@sealkeeper/schema 0.4.7 → 0.4.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api.d.ts +4411 -364
- package/dist/api.js +1753 -46
- package/dist/blocks.d.ts +21 -0
- package/dist/blocks.js +43 -0
- package/dist/cli-version.d.ts +7 -0
- package/dist/cli-version.js +89 -0
- package/dist/conformance.d.ts +4 -0
- package/dist/conformance.js +7 -0
- package/dist/credential.d.ts +763 -21
- package/dist/credential.js +253 -35
- package/dist/dimensions.d.ts +16 -3
- package/dist/dimensions.js +52 -2
- package/dist/fingerprint-conformance.d.ts +17 -0
- package/dist/fingerprint-conformance.js +83 -0
- package/dist/fingerprint.d.ts +133 -0
- package/dist/fingerprint.js +178 -0
- package/dist/game.d.ts +100 -0
- package/dist/game.js +190 -0
- package/dist/goal.d.ts +22 -1
- package/dist/goal.js +50 -7
- package/dist/handshake-conformance.d.ts +18 -0
- package/dist/handshake-conformance.js +276 -0
- package/dist/handshake.d.ts +66 -0
- package/dist/handshake.js +188 -0
- package/dist/index.d.ts +8 -0
- package/dist/index.js +8 -0
- package/dist/model-comparison.d.ts +70 -0
- package/dist/model-comparison.js +208 -0
- package/dist/model-name.d.ts +3 -0
- package/dist/model-name.js +48 -0
- package/dist/moderation.d.ts +2 -0
- package/dist/moderation.js +14 -8
- package/dist/policy.d.ts +1 -1
- package/dist/policy.js +1 -1
- package/dist/seal-conformance.js +263 -2
- package/dist/standing.d.ts +90 -0
- package/dist/standing.js +253 -13
- package/dist/task-templates.d.ts +47 -0
- package/dist/task-templates.js +506 -0
- package/dist/tasks.d.ts +97 -0
- package/dist/tasks.js +230 -0
- package/dist/template-conformance.d.ts +10 -0
- package/dist/template-conformance.js +298 -0
- package/dist/top-dimensions.d.ts +3 -3
- package/dist/top-dimensions.js +16 -6
- package/package.json +3 -3
package/dist/index.d.ts
CHANGED
|
@@ -3,18 +3,26 @@ export * from './agent-name.js';
|
|
|
3
3
|
export * from './api.js';
|
|
4
4
|
export * from './badge.js';
|
|
5
5
|
export * from './base64url.js';
|
|
6
|
+
export * from './blocks.js';
|
|
7
|
+
export * from './cli-version.js';
|
|
6
8
|
export * from './client-address.js';
|
|
7
9
|
export * from './credential.js';
|
|
8
10
|
export * from './dimensions.js';
|
|
9
11
|
export * from './envelope.js';
|
|
10
12
|
export * from './events.js';
|
|
13
|
+
export * from './fingerprint.js';
|
|
14
|
+
export * from './game.js';
|
|
11
15
|
export * from './goal.js';
|
|
16
|
+
export * from './handshake.js';
|
|
12
17
|
export * from './json-shape.js';
|
|
18
|
+
export * from './model-comparison.js';
|
|
19
|
+
export * from './model-name.js';
|
|
13
20
|
export * from './moderation.js';
|
|
14
21
|
export * from './operator-domains.js';
|
|
15
22
|
export * from './policy.js';
|
|
16
23
|
export * from './runtime.js';
|
|
17
24
|
export * from './seal-verify.js';
|
|
18
25
|
export * from './standing.js';
|
|
26
|
+
export * from './task-templates.js';
|
|
19
27
|
export * from './tasks.js';
|
|
20
28
|
export * from './top-dimensions.js';
|
package/dist/index.js
CHANGED
|
@@ -3,18 +3,26 @@ export * from './agent-name.js';
|
|
|
3
3
|
export * from './api.js';
|
|
4
4
|
export * from './badge.js';
|
|
5
5
|
export * from './base64url.js';
|
|
6
|
+
export * from './blocks.js';
|
|
7
|
+
export * from './cli-version.js';
|
|
6
8
|
export * from './client-address.js';
|
|
7
9
|
export * from './credential.js';
|
|
8
10
|
export * from './dimensions.js';
|
|
9
11
|
export * from './envelope.js';
|
|
10
12
|
export * from './events.js';
|
|
13
|
+
export * from './fingerprint.js';
|
|
14
|
+
export * from './game.js';
|
|
11
15
|
export * from './goal.js';
|
|
16
|
+
export * from './handshake.js';
|
|
12
17
|
export * from './json-shape.js';
|
|
18
|
+
export * from './model-comparison.js';
|
|
19
|
+
export * from './model-name.js';
|
|
13
20
|
export * from './moderation.js';
|
|
14
21
|
export * from './operator-domains.js';
|
|
15
22
|
export * from './policy.js';
|
|
16
23
|
export * from './runtime.js';
|
|
17
24
|
export * from './seal-verify.js';
|
|
18
25
|
export * from './standing.js';
|
|
26
|
+
export * from './task-templates.js';
|
|
19
27
|
export * from './tasks.js';
|
|
20
28
|
export * from './top-dimensions.js';
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { type TaskDifficulty } from './tasks.js';
|
|
3
|
+
export declare const MODEL_COMPARISON: {
|
|
4
|
+
readonly beforeDays: 30;
|
|
5
|
+
readonly afterDays: 30;
|
|
6
|
+
readonly minTasks: 20;
|
|
7
|
+
readonly rateDelta: 0.1;
|
|
8
|
+
readonly creditRatio: 0.15;
|
|
9
|
+
};
|
|
10
|
+
export declare const MODEL_VERDICTS: readonly ['better', 'worse', 'same', 'insufficient'];
|
|
11
|
+
export declare const ModelVerdict: z.ZodEnum<{
|
|
12
|
+
better: "better";
|
|
13
|
+
insufficient: "insufficient";
|
|
14
|
+
same: "same";
|
|
15
|
+
worse: "worse";
|
|
16
|
+
}>;
|
|
17
|
+
export type ModelVerdict = z.infer<typeof ModelVerdict>;
|
|
18
|
+
export type ModelComparisonSide = {
|
|
19
|
+
verified: number;
|
|
20
|
+
failed: number;
|
|
21
|
+
rejected: number;
|
|
22
|
+
credit: number | null;
|
|
23
|
+
};
|
|
24
|
+
export declare const completionRate: (s: ModelComparisonSide) => number;
|
|
25
|
+
export declare function verdictOf(before: ModelComparisonSide, after: ModelComparisonSide): ModelVerdict;
|
|
26
|
+
export declare const SUSPECTED_CHANGE: {
|
|
27
|
+
readonly recentDays: 7;
|
|
28
|
+
readonly beforeDays: 30;
|
|
29
|
+
readonly minTasks: 20;
|
|
30
|
+
readonly rateDelta: 0.05;
|
|
31
|
+
readonly minZ: 3;
|
|
32
|
+
readonly creditRatio: 0.15;
|
|
33
|
+
readonly lapseDays: 30;
|
|
34
|
+
};
|
|
35
|
+
export type SuspectedChangeRule = {
|
|
36
|
+
readonly minTasks: number;
|
|
37
|
+
readonly rateDelta: number;
|
|
38
|
+
readonly minZ: number;
|
|
39
|
+
readonly creditRatio: number;
|
|
40
|
+
};
|
|
41
|
+
export declare const SUSPECTED_STATES: readonly ['open', 'confirmed_declared', 'confirmed_network', 'dismissed', 'lapsed'];
|
|
42
|
+
export declare const SuspectedState: z.ZodEnum<{
|
|
43
|
+
confirmed_declared: "confirmed_declared";
|
|
44
|
+
confirmed_network: "confirmed_network";
|
|
45
|
+
dismissed: "dismissed";
|
|
46
|
+
lapsed: "lapsed";
|
|
47
|
+
open: "open";
|
|
48
|
+
}>;
|
|
49
|
+
export type SuspectedState = z.infer<typeof SuspectedState>;
|
|
50
|
+
export type SuspectedLevels = Partial<Record<TaskDifficulty, ModelComparisonSide>>;
|
|
51
|
+
export type SuspectedShift = {
|
|
52
|
+
shifted: boolean;
|
|
53
|
+
rate: number | null;
|
|
54
|
+
credit: number | null;
|
|
55
|
+
z: number | null;
|
|
56
|
+
};
|
|
57
|
+
export declare function shiftOf(before: SuspectedLevels, after: SuspectedLevels, rule?: SuspectedChangeRule): SuspectedShift;
|
|
58
|
+
export declare const MODEL_NETWORK: {
|
|
59
|
+
readonly minAgents: 3;
|
|
60
|
+
readonly minOperators: 2;
|
|
61
|
+
readonly perOperator: 3;
|
|
62
|
+
readonly models: 50;
|
|
63
|
+
readonly events: 20;
|
|
64
|
+
readonly changes: 20;
|
|
65
|
+
};
|
|
66
|
+
export type ModelVerdictCounts = {
|
|
67
|
+
better: number;
|
|
68
|
+
worse: number;
|
|
69
|
+
same: number;
|
|
70
|
+
};
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { TASK_DIFFICULTIES } from './tasks.js';
|
|
3
|
+
/*
|
|
4
|
+
* The model change comparison (VOU-551, UI-23, D-UI-6). When an agent
|
|
5
|
+
* declares another model (agent_model_changes, VOU-566), the nightly
|
|
6
|
+
* scoring run reads the agent's own tasks as claimant in the beforeDays
|
|
7
|
+
* before the change and in the days since, up to afterDays, per task
|
|
8
|
+
* category, and stores what it found beside the change. A report only.
|
|
9
|
+
* It moves no Trust Score, no category score, no stored credit and no
|
|
10
|
+
* level, and nothing reads it for standing.
|
|
11
|
+
*
|
|
12
|
+
* One side of one category holds four numbers. verified is the verified
|
|
13
|
+
* tasks that count for Trust, failed the claims ended at the failed submit
|
|
14
|
+
* cap or released after a failed submit, never a clean release (VOU-577),
|
|
15
|
+
* rejected the tasks whose poster answered the agent's success with
|
|
16
|
+
* failure, and credit the average stored base credit
|
|
17
|
+
* of the verified tasks that store one, null when none does.
|
|
18
|
+
*
|
|
19
|
+
* The rule, proposed by the VOU-551 pull request for Carl to agree before
|
|
20
|
+
* it ships, so the numbers live here and nowhere else. The completion rate
|
|
21
|
+
* is verified over verified plus failed plus rejected. Below minTasks
|
|
22
|
+
* verified tasks on either side there is no verdict, insufficient, since a
|
|
23
|
+
* rate over a handful of tasks swings on one of them. From there a side is
|
|
24
|
+
* better when the rate rose by at least rateDelta or the credit by at
|
|
25
|
+
* least creditRatio of the before side's, and neither fell by those
|
|
26
|
+
* amounts, worse the mirror, and same otherwise, so a rate that rose while
|
|
27
|
+
* the credit fell reads same. Ten points of completion is two tasks in
|
|
28
|
+
* twenty, so at minTasks one task alone never moves the rate to a
|
|
29
|
+
* verdict. Fifteen percent of credit is a starting value, to tune on real
|
|
30
|
+
* data like the Trust thresholds. Both are compared after rounding to six
|
|
31
|
+
* decimals, so 0.9 less 0.8 counts as the ten points it is.
|
|
32
|
+
*/
|
|
33
|
+
export const MODEL_COMPARISON = {
|
|
34
|
+
beforeDays: 30,
|
|
35
|
+
afterDays: 30,
|
|
36
|
+
minTasks: 20,
|
|
37
|
+
rateDelta: 0.1,
|
|
38
|
+
creditRatio: 0.15,
|
|
39
|
+
};
|
|
40
|
+
export const MODEL_VERDICTS = [
|
|
41
|
+
'better',
|
|
42
|
+
'worse',
|
|
43
|
+
'same',
|
|
44
|
+
'insufficient',
|
|
45
|
+
];
|
|
46
|
+
export const ModelVerdict = z.enum(MODEL_VERDICTS);
|
|
47
|
+
const round6 = (n) => Math.round(n * 1e6) / 1e6;
|
|
48
|
+
// verified over every attempt that ended, the rule above. A side with
|
|
49
|
+
// minTasks verified tasks always has attempts, so it never divides by 0.
|
|
50
|
+
export const completionRate = (s) => s.verified / (s.verified + s.failed + s.rejected);
|
|
51
|
+
// The verdict of one category, the rule above. Pure, so the run, a test
|
|
52
|
+
// and a later reader agree.
|
|
53
|
+
export function verdictOf(before, after) {
|
|
54
|
+
const { minTasks, rateDelta, creditRatio } = MODEL_COMPARISON;
|
|
55
|
+
if (before.verified < minTasks || after.verified < minTasks) {
|
|
56
|
+
return 'insufficient';
|
|
57
|
+
}
|
|
58
|
+
const rate = round6(completionRate(after) - completionRate(before));
|
|
59
|
+
const credit = before.credit === null || after.credit === null || before.credit <= 0
|
|
60
|
+
? 0
|
|
61
|
+
: round6(after.credit / before.credit - 1);
|
|
62
|
+
const up = rate >= rateDelta || credit >= creditRatio;
|
|
63
|
+
const down = rate <= -rateDelta || credit <= -creditRatio;
|
|
64
|
+
if (up && !down)
|
|
65
|
+
return 'better';
|
|
66
|
+
if (down && !up)
|
|
67
|
+
return 'worse';
|
|
68
|
+
return 'same';
|
|
69
|
+
}
|
|
70
|
+
/*
|
|
71
|
+
* Suspected changes (VOU-562, UI-30, D-UI-10, D-UI-11). When an agent's
|
|
72
|
+
* results in a category shift and nothing it reported explains it, the
|
|
73
|
+
* nightly scoring run keeps a suspected change of model for its operator,
|
|
74
|
+
* and for nobody else. A claim SealKeeper makes about the agent, so it is
|
|
75
|
+
* on no public route, in no feed item and in no SEAL, and it moves no
|
|
76
|
+
* Trust Score, no category score, no stored credit and no level.
|
|
77
|
+
*
|
|
78
|
+
* The windows, per UTC day of the run. after is the recentDays before the
|
|
79
|
+
* start of the day, before the beforeDays before that. Each side counts
|
|
80
|
+
* per difficulty level what the model comparison counts per side
|
|
81
|
+
* (ModelComparisonSide), the agent's own tasks as claimant.
|
|
82
|
+
*
|
|
83
|
+
* The shift, compared at the same difficulty. A level counts only with at
|
|
84
|
+
* least one attempt on each side, an attempt being a verified, failed or
|
|
85
|
+
* rejected task. Over the levels L that count, with w_d the attempts of
|
|
86
|
+
* level d on both sides and v_d its verified tasks on both sides,
|
|
87
|
+
*
|
|
88
|
+
* rate = sum over L of w_d * (rate_after_d - rate_before_d) / sum of w_d
|
|
89
|
+
* credit = sum over L' of v_d * (credit_after_d / credit_before_d - 1)
|
|
90
|
+
* / sum of v_d, L' the levels of L with a credit on both sides
|
|
91
|
+
* z = |rate| / sqrt(p * (1 - p) * (1 / n_before + 1 / n_after))
|
|
92
|
+
*
|
|
93
|
+
* rate_x_d being completionRate of level d on side x, n_x the attempts of
|
|
94
|
+
* L on side x and p the verified tasks of L on both sides over all their
|
|
95
|
+
* attempts. So a shift that comes only from harder or easier tasks reads
|
|
96
|
+
* as none, since every level is compared with itself. Below minTasks
|
|
97
|
+
* verified tasks of L on either side there is no shift, since tasks of a
|
|
98
|
+
* level with nothing to compare against say nothing about one. From there
|
|
99
|
+
* the results shifted when the rate moved by rateDelta or more with z at
|
|
100
|
+
* minZ or more, or the credit moved by creditRatio of the before side's or
|
|
101
|
+
* more, either way, compared after rounding to six decimals.
|
|
102
|
+
*
|
|
103
|
+
* The numbers wait for Carl's agreement at the VOU-562 pull request. The
|
|
104
|
+
* issue proposed 5 points of completion or 15 percent of credit alone.
|
|
105
|
+
* Over seeded steady history, one agent with the same chance per level
|
|
106
|
+
* every day, 5 points alone fired on about half the nights at 5 tasks a
|
|
107
|
+
* day and on a third at 10, since 5 points is one task in twenty. minZ is
|
|
108
|
+
* the tuning the seeded history asked for, a shift of 3 standard errors
|
|
109
|
+
* of the difference, which fired on under 0.4 percent of the nights at 3
|
|
110
|
+
* to 20 tasks a day, and on none of the 24 nights of the 60 days the API
|
|
111
|
+
* test seeds, while a real drop of 20 points is caught on most nights at
|
|
112
|
+
* 10 tasks a day or more. The credit side had no variance in the seeded
|
|
113
|
+
* history, where every task of a level earns the same.
|
|
114
|
+
*
|
|
115
|
+
* A suspected change stays open lapseDays at most. A declared change of
|
|
116
|
+
* model or a network event for the agent's model confirms it sooner, and
|
|
117
|
+
* the operator can dismiss it.
|
|
118
|
+
*/
|
|
119
|
+
export const SUSPECTED_CHANGE = {
|
|
120
|
+
recentDays: 7,
|
|
121
|
+
beforeDays: 30,
|
|
122
|
+
minTasks: 20,
|
|
123
|
+
rateDelta: 0.05,
|
|
124
|
+
minZ: 3,
|
|
125
|
+
creditRatio: 0.15,
|
|
126
|
+
lapseDays: 30,
|
|
127
|
+
};
|
|
128
|
+
// A suspected change's life. open until one of the others, each final.
|
|
129
|
+
export const SUSPECTED_STATES = [
|
|
130
|
+
'open',
|
|
131
|
+
'confirmed_declared',
|
|
132
|
+
'confirmed_network',
|
|
133
|
+
'dismissed',
|
|
134
|
+
'lapsed',
|
|
135
|
+
];
|
|
136
|
+
export const SuspectedState = z.enum(SUSPECTED_STATES);
|
|
137
|
+
// The shift of one category, the rule above. Pure, so the run, a test and
|
|
138
|
+
// a later reader agree. rule is SUSPECTED_CHANGE unless a test tunes it.
|
|
139
|
+
export function shiftOf(before, after, rule = SUSPECTED_CHANGE) {
|
|
140
|
+
const attempts = (s) => s.verified + s.failed + s.rejected;
|
|
141
|
+
let weight = 0;
|
|
142
|
+
let rateSum = 0;
|
|
143
|
+
let creditWeight = 0;
|
|
144
|
+
let creditSum = 0;
|
|
145
|
+
let verifiedBefore = 0;
|
|
146
|
+
let verifiedAfter = 0;
|
|
147
|
+
let attemptsBefore = 0;
|
|
148
|
+
let attemptsAfter = 0;
|
|
149
|
+
for (const d of TASK_DIFFICULTIES) {
|
|
150
|
+
const b = before[d];
|
|
151
|
+
const a = after[d];
|
|
152
|
+
if (!b || !a || attempts(b) === 0 || attempts(a) === 0)
|
|
153
|
+
continue;
|
|
154
|
+
const w = attempts(b) + attempts(a);
|
|
155
|
+
weight += w;
|
|
156
|
+
rateSum += w * (completionRate(a) - completionRate(b));
|
|
157
|
+
verifiedBefore += b.verified;
|
|
158
|
+
verifiedAfter += a.verified;
|
|
159
|
+
attemptsBefore += attempts(b);
|
|
160
|
+
attemptsAfter += attempts(a);
|
|
161
|
+
if (b.credit !== null && a.credit !== null && b.credit > 0) {
|
|
162
|
+
const v = b.verified + a.verified;
|
|
163
|
+
creditWeight += v;
|
|
164
|
+
creditSum += v * (a.credit / b.credit - 1);
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
if (verifiedBefore < rule.minTasks || verifiedAfter < rule.minTasks) {
|
|
168
|
+
return { shifted: false, rate: null, credit: null, z: null };
|
|
169
|
+
}
|
|
170
|
+
const rate = round6(rateSum / weight);
|
|
171
|
+
const credit = creditWeight > 0 ? round6(creditSum / creditWeight) : null;
|
|
172
|
+
const p = (verifiedBefore + verifiedAfter) / (attemptsBefore + attemptsAfter);
|
|
173
|
+
const se = Math.sqrt(p * (1 - p) * (1 / attemptsBefore + 1 / attemptsAfter));
|
|
174
|
+
const z = se > 0 ? round6(Math.abs(rate) / se) : 0;
|
|
175
|
+
const shifted = (Math.abs(rate) >= rule.rateDelta && z >= rule.minZ) ||
|
|
176
|
+
(credit !== null && Math.abs(credit) >= rule.creditRatio);
|
|
177
|
+
return { shifted, rate, credit, z };
|
|
178
|
+
}
|
|
179
|
+
/*
|
|
180
|
+
* Network events (VOU-551). Where minAgents or more agents, of at least
|
|
181
|
+
* minOperators operators, recorded the same change, the same old modelKey
|
|
182
|
+
* to the same new one, in the same UTC week from Monday (trustWeekOf), the
|
|
183
|
+
* nightly run keeps one model_network_events row with the count of those
|
|
184
|
+
* agents' verdicts per category. An agent with no declared name on either
|
|
185
|
+
* side is in none, and neither is a change of the fingerprint's model part
|
|
186
|
+
* alone, whose two keys are the same. Two operators, so one operator's own
|
|
187
|
+
* agents never make an event alone.
|
|
188
|
+
*
|
|
189
|
+
* perOperator (VOU-578). An event counts at most this many agents of any
|
|
190
|
+
* one operator, in agents and in its verdicts, the ones with the lowest
|
|
191
|
+
* agent ids, the rest left out. An operator may run 100 agents, so without
|
|
192
|
+
* it one operator with 99 agents and one colluding agent would make an
|
|
193
|
+
* event and supply almost every verdict. 3 is minAgents, so one
|
|
194
|
+
* operator's agents never count for more than the size an event needs,
|
|
195
|
+
* and in an event of 2 operators the other's agents are at least a
|
|
196
|
+
* quarter of the verdicts.
|
|
197
|
+
*
|
|
198
|
+
* The caps of the three reads, GET /v1/models (models and events) and GET
|
|
199
|
+
* /v1/agents/:id/model-changes (changes).
|
|
200
|
+
*/
|
|
201
|
+
export const MODEL_NETWORK = {
|
|
202
|
+
minAgents: 3,
|
|
203
|
+
minOperators: 2,
|
|
204
|
+
perOperator: 3,
|
|
205
|
+
models: 50,
|
|
206
|
+
events: 20,
|
|
207
|
+
changes: 20,
|
|
208
|
+
};
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import { EventPayload } from './events.js';
|
|
2
|
+
import { foldLetters, OWN_NAMES } from './moderation.js';
|
|
3
|
+
/*
|
|
4
|
+
* The name of the model an agent runs, as the agent declares it (VOU-566).
|
|
5
|
+
* It travels as text beside the fingerprint on a sync, see
|
|
6
|
+
* FingerprintDeclaration, so SealKeeper can show and compare the models
|
|
7
|
+
* agents run across Claude, GPT, Gemini and the rest. It is what the agent
|
|
8
|
+
* says about itself, inside a signed report. Nothing reads it as proof, and
|
|
9
|
+
* it earns no Trust. The model part of the fingerprint stays a hash.
|
|
10
|
+
*/
|
|
11
|
+
const OWN_KEYS = OWN_NAMES.map(foldLetters);
|
|
12
|
+
// The usage event's model name, the same characters and the same 64 cap, so
|
|
13
|
+
// a name an adapter already sends on usage is a name it can declare. A name
|
|
14
|
+
// must also leave a key, so one that ends in a slash, or is only a date, is
|
|
15
|
+
// refused. It shows on public pages, so the moderation rules names have
|
|
16
|
+
// apply as far as they fit a model name. It may not hold SealKeeper's own
|
|
17
|
+
// names anywhere in its letters and digits as foldLetters folds them, so
|
|
18
|
+
// sealkeeper-verified, official-vouched and 5ea1keeper are refused, nor in
|
|
19
|
+
// its key's, which GET /v1/models shows (VOU-551). foldLetters leaves the
|
|
20
|
+
// rn and vv pairs alone, since normaliseName folds v/vouched to wouched
|
|
21
|
+
// while its key is vouched, and sealkeepernova to sealkeepemova. The
|
|
22
|
+
// other protected names stay allowed, since claude-opus-4-5 and gpt-4.1
|
|
23
|
+
// name their makers' models, and the characters are Latin alone, so no
|
|
24
|
+
// lookalike of another script gets in.
|
|
25
|
+
export const ModelName = EventPayload.usage.shape.model
|
|
26
|
+
.unwrap()
|
|
27
|
+
.refine((name) => modelKey(name) !== '', 'A model name needs a name after its last slash')
|
|
28
|
+
.refine((name) => ![foldLetters(name), foldLetters(modelKey(name))].some((form) => OWN_KEYS.some((own) => form.includes(own))), 'A model name may not name SealKeeper');
|
|
29
|
+
// When two names are the same model. Lower case, everything up to the last
|
|
30
|
+
// slash dropped, so a provider or router prefix goes, a trailing date of 8
|
|
31
|
+
// digits dropped with its dash, and dots made dashes, so claude-sonnet-4.5
|
|
32
|
+
// and claude-sonnet-4-5 meet. A date written with dashes, as in
|
|
33
|
+
// gpt-4.1-2025-04-14, stays, and so does anything else the rule does not
|
|
34
|
+
// name. The one function that decides it, so the API, the CLI and the web
|
|
35
|
+
// agree.
|
|
36
|
+
export function modelKey(name) {
|
|
37
|
+
const lower = name.toLowerCase();
|
|
38
|
+
return lower
|
|
39
|
+
.slice(lower.lastIndexOf('/') + 1)
|
|
40
|
+
.replace(/-\d{8}$/, '')
|
|
41
|
+
.replaceAll('.', '-');
|
|
42
|
+
}
|
|
43
|
+
// The most model changes one agent records in any 24 hours. A real change
|
|
44
|
+
// of model is rare, and an operator who tries a model and goes back makes
|
|
45
|
+
// two in a day, so 3 leaves one to spare. Past it the name is still kept
|
|
46
|
+
// as the agent's current one and no change is recorded, so an agent cannot
|
|
47
|
+
// grow the table by flipping its name.
|
|
48
|
+
export const MODEL_CHANGES_PER_DAY = 3;
|
package/dist/moderation.d.ts
CHANGED
|
@@ -8,7 +8,9 @@ export type OperatorSlug = z.infer<typeof OperatorSlug>;
|
|
|
8
8
|
export declare const OperatorDisplayName: z.ZodString;
|
|
9
9
|
export type OperatorDisplayName = z.infer<typeof OperatorDisplayName>;
|
|
10
10
|
export declare function normaliseName(text: string): string;
|
|
11
|
+
export declare function foldLetters(text: string): string;
|
|
11
12
|
export declare const displayNameKey: typeof normaliseName;
|
|
13
|
+
export declare const OWN_NAMES: readonly string[];
|
|
12
14
|
export declare const PROTECTED_NAMES: readonly string[];
|
|
13
15
|
export declare const PROTECTED_SUFFIXES: readonly string[];
|
|
14
16
|
export declare const PROTECTED_PREFIXES: readonly string[];
|
package/dist/moderation.js
CHANGED
|
@@ -169,19 +169,26 @@ const NOT_LETTER_OR_DIGIT = /[^\p{L}\p{N}]/gu;
|
|
|
169
169
|
// Letters of scripts with no Latin lookalike are kept, so a display name
|
|
170
170
|
// in another script still has a key of its own.
|
|
171
171
|
export function normaliseName(text) {
|
|
172
|
-
|
|
172
|
+
return foldLetters(text).replace(/rn/g, 'm').replace(/vv/g, 'w');
|
|
173
|
+
}
|
|
174
|
+
// normaliseName before the letter pairs are folded. A word the text holds
|
|
175
|
+
// whole stays whole here, where a pair can take a letter from it in
|
|
176
|
+
// normaliseName, so vvouched keeps vouched and sealkeepernova keeps
|
|
177
|
+
// sealkeeper (VOU-551).
|
|
178
|
+
export function foldLetters(text) {
|
|
179
|
+
return Array.from(text.normalize('NFKC').normalize('NFKD').replace(MARKS, ''), foldLetter)
|
|
173
180
|
.join('')
|
|
174
|
-
.replace(I_AS_L, 'l')
|
|
175
|
-
|
|
176
|
-
.replace(NOT_LETTER_OR_DIGIT, '')
|
|
177
|
-
.replace(/rn/g, 'm')
|
|
178
|
-
.replace(/vv/g, 'w');
|
|
181
|
+
.replace(I_AS_L, 'l')
|
|
182
|
+
.replace(NOT_LETTER_OR_DIGIT, '');
|
|
179
183
|
}
|
|
180
184
|
// The display_name_key column. Unique across operators, so two display
|
|
181
185
|
// names that read the same cannot both be held.
|
|
182
186
|
export const displayNameKey = normaliseName;
|
|
183
187
|
// ---------------------------------------------------------------------------
|
|
184
188
|
// Protected names
|
|
189
|
+
// SealKeeper's own names, the first of PROTECTED_NAMES. A model name an
|
|
190
|
+
// agent declares may not hold one anywhere, see ModelName.
|
|
191
|
+
export const OWN_NAMES = ['SealKeeper', 'Vouched'];
|
|
185
192
|
// The top AI and technology companies and products. A slug or display name
|
|
186
193
|
// that normalises to one of these, with or without a common suffix, is
|
|
187
194
|
// refused. Written as the name reads. The list protects the distinctive
|
|
@@ -191,8 +198,7 @@ export const displayNameKey = normaliseName;
|
|
|
191
198
|
// would otherwise lose its slug. Claude is the one exception.
|
|
192
199
|
export const PROTECTED_NAMES = [
|
|
193
200
|
// SealKeeper itself
|
|
194
|
-
|
|
195
|
-
'Vouched',
|
|
201
|
+
...OWN_NAMES,
|
|
196
202
|
// AI labs and model makers
|
|
197
203
|
'Anthropic',
|
|
198
204
|
'Claude',
|
package/dist/policy.d.ts
CHANGED
package/dist/policy.js
CHANGED