clearotron 0.4.0-beta.2 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +5 -5
- package/build-info.json +2 -2
- package/driver/CHANGELOG.md +90 -0
- package/driver/ask-ledger.mjs +2 -2
- package/driver/coverage-form.mjs +6 -0
- package/driver/coverage-ledger.mjs +14 -2
- package/driver/driver.config.mjs +24 -7
- package/driver/engine/mcp/band-server.mjs +3 -3
- package/driver/engine/mcp/stdio-server.mjs +12 -1
- package/driver/gateway.mjs +27 -3
- package/driver/package.json +1 -1
- package/driver/pipeline-knockout.mjs +41 -10
- package/driver/pipeline.mjs +104 -12
- package/driver/progress.mjs +6 -1
- package/driver/provider-usage.mjs +16 -0
- package/driver/publish/index.mjs +11 -5
- package/driver/publish/knockout.mjs +16 -3
- package/driver/publish/render-knockout.mjs +38 -11
- package/driver/record-carry.mjs +17 -1
- package/driver/reference-strip-signatures.mjs +11 -1
- package/driver/register-plan.mjs +7 -3
- package/driver/register-served.mjs +91 -0
- package/driver/remedy-accounting.mjs +38 -9
- package/driver/reviewer-open-points.mjs +20 -23
- package/driver/score-redaction.mjs +416 -20
- package/driver/screen-gate.mjs +3 -3
- package/driver/skills/clearance-register/SKILL.md +0 -8
- package/driver/skills/clearance-register/digest.md +1 -1
- package/driver/skills/clearance-register/providers/corsearch.md +1 -1
- package/driver/skills/clearance-register/register-recipes.md +3 -9
- package/driver/skills/clearance-search/SKILL.md +2 -2
- package/driver/skills/clearance-search/phase2-execution.md +1 -1
- package/driver/skills/knockout-assess/SKILL.md +2 -0
- package/driver/stages-knockout.mjs +22 -0
- package/driver/suite-census.json +152 -26
- package/driver/unit-inventory.mjs +38 -43
- package/mcp-server/CHANGELOG.md +8 -0
- package/mcp-server/package.json +1 -1
- package/node_modules/brace-expansion/index.js +78 -22
- package/node_modules/brace-expansion/package.json +1 -1
- package/node_modules/readdir-glob/node_modules/brace-expansion/index.js +78 -22
- package/node_modules/readdir-glob/node_modules/brace-expansion/package.json +1 -1
- package/package.json +1 -1
- package/portal-ui/package.json +2 -2
- package/providers/_shared/term-shape.mjs +1 -1
- package/providers/jx/src/core.js +0 -1
- package/providers/jx/src/judge.js +0 -1
- package/providers/jx/src/nativeread.js +0 -1
- package/providers/oauth-mcp-bridge/CHANGELOG.md +8 -0
- package/providers/oauth-mcp-bridge/package.json +1 -1
- package/providers/signa/src/core.js +125 -17
- package/scripts/e2e-first-time.mjs +12 -4
- package/scripts/e2e-scenario-ops.mjs +25 -2
- package/scripts/e2e.mjs +246 -24
- package/scripts/mint-reference-strip-backlog.mjs +36 -2
- package/scripts/score.mjs +102 -29
- package/shared/identifier-scan.mjs +31 -1
|
@@ -41,11 +41,43 @@ const PROSE_KEYS = new Set([
|
|
|
41
41
|
// Engine prose about a matter, classified from reading its site rather than from a warning: the
|
|
42
42
|
// reason a finding was ruled out is written about that finding and can name the party it concerns.
|
|
43
43
|
"ruled_out_reason",
|
|
44
|
+
// The findings document's own prose, surfaced the first time the run was walked rather than only the
|
|
45
|
+
// reference. Each of these is the engine reasoning about a party in sentences.
|
|
46
|
+
"net", "practical_position", "legal_position", "off_field_ground", "reason", "bears_on", "read",
|
|
47
|
+
"condition", "basis", "quality",
|
|
48
|
+
// THE SEVEN THE WARNING HAD BEEN NAMING, each read at the site that WRITES it rather than classified
|
|
49
|
+
// from its key name — which is what the warning asks a reader to do, and what it now says.
|
|
50
|
+
//
|
|
51
|
+
// Four were measured against the real-name list, one key at a time: `email_line` hit 104 times across
|
|
52
|
+
// 12 values, `transcription_note` 3 of 4, `scoring_caveat` 1 of 1, `what` 1 of 2. A line written to be
|
|
53
|
+
// sent to somebody about their matter names them, which is what `email_line` is for.
|
|
54
|
+
"email_line", "transcription_note", "scoring_caveat", "what",
|
|
55
|
+
// Three read clean against that list TODAY and are protected on what their producer is for, because
|
|
56
|
+
// clean today is not safe. `owner_description` sits at `common_law[].owner_description` and its job is
|
|
57
|
+
// to describe a party: its values are generic in the references we hold and the next one written is
|
|
58
|
+
// the one that names somebody. `channels_note` is 97 words of the lawyer's prose about a matter.
|
|
59
|
+
"owner_description", "channels_note",
|
|
60
|
+
// `sheet` is the one that argues for reading the producer rather than the name. Two files write it
|
|
61
|
+
// and they mean different things: `presence-reconciliation.mjs` writes a two-word workbook label,
|
|
62
|
+
// plainly safe, and a gold writes ten values, all distinct, up to 165 characters — tabulated register
|
|
63
|
+
// rows carrying application and registration numbers. Only the gold's reaches this function today
|
|
64
|
+
// (zero `sheet` keys in a real run's findings, checked on R19 `a46abdad`), so prose costs nothing and
|
|
65
|
+
// is right for what arrives. See the note below on why that is luck rather than design.
|
|
66
|
+
"sheet",
|
|
44
67
|
]);
|
|
45
68
|
|
|
46
69
|
/** Fields that carry a name: a mark, a proprietor, a subject, a form of a mark. */
|
|
47
70
|
const NAME_KEYS = new Set([
|
|
48
71
|
"mark", "owner", "subject", "name", "matched", "entry", "noise", "form", "use_form",
|
|
72
|
+
// `refused` READS LIKE A BOOLEAN AND HOLDS A MARK. `reference-score.mjs` builds it as
|
|
73
|
+
// `refused: refused.mark`, so a key whose name suggests a flag carries a name. Classified from its
|
|
74
|
+
// producer rather than from the word — I had it in the safe set for one draft, which would have
|
|
75
|
+
// shipped a leak inside the change that closes one.
|
|
76
|
+
"refused",
|
|
77
|
+
// `item` is a gap's identifier, `g.item ?? g.slice`, and I could not establish from its producer that
|
|
78
|
+
// it never carries matter text. Protected rather than assumed safe: tokenising an identifier costs a
|
|
79
|
+
// little readability, and the other error costs a name on the page.
|
|
80
|
+
"item",
|
|
49
81
|
// A SEARCH TERM IS A SPELLING OF A MARK. `term` carries the string a register question was actually
|
|
50
82
|
// asked with — the same class of thing as `close_variations`, which was already protected, arriving
|
|
51
83
|
// by a different route. Read at its site (the per-territory query roll) rather than inferred: it is
|
|
@@ -66,6 +98,12 @@ const NAME_LIST_KEYS = new Set(["covers_marks", "close_variations", "variations"
|
|
|
66
98
|
* steady state, so a reference that gains a field says so the first time it is scored.
|
|
67
99
|
*/
|
|
68
100
|
const SAFE_KEYS = new Set([
|
|
101
|
+
// NOT A REFERENCE KEY AT ALL, read at its producer by the lane that owns this file: `coverage` appears
|
|
102
|
+
// zero times in every gold. `reference-score.mjs` sets it from the scorer's own verdict word and the
|
|
103
|
+
// scorer prints it as a state and a `why` — the harness's sentence about what it could measure, not a
|
|
104
|
+
// sentence about anybody. Classified here because the warning below named it and a reader would
|
|
105
|
+
// otherwise go looking in a gold file for a key the scorer invented.
|
|
106
|
+
"coverage",
|
|
69
107
|
"scenario", "schema_version", "schema", "framework", "id", "title", "door", "lane", "state", "why",
|
|
70
108
|
"lawyer_band", "legal_level", "dispute_type", "lawyer_risk", "band", "rating", "ratingQualifier",
|
|
71
109
|
"channel", "channels", "jurisdictions", "classes", "territories", "country", "rule", "bucket",
|
|
@@ -73,6 +111,14 @@ const SAFE_KEYS = new Set([
|
|
|
73
111
|
// Label-shaped fields the real references carry, confirmed one by one rather than assumed: a grade,
|
|
74
112
|
// an expected-value word, and the territory a reference entry sits in.
|
|
75
113
|
"grade", "expected", "jurisdiction",
|
|
114
|
+
// Band and classification labels the findings document carries. Words from a fixed ladder, not names.
|
|
115
|
+
"Dispute Type", "Legal Risk Level", "category", "source_type", "resolved_link",
|
|
116
|
+
// The refusal record's own fields, surfaced once the run was walked: `refusedRule` and
|
|
117
|
+
// `refusedEvidence` name the RULE that fired and the evidence class it wanted, both from fixed
|
|
118
|
+
// vocabularies, and `refused` is its boolean. Read at their site rather than inferred from the names.
|
|
119
|
+
// `refusedRule` names the RULE that fired and `refusedEvidence` the evidence class it wanted, both
|
|
120
|
+
// fixed vocabularies. `refused` is NOT here: see NAME_KEYS.
|
|
121
|
+
"refusedRule", "refusedEvidence",
|
|
76
122
|
// THE TWENTY-ONE THE WARNING NAMED ON THE WITHHELD CORPUS, each classified by what its site holds
|
|
77
123
|
// rather than by what its name suggests — two of them would have been got wrong by the name alone.
|
|
78
124
|
//
|
|
@@ -113,6 +159,14 @@ export function protectedStrings(root) {
|
|
|
113
159
|
if (Array.isArray(node)) { for (const v of node) walk(v); return; }
|
|
114
160
|
for (const [k, v] of Object.entries(node)) {
|
|
115
161
|
if (k.startsWith("_")) continue; // `_why` keys document the file, not the matter
|
|
162
|
+
// CLASSIFIED BY THE LEAF KEY NAME, AND THAT IS A KNOWN LIMIT rather than an oversight. Two files
|
|
163
|
+
// can write the same key and mean different things: `sheet` is a two-word workbook label in
|
|
164
|
+
// `presence-reconciliation.mjs` and a tabulated register row in a gold. Only one of them reaches
|
|
165
|
+
// this function today, so the sets are right about what arrives — but they are right by luck, and
|
|
166
|
+
// the day the other one arrives whichever answer is held is wrong for it. The fix, when it is
|
|
167
|
+
// needed, is to decide on the FULL PATH before the leaf, the way the scenario-label guard does for
|
|
168
|
+
// `cost.note` against `scoring.note`. Not done here because nothing yet needs it, and a path-aware
|
|
169
|
+
// classifier built against a collision that has not happened would be guessing at its shape.
|
|
116
170
|
if (isStr(v) && NAME_KEYS.has(k)) names.add(v.trim());
|
|
117
171
|
else if (isStr(v) && PROSE_KEYS.has(k)) prose.add(v.trim());
|
|
118
172
|
else if (Array.isArray(v) && NAME_LIST_KEYS.has(k)) { for (const e of v) if (isStr(e)) names.add(e.trim()); }
|
|
@@ -122,21 +176,138 @@ export function protectedStrings(root) {
|
|
|
122
176
|
}
|
|
123
177
|
};
|
|
124
178
|
walk(root);
|
|
125
|
-
// A
|
|
126
|
-
//
|
|
179
|
+
// A MULTI-WORD NAME IS ALSO PROTECTED BY ITS DISTINCTIVE WORD, because prose shortens it.
|
|
180
|
+
//
|
|
181
|
+
// Measured on R18 `f5764a2f`: a two-word proprietor was collected in full and still printed in clear,
|
|
182
|
+
// because the sentence that named it used the first word with a possessive and nothing else. Matching
|
|
183
|
+
// only the whole string protects the form in the record and misses the form a reader actually meets —
|
|
184
|
+
// and a redaction that covers the tidy case and not the prose one is worse than none, because the page
|
|
185
|
+
// looks redacted.
|
|
186
|
+
//
|
|
187
|
+
// THE FLOOR IS FIVE CHARACTERS AND THE LEGAL FORMS ARE EXCLUDED. A short word, or "Holdings", or "Ltd",
|
|
188
|
+
// is a word of ordinary English before it is anybody's name, and swapping those everywhere would shred
|
|
189
|
+
// the surrounding text while protecting nobody. A distinctive five-letter-plus word is the part a
|
|
190
|
+
// reader would recognise the party from, which is the thing being withheld.
|
|
191
|
+
const LEGAL_FORM = new Set(["inc", "llc", "ltd", "limited", "gmbh", "corp", "corporation", "company",
|
|
192
|
+
"holdings", "group", "plc", "sarl", "bv", "nv", "ag", "sa", "spa", "pty", "kk", "co", "and", "the", "of"]);
|
|
193
|
+
// KEPT APART FROM THE GIVEN NAMES, because the two are not equally safe to
|
|
194
|
+
// apply. A given name is a party's name and appears nowhere else by construction. A DERIVED word is an
|
|
195
|
+
// ordinary word that happens to sit inside one, so it collides with the scorer's own scaffolding: a
|
|
196
|
+
// party called "Depth Charge" turns `per-territory depth` into `per-territory «name 2»`, and a reader
|
|
197
|
+
// cannot tell a redaction from a word. Both layers still apply to everything the run prints; only a
|
|
198
|
+
// line the SCORER ITSELF authored drops the derived one, through `authoredRedactor` below.
|
|
199
|
+
const derived = new Set();
|
|
200
|
+
for (const n of [...names]) {
|
|
201
|
+
const parts = n.trim().split(/\s+/);
|
|
202
|
+
if (parts.length < 2) continue;
|
|
203
|
+
for (const w of parts) {
|
|
204
|
+
// FOLDED BEFORE THE FLOOR AND THE LEGAL-FORM LOOKUP, and the folded form is what is stored.
|
|
205
|
+
// Both tests are wrong on the unfolded word, in opposite directions. `\uFF2C\uFF34\uFF24`
|
|
206
|
+
// lowercases to `\uFF4C\uFF54\uFF44`, which is not in LEGAL_FORM, so the legal form became a
|
|
207
|
+
// protected word — and once the text is folded too, that word matches every "Ltd" the run prints,
|
|
208
|
+
// which is the shredding LEGAL_FORM exists to prevent. The floor has the mirror fault: the "fi"
|
|
209
|
+
// ligature is four characters raw and five folded, so a distinctive word was excluded for being
|
|
210
|
+
// short when it is not. Deriving from the folded form makes both decisions right by construction.
|
|
211
|
+
const bare = foldForMatching(w).folded.replace(/[^\p{L}\p{N}]/gu, "");
|
|
212
|
+
if (bare.length >= 5 && !LEGAL_FORM.has(bare.toLowerCase())) { names.add(bare); derived.add(bare); }
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
// THE FLOOR STAYS AND THE CASE IT MISSES IS NAMED. A fleet decision on an internal page rather than
|
|
216
|
+
// this file's preference: keep the floor, and name the case it does not cover.
|
|
217
|
+
//
|
|
218
|
+
// WHAT IT BUYS: a two-character token is cheap to collide with, and every standalone occurrence of one
|
|
219
|
+
// would go — including words of this harness's own vocabulary. Lowering the floor is a real option and
|
|
220
|
+
// its cost has not been measured, so it is left open on evidence rather than closed on an argument.
|
|
221
|
+
//
|
|
222
|
+
// WHAT IT COSTS, SAID PLAINLY BECAUSE THE OLD COMMENT DENIED IT. This used to claim a short name
|
|
223
|
+
// protects "nothing a reader could identify anyone from". That is false, and it was measured: on one
|
|
224
|
+
// scenario the two-character SUBJECT MARK of the matter is deleted here and prints in clear, beside a
|
|
225
|
+
// name that was correctly withheld. The scenario is named after it and its run directory carries it.
|
|
226
|
+
// `ops/real-names/scan.mjs` holds that same string as a list entry and refuses a page that shows it —
|
|
227
|
+
// so two instruments on this box disagree about it, and this is the one that lets it through.
|
|
228
|
+
//
|
|
229
|
+
// The other clause was stale rather than false: "a substring of ordinary words" described a matcher
|
|
230
|
+
// without boundaries. The matcher has them now — no letter or digit may sit immediately before a
|
|
231
|
+
// match, and only a plural or possessive after — so a short name matches standing alone, not inside a
|
|
232
|
+
// word. That narrows the shredding this floor was written to prevent, which is why lowering it is
|
|
233
|
+
// worth measuring rather than dismissing.
|
|
234
|
+
//
|
|
235
|
+
// THE SCAN IS WHAT CATCHES THE MISS, and a reader of this file should know that rather than infer it.
|
|
127
236
|
for (const n of [...names]) if (n.length < 3) names.delete(n);
|
|
128
|
-
return { names, prose, unclassified: [...unclassified].sort() };
|
|
237
|
+
return { names, prose, derived, unclassified: [...unclassified].sort() };
|
|
129
238
|
}
|
|
130
239
|
|
|
131
|
-
/**
|
|
240
|
+
/**
|
|
241
|
+
* The line a caller prints when `unclassified` is not empty. Names keys, never values.
|
|
242
|
+
*
|
|
243
|
+
* IT DOES NOT SAY "THE REFERENCE", and that wording was a real misdirection rather than a looseness.
|
|
244
|
+
* `protectedStrings` walks the reference, the scored buckets AND the run, so a key it names may come from
|
|
245
|
+
* any of the three — and the first key it named after the run was added, `coverage`, is the scorer's own
|
|
246
|
+
* field, absent from every gold. Anyone acting on the old sentence went looking in a gold file for a key
|
|
247
|
+
* the scorer invented. Saying it does not know which source is worse than naming it and better than
|
|
248
|
+
* naming the wrong one; carrying the source per key is a further change and is not this one.
|
|
249
|
+
*/
|
|
132
250
|
export const unclassifiedNotice = (keys) =>
|
|
133
|
-
`*** THE REFERENCE CARRIES ${keys.length} KEY(S) THIS REDACTION DOES NOT
|
|
251
|
+
`*** THE REFERENCE, THE SCORED BUCKETS OR THE RUN CARRIES ${keys.length} KEY(S) THIS REDACTION DOES NOT `
|
|
252
|
+
+ `CLASSIFY, and this line does not know which: ${keys.join(", ")}. `
|
|
134
253
|
+ `Their text was NOT withheld and may name somebody. Classify each in driver/score-redaction.mjs as a `
|
|
135
|
-
+ `name, as prose, or as safe
|
|
136
|
-
+ `
|
|
254
|
+
+ `name, as prose, or as safe, reading the site that WRITES it rather than its key name — and protect `
|
|
255
|
+
+ `anything you cannot establish is free of matter text.`;
|
|
137
256
|
|
|
138
257
|
const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
139
258
|
|
|
259
|
+
/**
|
|
260
|
+
* A name is matched however it is rendered, and the token is spliced back into the ORIGINAL text.
|
|
261
|
+
*
|
|
262
|
+
* WHY. A score taken without `--names` printed a mark and its proprietor in clear, in the same sentence
|
|
263
|
+
* as three that redacted correctly, because that one renders at full width: the protected set matched
|
|
264
|
+
* only the exact code points the reference happened to carry. Full-width Latin is
|
|
265
|
+
* ordinary in East Asian filings, so the failure concentrated in exactly the matters read in more than
|
|
266
|
+
* one script. A register can also hand back full-width digits and punctuation, half-width katakana, and
|
|
267
|
+
* an accented name either composed or decomposed. All of those are the same class, and one compatibility
|
|
268
|
+
* fold answers all of them.
|
|
269
|
+
*
|
|
270
|
+
* WHY NOT FOLD THE WHOLE STRING. The fold has to be reversible enough to put the token back where the
|
|
271
|
+
* name stood in the text the reader will see. Normalising the page and printing the normalised copy would
|
|
272
|
+
* silently rewrite text that names nobody. So the fold is built as a PROJECTION: a folded string to match
|
|
273
|
+
* in, and two arrays mapping each folded position back to the original range it came from.
|
|
274
|
+
*
|
|
275
|
+
* WHY BY CLUSTER, NOT BY CHARACTER. Folding one code point at a time keeps the mapping simple but never
|
|
276
|
+
* recombines: `e` followed by a combining acute stays two characters and never matches the composed name.
|
|
277
|
+
* So a character is grouped with the marks that follow it and the group is folded whole. The group has to
|
|
278
|
+
* include the half-width katakana voiced marks, which are modifier LETTERS and not marks — without them
|
|
279
|
+
* the fold yields U+30C8 U+3099 where the reference carries U+30C9, which is the same glyph on screen and
|
|
280
|
+
* not a match. Measured, 2026-09-27; the character classes are why reading the output was not enough.
|
|
281
|
+
*
|
|
282
|
+
* EVERY FAILURE MODE HERE IS OVER-REDACTION. The spliced range runs from the start of the first cluster
|
|
283
|
+
* the match touched to the end of the last, so it always contains the matched text and never less. A
|
|
284
|
+
* match that lands part-way into one cluster removes the whole cluster.
|
|
285
|
+
*/
|
|
286
|
+
export function foldForMatching(text) {
|
|
287
|
+
const src = String(text);
|
|
288
|
+
// A character plus any marks that follow it. `\uFF9E`/`\uFF9F` are the half-width voiced sound marks:
|
|
289
|
+
// they behave as marks here and are not in `\p{M}`.
|
|
290
|
+
const cluster = /\P{M}[\p{M}\uFF9E\uFF9F]*|[\p{M}\uFF9E\uFF9F]+/gu;
|
|
291
|
+
let folded = "";
|
|
292
|
+
const start = []; // folded position -> where its cluster starts in the original
|
|
293
|
+
const end = []; // folded position -> where its cluster ends in the original
|
|
294
|
+
for (let m = cluster.exec(src); m !== null; m = cluster.exec(src)) {
|
|
295
|
+
// FORMAT CHARACTERS ARE DROPPED FOR MATCHING, never from the output. NFKC keeps every one of them:
|
|
296
|
+
// a zero-width space, joiner, non-joiner, soft hyphen or byte-order mark sitting INSIDE a name
|
|
297
|
+
// survives the fold, and the name is then printed in clear. Measured 2026-09-27: a zero-width space
|
|
298
|
+
// one character into a protected name defeated the match completely. They carry no width and a reader
|
|
299
|
+
// cannot see them, so a page can carry a party's name looking exactly like the redacted rows beside
|
|
300
|
+
// it. Dropping them here only ever makes a match MORE likely, and the splice still removes the whole
|
|
301
|
+
// original range, format characters included.
|
|
302
|
+
const f = m[0].normalize("NFKC").replace(/\p{Cf}/gu, "");
|
|
303
|
+
for (let k = 0; k < f.length; k++) { start.push(m.index); end.push(m.index + m[0].length); }
|
|
304
|
+
folded += f;
|
|
305
|
+
}
|
|
306
|
+
start.push(src.length);
|
|
307
|
+
end.push(src.length);
|
|
308
|
+
return { folded, start, end };
|
|
309
|
+
}
|
|
310
|
+
|
|
140
311
|
/**
|
|
141
312
|
* A name matches where it stands on its own, not where it happens to be inside a word.
|
|
142
313
|
*
|
|
@@ -153,8 +324,32 @@ const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
|
153
324
|
*
|
|
154
325
|
* WHAT IT STILL DOES NOT CATCH, stated because a redaction that overclaims is the thing this module
|
|
155
326
|
* warns about: a name fused into a longer alphanumeric token with no separator — a slug, a domain, a
|
|
156
|
-
* run-together identifier.
|
|
157
|
-
*
|
|
327
|
+
* run-together identifier.
|
|
328
|
+
*
|
|
329
|
+
* THIS COMMENT USED TO SAY THE SCORER DOES NOT PRINT THE REFERENCE IN THOSE FORMS. It does, and the
|
|
330
|
+
* claim was the dangerous half of the paragraph: a reader who believed it would stop looking.
|
|
331
|
+
*
|
|
332
|
+
* · The `run:` line prints the run directory, and a run directory is named after the matter — five
|
|
333
|
+
* of forty scenario slugs measured on the box are a client's name with nothing done to them.
|
|
334
|
+
* · Carry-through prints source URLs, and a mark inside a URL path is fused to what surrounds it.
|
|
335
|
+
* · A native-script page produced two fused matches on the first real page it was read against.
|
|
336
|
+
*
|
|
337
|
+
* So the honest position is narrower than it was written: this rule protects a name standing on its
|
|
338
|
+
* own, with a plural or possessive, at any width. It does not protect one fused into a longer token,
|
|
339
|
+
* that case is not rare, and the forms it misses are being counted rather than assumed away.
|
|
340
|
+
*
|
|
341
|
+
* A URL IS NOT AN EXEMPTION, and that is a decision rather than a side effect. A locator is not prose, and there was an argument that a name inside one is an address
|
|
342
|
+
* rather than a disclosure — it was not taken. A reader who can see the address can fetch it, and what
|
|
343
|
+
* comes back names the party as plainly as the page would have. So a protected name is withheld inside
|
|
344
|
+
* a URL exactly as it is anywhere else: as a path segment, against a hyphen, between slashes, as a
|
|
345
|
+
* subdomain, as a query value. An arm holds all five, because this used to be true by accident of the
|
|
346
|
+
* boundary rule and is now true on purpose.
|
|
347
|
+
*
|
|
348
|
+
* THE ONE DECLINE INSIDE A URL IS THE FUSED CASE ABOVE, not a rule about URLs. `…/about/NAMEgroup`
|
|
349
|
+
* survives for the same reason `NAMEgroup` survives in prose, and the measurement that decided against
|
|
350
|
+
* refining the boundary counted it: 789 protected names across six scored pages, three declines, one of
|
|
351
|
+
* them inside a URL. Nothing here narrows that; it only stops a reader concluding from a URL decline
|
|
352
|
+
* that locators are treated differently.
|
|
158
353
|
*/
|
|
159
354
|
const matcher = (name) => new RegExp(`(?<![\\p{L}\\p{N}])${escapeRe(name)}(?:['’]s|s)?(?![\\p{L}\\p{N}])`, "giu");
|
|
160
355
|
|
|
@@ -170,24 +365,93 @@ const matcher = (name) => new RegExp(`(?<![\\p{L}\\p{N}])${escapeRe(name)}(?:['
|
|
|
170
365
|
*/
|
|
171
366
|
export function redactor({ names = new Set(), prose = new Set(), hint = "run again with --names to read it" } = {}) {
|
|
172
367
|
const index = new Map();
|
|
173
|
-
const ordered = [...names].sort((a, b) => b.length - a.length || a.localeCompare(b));
|
|
174
368
|
// The token is the entry's position in a STABLE ordering of the protected set, not its position in the
|
|
175
369
|
// reference: the same run scored twice must redact to the same tokens, and two entries that differ only
|
|
176
370
|
// in case are one name.
|
|
177
371
|
[...names].sort((a, b) => a.localeCompare(b)).forEach((n, i) => index.set(n.toLowerCase(), i + 1));
|
|
178
|
-
|
|
372
|
+
// LONGEST FOLDED FIRST, not longest raw. The reason for longest-first is about the text being matched,
|
|
373
|
+
// and that text is now the folded projection — folding does not preserve length, so raw order can put
|
|
374
|
+
// a shorter name first and leave the tail of a longer one beside a token claiming it was removed.
|
|
375
|
+
const withFold = (s) => ({ raw: s, fold: foldForMatching(s).folded });
|
|
376
|
+
const ordered = [...names].map(withFold).sort((a, b) => b.fold.length - a.fold.length || a.raw.localeCompare(b.raw));
|
|
377
|
+
const proseOrdered = [...prose].map(withFold).sort((a, b) => b.fold.length - a.fold.length);
|
|
179
378
|
|
|
180
379
|
return function redact(text) {
|
|
181
|
-
|
|
380
|
+
const src = String(text);
|
|
381
|
+
// The page is folded ONCE, not once per name: the protected set runs to hundreds of entries on a real
|
|
382
|
+
// reference, and re-folding for each would be a second full scan of the text per name.
|
|
383
|
+
const { folded, start, end } = foldForMatching(src);
|
|
384
|
+
// Accepted replacements, in ORIGINAL offsets, each accepted only where it overlaps nothing already
|
|
385
|
+
// accepted. That is what makes prose win over the names inside it.
|
|
386
|
+
const taken = [];
|
|
387
|
+
// UNION, NEVER DROP. A candidate overlapping something already accepted is MERGED into it rather than
|
|
388
|
+
// discarded, and dropping was the one step in this file that could SHRINK coverage. A name straddling
|
|
389
|
+
// the edge of an accepted prose range was refused whole, and the text outside that range — the rest of
|
|
390
|
+
// the name — was then emitted as it stood. Measured: prose "hello DRAV" beside a protected "DRAVOLINE"
|
|
391
|
+
// printed "OLINE" in clear, immediately after a withheld token. That is the worst shape this module
|
|
392
|
+
// has, because the token beside the fragment tells the reader the redaction is working. A name wholly
|
|
393
|
+
// inside a prose range was always correct; only the straddle leaked. Merging restores the invariant
|
|
394
|
+
// the fold already relies on: every failure here is over-redaction and none is a leak.
|
|
395
|
+
//
|
|
396
|
+
// WHICH TEXT A MERGED RANGE CARRIES. Prose beats a name, because a sentence with its names swapped out
|
|
397
|
+
// still says what the matter is about. Between two names the INCUMBENT keeps its token — the earliest
|
|
398
|
+
// entry in a stable ordering — because the same run scored twice must redact to the same tokens.
|
|
399
|
+
const take = (at, stop, isFolded, withText, isProse) => {
|
|
400
|
+
let from = isFolded ? start[at] : at;
|
|
401
|
+
let to = isFolded ? end[stop - 1] : stop;
|
|
402
|
+
const hits = [];
|
|
403
|
+
for (let i = 0; i < taken.length; i++) if (from < taken[i].to && to > taken[i].from) hits.push(i);
|
|
404
|
+
if (!hits.length) { taken.push({ from, to, withText, isProse }); return; }
|
|
405
|
+
let text = withText;
|
|
406
|
+
let prose = isProse;
|
|
407
|
+
for (const i of hits) {
|
|
408
|
+
const t = taken[i];
|
|
409
|
+
from = Math.min(from, t.from);
|
|
410
|
+
to = Math.max(to, t.to);
|
|
411
|
+
if (t.isProse || !prose) { text = t.withText; prose = prose || t.isProse; }
|
|
412
|
+
}
|
|
413
|
+
// Accepted ranges are pairwise disjoint, so every range inside the merged extent is among the hits
|
|
414
|
+
// and one pass is enough. `hits` ascends, so removing from the end keeps the earlier indexes valid.
|
|
415
|
+
for (let k = hits.length - 1; k >= 0; k--) taken.splice(hits[k], 1);
|
|
416
|
+
taken.push({ from, to, withText: text, isProse: prose });
|
|
417
|
+
};
|
|
418
|
+
|
|
419
|
+
// PROSE FIRST, AND IT WINS ANY OVERLAP: a sentence with its names swapped out still says what the
|
|
420
|
+
// matter is about, so the whole sentence goes rather than the names inside it.
|
|
182
421
|
for (const p of proseOrdered) {
|
|
183
|
-
|
|
184
|
-
|
|
422
|
+
const needle = p.fold || p.raw; // a string that folds to nothing is matched as it stands
|
|
423
|
+
// AN EMPTY ENTRY IS SKIPPED, and the guard is here rather than left to the producer. `indexOf("")`
|
|
424
|
+
// is 0 and the loop below advances by the needle's length, so an empty string spins forever. Nothing
|
|
425
|
+
// empty can arrive from a reference today, because every branch that collects one tests it first —
|
|
426
|
+
// but that is the producer's property, not this function's, and a hang has no error message.
|
|
427
|
+
if (!needle) continue;
|
|
428
|
+
const hay = p.fold ? folded : src;
|
|
429
|
+
for (let at = hay.indexOf(needle); at !== -1; at = hay.indexOf(needle, at + needle.length)) {
|
|
430
|
+
take(at, at + needle.length, Boolean(p.fold), `[withheld — ${hint}]`, true);
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
for (const n of ordered) {
|
|
434
|
+
const needle = n.fold || n.raw;
|
|
435
|
+
if (!needle) continue; // same reason as the prose loop above
|
|
436
|
+
const hay = n.fold ? folded : src;
|
|
437
|
+
const re = matcher(needle);
|
|
438
|
+
const withText = `«name ${index.get(n.raw.toLowerCase())}»`;
|
|
439
|
+
for (let m = re.exec(hay); m !== null; m = re.exec(hay)) {
|
|
440
|
+
if (m[0].length === 0) { re.lastIndex += 1; continue; } // never advance on an empty match
|
|
441
|
+
take(m.index, m.index + m[0].length, Boolean(n.fold), withText, false);
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
if (!taken.length) return src;
|
|
446
|
+
taken.sort((a, b) => a.from - b.from);
|
|
447
|
+
let out = "";
|
|
448
|
+
let at = 0;
|
|
449
|
+
for (const t of taken) {
|
|
450
|
+
if (t.from < at) continue;
|
|
451
|
+
out += src.slice(at, t.from) + t.withText;
|
|
452
|
+
at = t.to;
|
|
185
453
|
}
|
|
186
|
-
|
|
187
|
-
// so the guard did no work the replace did not, and on a long protected set it was a second full
|
|
188
|
-
// scan of the text for every name.
|
|
189
|
-
for (const n of ordered) out = out.replace(matcher(n), `«name ${index.get(n.toLowerCase())}»`);
|
|
190
|
-
return out;
|
|
454
|
+
return out + src.slice(at);
|
|
191
455
|
};
|
|
192
456
|
}
|
|
193
457
|
|
|
@@ -205,7 +469,131 @@ export function redactor({ names = new Set(), prose = new Set(), hint = "run aga
|
|
|
205
469
|
* rather than rethrowing: a rethrow would reach node's own printer, which writes to the real file
|
|
206
470
|
* descriptor and never passes through anything installed here.
|
|
207
471
|
*/
|
|
208
|
-
|
|
472
|
+
/**
|
|
473
|
+
* The redactor for a line the SCORER ITSELF wrote — a heading, a column ruler, a label.
|
|
474
|
+
*
|
|
475
|
+
* It applies the party names and the reference's own sentences, and DROPS the derived words. So a
|
|
476
|
+
* scaffolding word that happens to sit inside a party's name survives, and a reader keeps the structure
|
|
477
|
+
* that tells them what was withheld.
|
|
478
|
+
*
|
|
479
|
+
* IT IS NOT A BYPASS, and that is the whole reason it is a redactor rather than a raw write. Routing
|
|
480
|
+
* structural lines straight to the original writer would make this a hole: a data line sent through it by
|
|
481
|
+
* mistake would print a party name in clear. Here the worst a mistake costs is a shortened form — the full
|
|
482
|
+
* name is still taken out, and the given-name layer is exactly as strict as the main one. Every failure
|
|
483
|
+
* mode of this path is over-redaction; none is a name escaping.
|
|
484
|
+
*/
|
|
485
|
+
export function authoredRedactor({ names = new Set(), prose = new Set(), derived = new Set(), hint } = {}) {
|
|
486
|
+
// SUBTRACTS, rather than being handed a narrower set to union. The first shape of this had `redactor`
|
|
487
|
+
// take the derived words as a separate argument, so every existing caller that did not pass them lost
|
|
488
|
+
// the layer silently — including the live install. Two arms caught it. Dropping a layer has to be the
|
|
489
|
+
// thing a caller asks for explicitly; the default cannot be the weaker one.
|
|
490
|
+
const kept = new Set([...names].filter((x) => !derived.has(x)));
|
|
491
|
+
return redactor({ names: kept, prose, ...(hint ? { hint } : {}) });
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
// The authored printer, live only while a redaction is installed. Held here rather than threaded through
|
|
495
|
+
// every caller so a call site needs nothing but the function.
|
|
496
|
+
let authoredPrint = null;
|
|
497
|
+
|
|
498
|
+
/**
|
|
499
|
+
* Print a line this tool authored, with the derived-word layer dropped. With no redaction installed it is
|
|
500
|
+
* an ordinary write, which is the `--names` case: nothing is withheld, so there is nothing to drop.
|
|
501
|
+
*/
|
|
502
|
+
/**
|
|
503
|
+
* Redact a STRUCTURE rather than its serialisation.
|
|
504
|
+
*
|
|
505
|
+
* WHY A SERIALISED PAYLOAD CANNOT BE REDACTED AS TEXT. `--json` used to be built, stringified, and sent
|
|
506
|
+
* through the boundary like any other line, which redacts a document that has no prose in it — every
|
|
507
|
+
* string in it is either a field NAME the code wrote or a VALUE out of the data, and the two need
|
|
508
|
+
* opposite treatment. Measured on one real score: the keys `owner` and `additional` came back as
|
|
509
|
+
* `«name 141»` and `«name 1»`, because both are distinctive words of multi-word parties in that
|
|
510
|
+
* reference. Two of the payload's fields were unaddressable — a consumer asking for `owner` finds
|
|
511
|
+
* nothing, and cannot discover what to ask for instead. That is past readability: the JSON is an
|
|
512
|
+
* interface.
|
|
513
|
+
*
|
|
514
|
+
* KEYS ARE AUTHORED AND VALUES ARE DATA, so each gets the instrument that fits it. A key is a literal in
|
|
515
|
+
* this repository's source: the derived layer has no business in it, and dropping that layer is exactly
|
|
516
|
+
* what `authoredRedactor` is for. A value came out of a reference or a run, so it keeps every layer,
|
|
517
|
+
* including the derived one that closed the leak where a two-word proprietor's first word survived in
|
|
518
|
+
* prose.
|
|
519
|
+
*
|
|
520
|
+
* A KEY IS STILL REDACTED, never passed through. If a party's whole name is also a field name, the field
|
|
521
|
+
* name goes — that collision is worth a token, and it is not the ordinary-word case this exists for.
|
|
522
|
+
*
|
|
523
|
+
* Numbers, booleans and null are returned as they are: there is nothing in them to withhold, and turning
|
|
524
|
+
* them into strings would change the payload's shape.
|
|
525
|
+
*/
|
|
526
|
+
export function redactDeep(value, { redactValue, redactKey }) {
|
|
527
|
+
const walk = (v, path = "") => {
|
|
528
|
+
if (typeof v === "string") return redactValue(v);
|
|
529
|
+
if (Array.isArray(v)) return v.map((x) => walk(x, path));
|
|
530
|
+
if (v && typeof v === "object") {
|
|
531
|
+
const out = {};
|
|
532
|
+
// TWO SIBLING KEYS CAN REDUCE TO ONE TOKEN, and the later write would take the earlier field with
|
|
533
|
+
// it. The matcher tolerates a trailing plural, so a protected name and that name plus `s` both
|
|
534
|
+
// match; it is case-insensitive, so two spellings of one name collide too. Rebuilding the object
|
|
535
|
+
// key by key, the second write wins and the first field is gone — valid JSON, no warning, and a
|
|
536
|
+
// consumer cannot tell a field ever existed. That is the failure mode this whole module exists to
|
|
537
|
+
// prevent, so it refuses rather than guessing which field to keep.
|
|
538
|
+
//
|
|
539
|
+
// MEASURED BEFORE BEING BUILT: across four real scored payloads, 585 objects and 1634 protected
|
|
540
|
+
// names, there is no such pair today. The mechanism is real and the path is not reachable on
|
|
541
|
+
// anything this box produces, which is why this is a guard and not a repair.
|
|
542
|
+
//
|
|
543
|
+
// THE REFUSAL NAMES THE TOKEN AND THE PATH, NEVER THE KEYS. Saying which keys collided would print
|
|
544
|
+
// the very name being withheld — the keys are the protected string, that is why they collided. A
|
|
545
|
+
// reader who needs them asks with `--names`, where nothing is withheld and the collision cannot
|
|
546
|
+
// happen.
|
|
547
|
+
const seen = new Map();
|
|
548
|
+
for (const [k, x] of Object.entries(v)) {
|
|
549
|
+
const rk = redactKey(k);
|
|
550
|
+
if (seen.has(rk)) {
|
|
551
|
+
const e = new Error(`two sibling keys reduce to the same token ${rk} at ${path || "the payload root"}`
|
|
552
|
+
+ ` — this payload cannot be rendered addressably, so it is refused rather than written with a`
|
|
553
|
+
+ ` field silently dropped. Run again with --names to see which keys they are.`);
|
|
554
|
+
e.code = "REDACTED_KEY_COLLISION";
|
|
555
|
+
e.token = rk;
|
|
556
|
+
e.at = path || "(root)";
|
|
557
|
+
throw e;
|
|
558
|
+
}
|
|
559
|
+
seen.set(rk, true);
|
|
560
|
+
out[rk] = walk(x, path ? `${path}.${rk}` : rk);
|
|
561
|
+
}
|
|
562
|
+
return out;
|
|
563
|
+
}
|
|
564
|
+
return v;
|
|
565
|
+
};
|
|
566
|
+
return walk(value);
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
export function printAuthored(line) {
|
|
570
|
+
const text = `${line}\n`;
|
|
571
|
+
if (authoredPrint) authoredPrint(text);
|
|
572
|
+
else process.stdout.write(text);
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
/**
|
|
576
|
+
* Write text that has ALREADY been redacted, part by part, through the original stream.
|
|
577
|
+
*
|
|
578
|
+
* THE ONE CALLER THIS IS FOR is a `redactDeep` payload, and the contract is narrow on purpose. Every
|
|
579
|
+
* string in such a payload has been through a redactor already — values through the full one, keys
|
|
580
|
+
* through the authored one — so sending it round again would re-apply the derived layer to keys and put
|
|
581
|
+
* back the exact defect `redactDeep` exists to remove.
|
|
582
|
+
*
|
|
583
|
+
* IT IS A HOLE IF IT IS MISUSED, and unlike `printAuthored` there is no weaker layer standing behind it:
|
|
584
|
+
* hand this raw run text and it prints in clear. So it takes text that a redactor produced, never text a
|
|
585
|
+
* caller assembled, and the two arms that drive it both assert a name is absent rather than that a
|
|
586
|
+
* structure survived.
|
|
587
|
+
*/
|
|
588
|
+
export function printPreRedacted(text) {
|
|
589
|
+
if (rawPrint) rawPrint(text);
|
|
590
|
+
else process.stdout.write(text);
|
|
591
|
+
}
|
|
592
|
+
|
|
593
|
+
// The raw printer, live only while a redaction is installed, for the reason above.
|
|
594
|
+
let rawPrint = null;
|
|
595
|
+
|
|
596
|
+
export function installRedaction(redact, io = { console, process }, authored = null) {
|
|
209
597
|
const { console: con, process: proc } = io;
|
|
210
598
|
// THE REFERENCE TO PUT BACK AND THE FUNCTION TO CALL ARE NOT THE SAME THING. A stream write has to be
|
|
211
599
|
// called bound to its stream, but restoring the BOUND copy leaves a different function on the object
|
|
@@ -227,14 +615,22 @@ export function installRedaction(redact, io = { console, process }) {
|
|
|
227
615
|
original.stderr(`${redact(e?.stack ?? String(e))}\n`);
|
|
228
616
|
proc.exit(1);
|
|
229
617
|
};
|
|
618
|
+
// Writes through the ORIGINAL stream, so the wrap above cannot re-apply the
|
|
619
|
+
// derived layer to a line that just had it dropped.
|
|
620
|
+
authoredPrint = (text) => original.stdout(authored ? authored(text) : redact(text));
|
|
621
|
+
// Straight through, because the caller's contract is that every part of it is already redacted.
|
|
622
|
+
rawPrint = (text) => original.stdout(text);
|
|
230
623
|
proc.on?.("uncaughtException", onUncaught);
|
|
231
624
|
proc.on?.("unhandledRejection", onUncaught);
|
|
232
625
|
return function uninstall() {
|
|
626
|
+
authoredPrint = null;
|
|
627
|
+
rawPrint = null;
|
|
233
628
|
for (const m of ["log", "error", "warn", "info", "debug"]) con[m] = original[m];
|
|
234
629
|
proc.stdout.write = original.stdoutRef;
|
|
235
630
|
proc.stderr.write = original.stderrRef;
|
|
236
631
|
proc.off?.("uncaughtException", onUncaught);
|
|
237
632
|
proc.off?.("unhandledRejection", onUncaught);
|
|
633
|
+
authoredPrint = null;
|
|
238
634
|
};
|
|
239
635
|
}
|
|
240
636
|
|
package/driver/screen-gate.mjs
CHANGED
|
@@ -99,9 +99,9 @@ function parseVerdict(notes) {
|
|
|
99
99
|
*/
|
|
100
100
|
export function findScreenGateViolations(findingsContent, fetchedUriSet) {
|
|
101
101
|
// URI-granularity normalization: the record_fetch ledger logs the registration-INSTANCE URI (e.g.
|
|
102
|
-
// /mark/ch/
|
|
103
|
-
// SLASH (→ /mark/ch/
|
|
104
|
-
// other instance) suffix is not a false-negative on the membership test —
|
|
102
|
+
// /mark/ch/30419/2014) while the gate parses each Notes URI through URI_RE, which stops at the first
|
|
103
|
+
// SLASH (→ /mark/ch/30419). Reduce BOTH sides through the SAME regex so a slash-separated /<year> (or
|
|
104
|
+
// other instance) suffix is not a false-negative on the membership test — a false hard-halt
|
|
105
105
|
// that blocked a live pharma matter twice on 2026-06-17 (the record WAS fetched, logged as …/2014).
|
|
106
106
|
//
|
|
107
107
|
// …and CASE-FOLD, for the same reason at a different granularity (2026-07-28). The fetched universe is
|
|
@@ -92,14 +92,6 @@ cheap by construction**, so they enumerate fully and cheaply. A saturation crowd
|
|
|
92
92
|
count-only call (`limit:1`), not enumerated. The expensive failure mode of the old funnel — deep-paging a
|
|
93
93
|
saturated raw pile — does **not** exist: crowds are descriptors, named slices are bounded.
|
|
94
94
|
|
|
95
|
-
| Worker | register_enumerate calls (named slices) | count-only crowd descriptors | Phoneme | Image |
|
|
96
|
-
|---|---|---|---|---|
|
|
97
|
-
| `saturation-probe` unit | 0 | ~3–4 (count-only) | 0 | 0 |
|
|
98
|
-
| `primary-sweep` unit | ~8–14 (exact + substring band + per-major + meaning, where applicable) | ~1–2 | phonetic recipes | device-led |
|
|
99
|
-
| `transliteration-numeric` unit | ~4–6 (one per script/variant query) | ~1 | 0 | 0 |
|
|
100
|
-
| `incumbent-class` unit | ~2–4 | 0 | 0 | 0 |
|
|
101
|
-
| digest worker (merch-sweep + Option-D follow-ups) | ~2–4 | 0 | 0 | 0 |
|
|
102
|
-
|
|
103
95
|
**A genuine resource/time limit produces an `incomplete` block, never a sufficiency accept.** If
|
|
104
96
|
`register_enumerate` hits the provider 5000-record window or a resource ceiling on a slice, it returns
|
|
105
97
|
`{state:"incomplete", …, reason}` — the funnel writes that block verbatim and **stops there for that slice**.
|
|
@@ -372,7 +372,7 @@ loop and never a no-deliver halt.
|
|
|
372
372
|
|
|
373
373
|
# MODE B — DIGEST (combine the complete named band → register findings)
|
|
374
374
|
|
|
375
|
-
You are JUDGMENT, not the machine. The funnel (the unit-mode workers)
|
|
375
|
+
You are JUDGMENT, not the machine. The funnel (the unit-mode workers) either **enumerated** a search to completion or reported it **incomplete** (a crowd descriptor). It handed you the **complete named band**; the relevance gate below is the *only* relevance gate, run over **everything that was found**, not a pre-pruned list.
|
|
376
376
|
|
|
377
377
|
Your task gives the paths to: the per-axis prose digests (`register-units/<axis>.md` — audit-trail summary only), the variant manifest, `matter-context.md`, and `placement-recommendations.md`. The **band itself is read through the band tools** — you hold `band_shape` / `band_lookup` / `band_record`, and every call you make lands in the run's reading audit (that on-the-record trail is the point: the reading layer is as auditable as the frozen search plan).
|
|
378
378
|
|
|
@@ -187,7 +187,7 @@ Be aware these capabilities are missing, so the skill doesn't promise them:
|
|
|
187
187
|
|
|
188
188
|
## API budget
|
|
189
189
|
|
|
190
|
-
Sustained-burst behaviour is UNDECLARED.
|
|
190
|
+
Sustained-burst behaviour is UNDECLARED.
|
|
191
191
|
|
|
192
192
|
Session-key validation: send one cheap `register_search({ name: "EXAMPLEMARK", limit: 1 })` before doing any real work. If it returns 401/403, halt and surface to orchestrator immediately.
|
|
193
193
|
|
|
@@ -215,22 +215,16 @@ For each transliteration variant in the manifest:
|
|
|
215
215
|
→ name:<root> nice-class:<class N> limit=1 fields=[uri] → `incomplete` crowd-descriptor block
|
|
216
216
|
(the count tells the lawyer how crowded; it clears nothing)
|
|
217
217
|
|
|
218
|
-
1b.
|
|
219
|
-
count is over the resource ceiling (a crowd — GREAT≈28k, OUTDOORS≈2.7k), the Step-2 substring enumerate
|
|
220
|
-
would return `incomplete`, and fanning it out per-major + phonetic is the grind that double-SIGKILLed the
|
|
221
|
-
stage. STOP the fan-out: do NOT run Step 2b (per-major) or Step 3 (phonetic) on a crowd root; keep the exact
|
|
222
|
-
name-list (Step 2c). Then WRITE depends on WHICH root it is:
|
|
218
|
+
1b. WRITE depends on WHICH root it is:
|
|
223
219
|
• the DISTINCTIVE dominant category (the highest-relevance slice) → write ONE `incomplete` block (a
|
|
224
220
|
material could-not-finish for judgment — the COLORA→色彩 case);
|
|
225
221
|
• a stripped COMMON component (NOT the distinctive anchor — GREAT/OUTDOORS) → write NO block; the
|
|
226
222
|
`saturation-probe` count is the sole, immaterial signal (a duplicate primary-sweep crowd risks
|
|
227
223
|
mis-reading as a material in-class gap).
|
|
228
|
-
|
|
229
|
-
that narrows per-region). For an all-common-words phrase mark, the exact phrase + near-neighbours is the
|
|
224
|
+
For an all-common-words phrase mark, the exact phrase + near-neighbours is the
|
|
230
225
|
dangerous band (Recipe 1), not the component substrings.
|
|
231
226
|
|
|
232
|
-
2. ENUMERATE the substring band (register_enumerate) —
|
|
233
|
-
class + region scoped (breadth, not sufficiency)
|
|
227
|
+
2. ENUMERATE the substring band (register_enumerate) — class + region scoped (breadth, not sufficiency)
|
|
234
228
|
→ register_enumerate name:<root> match=default nice_classes:<in-scope full Nice set, goods AND
|
|
235
229
|
services 42/44 — never a goods-only 1/5 subset> regions:<in-scope>
|
|
236
230
|
→ the tool pages to has_more:false and returns:
|
|
@@ -150,7 +150,7 @@ The orchestrator itself makes few direct tool calls. Most calls happen inside su
|
|
|
150
150
|
- `clearance-common-law`: **15** `perplexity_research` per workflow
|
|
151
151
|
*(rationale: cost-based overflow protection against a looping worker — the API is usage-billed; a search-as-code grid call ≈ $0.06, prose follow-ups ≈ $0.01–0.15 each (measured 2026-06-10). Typical workflow uses 2–6 calls/mark; 15 leaves headroom for thinness re-spawns)*
|
|
152
152
|
- `clearance-register`: **150** provider calls per workflow, across all marks
|
|
153
|
-
*(rationale: Corsearch billing-tier ceiling; calibrated against May runs which used 80–110 calls each
|
|
153
|
+
*(rationale: Corsearch billing-tier ceiling; calibrated against May runs which used 80–110 calls each)*
|
|
154
154
|
- This skill: file read/write and memory write — bounded by workflow steps. It builds no workbook and sends no mail: the driver does both at publish, in code.
|
|
155
155
|
|
|
156
156
|
Cross-pollination dispatches add at most **10** calls split across the two sub-skills (Option D cap).
|
|
@@ -306,7 +306,7 @@ The **Methodology** sheet carries: matter-context summary, search approach, plac
|
|
|
306
306
|
**Detail-fetch coverage** (register search-depth floor — kept here, not in the Excel spec, because the Step 2.6 skeptic review depends on it) — rank the union of unique URIs returned across all register searches by signal strength, then detail-fetch as follows:
|
|
307
307
|
|
|
308
308
|
1. **Identical-mark hits** — fetch all, no cap.
|
|
309
|
-
2. **Near-exact (dominant-token-substring) in-class-live hits — fetch all, no cap (Slice A).** The mirror, at the orchestrator floor, of the unit-side [exact-in-class-live floor](../clearance-register/unit.md#exact-in-class-live-floor-primary-sweep-unit-owns-it). The **near-exact band** — where the **dominant element** (from the variant manifest) appears as a *substring* of `mark_text` (case-insensitive, after the `normalize()` strip in [status-rules.md](../clearance-register/status-rules.md), see the Identical-match normalisation shape) in a **filed target class** with **live** status, but is not an identical match (e.g. NORDWAVE NOVAPULSE, NOVAPULSE.com on dominant element NOVAPULSE) — is **enumerate-and-fetch, no top-N, no score gate**. This is the F-1/F-3 hole: the near-exact band otherwise falls into the Top-K sample (item 5) and a dangerous in-class-live conflict gets paged past the cliff.
|
|
309
|
+
2. **Near-exact (dominant-token-substring) in-class-live hits — fetch all, no cap (Slice A).** The mirror, at the orchestrator floor, of the unit-side [exact-in-class-live floor](../clearance-register/unit.md#exact-in-class-live-floor-primary-sweep-unit-owns-it). The **near-exact band** — where the **dominant element** (from the variant manifest) appears as a *substring* of `mark_text` (case-insensitive, after the `normalize()` strip in [status-rules.md](../clearance-register/status-rules.md), see the Identical-match normalisation shape) in a **filed target class** with **live** status, but is not an identical match (e.g. NORDWAVE NOVAPULSE, NOVAPULSE.com on dominant element NOVAPULSE) — is **enumerate-and-fetch, no top-N, no score gate**. This is the F-1/F-3 hole: the near-exact band otherwise falls into the Top-K sample (item 5) and a dangerous in-class-live conflict gets paged past the cliff. Slice A is **exhaustive** — distinct from the sample-with-disclosure Slice B below.
|
|
310
310
|
3. **Phonetic-equivalent fringe — floor, sample-with-disclosure (Slice B).** Run the provider phonetic capability on the dominant token for the matter languages (provider-agnostic: `<provider>_expand_phoneme` then `match_mode: phonetic`; Clarivate uses native `match_mode: phonetic`). It **is a floor** — it MUST run for the dominant token in the filed class — but phonetic sets are unbounded, so it is **sample-with-disclosure**: where it cannot be fully worked, the coverage unit is **`coverage-limited`** (reason: "phonetic fringe sampled, not enumerated"), never `confirmed-clean`. Keep Slice A (exhaustive) and Slice B (sampled) **structurally distinct** — they have different correctness properties.
|
|
311
311
|
4. **Watchlist-owner hits** — fetch all.
|
|
312
312
|
5. **Top-K by relevance** — sample from each match-mode (exact / phrase / default) and across regions to ensure representative coverage. If the worker stops detail-fetching before ~25 URIs across all match-modes, the worker MUST answer in its digest audit: did the result set genuinely run out, or is this the empirical execution-tier truncation pattern (~10 URIs, no explanation)? The 25 figure is a tripwire-with-question, not a target — the Step 2.6 skeptic reads the worker's answer and re-spawns escalated to Opus if the answer is missing or unconvincing.
|