clearotron 0.3.2-beta.13 → 0.3.2-beta.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/clearotron.mjs +7 -1
- package/bin/example.mjs +49 -22
- package/build-info.json +2 -2
- package/driver/CHANGELOG.md +22 -0
- package/driver/contract-vocabulary.mjs +5 -5
- package/driver/engine/mcp/codex-config.mjs +3 -11
- package/driver/named-band.mjs +1 -1
- package/driver/package.json +1 -1
- package/driver/pipeline-knockout.mjs +3 -1
- package/driver/pipeline.mjs +30 -7
- package/driver/publish/index.mjs +10 -1
- package/driver/publish/render-knockout.mjs +31 -6
- package/driver/publish/render.mjs +54 -2
- package/driver/publish/templates/report.css +16 -1
- package/driver/register-availability.mjs +2 -2
- package/driver/register-plan.mjs +229 -55
- package/driver/skills/clearance-register/unit.md +1 -1
- package/driver/skills/clearance-variants/SKILL.md +10 -4
- package/driver/stages.mjs +1 -1
- package/driver/suite-census.json +45 -3
- package/driver/variant-manifest-model.mjs +39 -1
- package/mcp-server/CHANGELOG.md +8 -0
- package/mcp-server/lib/audit-view.mjs +5 -3
- package/mcp-server/lib/brief.mjs +10 -4
- package/mcp-server/lib/runs.mjs +35 -1
- package/mcp-server/package.json +1 -1
- package/mcp-server/server.mjs +6 -3
- package/package.json +1 -1
- package/portal-ui/dist/assets/{index-7Lq-dXDV.css → index-5CCwiJG7.css} +10 -0
- package/portal-ui/dist/assets/{index-w8GFZftk.js → index-DMthc7PQ.js} +8 -1
- package/portal-ui/dist/index.html +2 -2
- package/portal-ui/package.json +1 -1
- package/providers/_shared/execute-plan.mjs +11 -1
- package/providers/_shared/term-shape.mjs +76 -0
- package/providers/clarivate/src/capabilities.js +36 -0
- package/providers/clarivate/src/core.js +119 -0
- package/providers/corsearch/src/capabilities.js +24 -0
- package/providers/corsearch/src/core.js +7 -0
- package/providers/oauth-mcp-bridge/CHANGELOG.md +8 -0
- package/providers/oauth-mcp-bridge/package.json +1 -1
- package/providers/signa/src/capabilities.js +29 -0
- package/providers/signa/src/core.js +17 -0
- package/shared/demo-start-args.mjs +10 -0
- package/shared/stdio-connect.mjs +41 -10
- package/shared/toml-string.mjs +25 -0
package/driver/register-plan.mjs
CHANGED
|
@@ -50,7 +50,7 @@ import { REGISTER_AXES } from "./coverage-ledger.mjs";
|
|
|
50
50
|
import { canonicalJurisdictionCode, isKnownJurisdictionCode } from "./jurisdiction-codes.mjs"; // item 13 — a searched territory traces to an executed query
|
|
51
51
|
import { normalizeTerritory } from "../providers/_shared/territory-codes.mjs";
|
|
52
52
|
import { bindingLayersFor, layerCoverageFor } from "./binding-layers.mjs";
|
|
53
|
-
import { entryTermIssues, hasAnchoredWildcard, termAnnotationIssue, termMarkupIssue, termShapeIssue, termSubstanceIssue } from "../providers/_shared/term-shape.mjs";
|
|
53
|
+
import { entryTermIssues, goodsTermsList, stripGoodsReservedWords, hasAnchoredWildcard, termAnnotationIssue, termMarkupIssue, termShapeIssue, termSubstanceIssue } from "../providers/_shared/term-shape.mjs";
|
|
54
54
|
import { formKey, romanizationSpellings, isNonLatinTerm } from "../providers/_shared/script-form.mjs";
|
|
55
55
|
|
|
56
56
|
export const PLAN_SCHEMA_VERSION = 1;
|
|
@@ -188,6 +188,37 @@ export function ownerIntersectionGap(entry, capabilities) {
|
|
|
188
188
|
return ownerIntersectionUnsupportedReason(capabilities.id ?? "unknown");
|
|
189
189
|
}
|
|
190
190
|
|
|
191
|
+
// ── GOODS-TEXT NARROWING: a class is not a specification ───────────────────────────────────────────
|
|
192
|
+
//
|
|
193
|
+
// An entry carrying `goods_text` asks the register to match the goods and services DESCRIPTION, not
|
|
194
|
+
// only the Nice class the slice already carries. Both are needed and they are not the same question:
|
|
195
|
+
// a class is a filing-fee bucket that holds everything from headphones to jukeboxes, so a contains
|
|
196
|
+
// sweep scoped to a class alone comes back a crowd on any crowded word.
|
|
197
|
+
//
|
|
198
|
+
// WHY THIS IS A CAPABILITY AND NOT A PREDICATE. It is a SECOND FIELD on the same request, exactly as
|
|
199
|
+
// the owner scope field is — it composes with the mark clause rather than replacing it. A provider
|
|
200
|
+
// whose connector cannot send it must DEFER the slice: dropping the goods clause and running the
|
|
201
|
+
// sweep anyway would return the crowd this entry exists to avoid, and record it as a searched slice.
|
|
202
|
+
// That is a widened search wearing a narrowed slice's qid, and it is the failure mode the whole
|
|
203
|
+
// `unsupported` lane exists to prevent.
|
|
204
|
+
export const goodsTextUnsupportedReason = (providerId) =>
|
|
205
|
+
`goods-and-services text narrowing is not supported by the active register provider (${providerId}) — the `
|
|
206
|
+
+ `goods description cannot be searched there, so the narrowed sweep this entry asks for was never run. `
|
|
207
|
+
+ `Running it on the class alone would have returned the crowd the narrowing exists to cut, so the slice `
|
|
208
|
+
+ `is deferred. It is a gap for judgment, never a clean negative.`;
|
|
209
|
+
|
|
210
|
+
/** Is a goods-narrowed entry executable on the provider? null when it is (or when it asks for no goods text). */
|
|
211
|
+
export function goodsTextGap(entry, capabilities) {
|
|
212
|
+
if (!capabilities) return null;
|
|
213
|
+
const goods = goodsTermsList(entry);
|
|
214
|
+
if (!goods.length) return null;
|
|
215
|
+
if (capabilities.goodsTextSearch === true) return null;
|
|
216
|
+
return goodsTextUnsupportedReason(capabilities.id ?? "unknown");
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
// The reader is `goodsTermsList`, imported from the shared term vocabulary — the compiler, the
|
|
220
|
+
// executor and the connectors all ask the question with the same function.
|
|
221
|
+
|
|
191
222
|
/**
|
|
192
223
|
* 2026-07-29 hardening — is a VARIANTS-MODEL value un-searchable as a mark term? Returns the
|
|
193
224
|
* plain-English reason, or null. Two classes shipped as silent nil "cleans" on the 2026-07-28 run
|
|
@@ -458,9 +489,19 @@ export function variantsFingerprint(manifest) {
|
|
|
458
489
|
elements: manifest.elements.map((e) => [e.value, e.kind]),
|
|
459
490
|
variants: manifest.variants.map((v) => [v.value, v.category]),
|
|
460
491
|
incumbent: manifest.incumbent_classes,
|
|
461
|
-
//
|
|
462
|
-
//
|
|
463
|
-
|
|
492
|
+
// ── WHAT BELONGS HERE IS WHAT CHANGES AN ENTRY ─────────────────────────────────────────────
|
|
493
|
+
//
|
|
494
|
+
// `watchlist_owners` is GONE from this fingerprint, and its removal follows the same rule that put
|
|
495
|
+
// it here: a field belongs when a manifest gaining it must not reuse a stored plan byte-identical.
|
|
496
|
+
// That list no longer compiles anything, so two manifests differing only in it now compile to the
|
|
497
|
+
// same plan and must fingerprint the same. The one cost is a single re-mint for a matter whose
|
|
498
|
+
// plan was stored before the owner lane was removed — and that re-mint produces the entries the
|
|
499
|
+
// compiler would produce today, which is the point of it.
|
|
500
|
+
//
|
|
501
|
+
// `goods_words` is here for the mirror-image reason: it DOES change an entry. Added conditionally,
|
|
502
|
+
// exactly as the romanisations below are, so a manifest carrying none fingerprints byte-identically
|
|
503
|
+
// to the way it always did and nothing re-mints for a field it never had.
|
|
504
|
+
...(manifest.goods_words?.length ? { goods_words: manifest.goods_words } : {}),
|
|
464
505
|
// Same rule for the romanisations: a manifest that carries none fingerprints exactly as it always
|
|
465
506
|
// did, and one that gains them is a DIFFERENT manifest, so a stored plan minted before the
|
|
466
507
|
// romanisations existed is never REUSED byte-identical for a manifest that now carries them.
|
|
@@ -994,7 +1035,7 @@ export function compileRegisterPlan({ manifest, job, form = null, skillVersion =
|
|
|
994
1035
|
// deferred coverage row is the loud form of "this was never really searchable", and stamping it
|
|
995
1036
|
// here keeps the plan byte-identical across compiles (a pure function of the value).
|
|
996
1037
|
const gap = allJurisdictionsDeferred ?? substanceGap ?? dropIssue ?? markupGap ?? predicateGap(e.predicate, e.term ?? e.terms?.[0], caps)
|
|
997
|
-
?? ownerIntersectionGap(e, caps);
|
|
1038
|
+
?? ownerIntersectionGap(e, caps) ?? goodsTextGap(e, caps);
|
|
998
1039
|
entries.push({ qid, nice_classes: classes, regions, ...romanStamp(e), ...rest,
|
|
999
1040
|
...(gap ? { unsupported: true, unsupported_reason: gap } : {}) });
|
|
1000
1041
|
return qid;
|
|
@@ -1106,6 +1147,140 @@ export function compileRegisterPlan({ manifest, job, form = null, skillVersion =
|
|
|
1106
1147
|
// closed when present, like `provenance`: a frozen pre-2050 plan carries none and its reader falls
|
|
1107
1148
|
// back to the old rule, so a resumed run does not change its answer because a field arrived.
|
|
1108
1149
|
const parentQid = push({ axis: "primary-sweep", predicate: "default", term: manifest.dominant_element, expected_kind: "enumerate", provenance: "mark", crowd_gate_parent: true });
|
|
1150
|
+
|
|
1151
|
+
// ── THE SAME SWEEP, NARROWED TO WHAT THE FILINGS COVER ──────────────────────────────────────────
|
|
1152
|
+
//
|
|
1153
|
+
// The parent above is the whole reason this entry exists. On a measured run it asked for the mark's
|
|
1154
|
+
// dominant word, already scoped to the matter's classes across 186 offices, and the register
|
|
1155
|
+
// answered 10,483 — over the 600 enumerate ceiling, so it was recorded as a crowd and NOT ONE
|
|
1156
|
+
// RECORD WAS READ. A class cannot cut that: class 9 holds headphones and jukeboxes alike, so
|
|
1157
|
+
// class-scoping is already spent by the time the crowd appears. The goods and services DESCRIPTION
|
|
1158
|
+
// is the only axis left that narrows the sweep without narrowing the mark.
|
|
1159
|
+
//
|
|
1160
|
+
// IT IS COMPILED ALWAYS, NOT ONLY WHEN THE PARENT CROWDS. A `when` guard here would be a second
|
|
1161
|
+
// crowd-gate mechanism, and it would buy nothing: on an uncrowded matter this entry returns a
|
|
1162
|
+
// SUBSET of the parent's own records, and mergeNamedBands folds them by `record_id` — one row per
|
|
1163
|
+
// record, first occurrence wins, and the survivor carries both qids in `_qids`. So the duplicate
|
|
1164
|
+
// costs one register question and changes no count downstream. Gating it would instead make the
|
|
1165
|
+
// plan's shape depend on a result, which is the property this compiler exists not to have: the same
|
|
1166
|
+
// manifest must compile to the same plan every time.
|
|
1167
|
+
//
|
|
1168
|
+
// The words come from the manifest (`goods_words`) and from nowhere else. Code never invents a
|
|
1169
|
+
// search term, and it never adds a synonym: what this entry can find is exactly what was written
|
|
1170
|
+
// there, which is why the list is the client's wording plus the model's synonyms and is recorded on
|
|
1171
|
+
// the run.
|
|
1172
|
+
// TWO REASONS NOT TO COMPILE IT AT ALL, and neither is a coverage loss.
|
|
1173
|
+
//
|
|
1174
|
+
// No words: the matter stated no goods wording and the profile carries none, so there is nothing to
|
|
1175
|
+
// narrow BY. An entry with an empty clause is the broad sweep under a second name.
|
|
1176
|
+
//
|
|
1177
|
+
// No capability: on a provider whose connector cannot send the clause, this entry would be stamped
|
|
1178
|
+
// `unsupported` and print a deferred coverage gap in every instructed class, on every matter —
|
|
1179
|
+
// "this was not searched" about a slice whose population is a STRICT SUBSET of the broad sweep that
|
|
1180
|
+
// did run. Same term, same classes, one clause more: wherever the parent ran, nothing here went
|
|
1181
|
+
// unsearched. The deferral lane is for coverage genuinely lost, and claiming a loss that did not
|
|
1182
|
+
// happen is the same false statement as hiding one.
|
|
1183
|
+
// A THIRD REASON, and it is a property of the register rather than of the matter: some registers
|
|
1184
|
+
// cannot express a LIST of alternatives on this field at all. Where that is so, a multi-word list
|
|
1185
|
+
// has no honest form — joining it would intersect the words instead of offering them as
|
|
1186
|
+
// alternatives, which asks for filings covering ALL of them and answers 200 with a population that
|
|
1187
|
+
// shrinks as the list grows. So the entry is not compiled; the broad sweep still runs.
|
|
1188
|
+
const goodsWords = Array.isArray(manifest.goods_words) ? manifest.goods_words : [];
|
|
1189
|
+
// A FOURTH REASON, and it is the one the parser deliberately does NOT decide: the list may carry a
|
|
1190
|
+
// short phrase, and only some registers match a phrase as a phrase. Where this one does not, the
|
|
1191
|
+
// connector would refuse the entry at the door — so the entry must not be compiled in the first
|
|
1192
|
+
// place. Checking it here keeps the parser free to accept what the model was told to write, and
|
|
1193
|
+
// keeps each register's behaviour in the one file that describes that register.
|
|
1194
|
+
// A MULTI-WORD ITEM THIS REGISTER CANNOT TAKE DROPS ON ITS OWN, and the single words beside it
|
|
1195
|
+
// still compile. Suppressing the whole entry over one phrase threw away the words that WOULD have
|
|
1196
|
+
// narrowed the sweep, which is the wrong direction to fail in: a narrowing that asks for less than
|
|
1197
|
+
// intended is narrower than intended, never wider, and the class-wide sweep still runs beside it.
|
|
1198
|
+
// What was dropped rides on the entry, so the omission is a fact on the plan rather than a silence.
|
|
1199
|
+
// ── A WORD THE REGISTER READS AS AN OPERATOR COMES OUT OF THE ITEM, NOT THE ITEM OUT OF THE LIST ──
|
|
1200
|
+
//
|
|
1201
|
+
// AND, OR, NOT, ADJ and NEAR are operators inside the value there and the field has no escape
|
|
1202
|
+
// syntax, so "near field communication" is a 400 rather than a narrower search — and since the list
|
|
1203
|
+
// rides one OR-joined value, that one term would take the whole narrowing down for the run.
|
|
1204
|
+
//
|
|
1205
|
+
// The word is removed and the rest of the item is still asked: "field communication" narrows
|
|
1206
|
+
// usefully, and throwing the term away over one word the vendor happens to reserve would lose a
|
|
1207
|
+
// choice the model made. An item that is nothing BUT reserved words has nothing left and goes.
|
|
1208
|
+
// Every removal is disclosed, because a term that reached the wire in a different shape from the
|
|
1209
|
+
// one written must never do so in silence.
|
|
1210
|
+
//
|
|
1211
|
+
// STRIPPED ONCE, HERE, AND THE DISTANCES TRAVEL WITH THE WORDS. The plan stores what will be asked,
|
|
1212
|
+
// so the gap a removal opened has to ride beside the term: the connector receives "controllers
|
|
1213
|
+
// peripherals" with nothing left to strip, and on its own would join it as a plain adjacency — the
|
|
1214
|
+
// query that matches nothing, which is the defect this carriage exists to prevent. A plan that
|
|
1215
|
+
// states the terms but not the distances does not state what will be asked.
|
|
1216
|
+
const goodsRewritten = [];
|
|
1217
|
+
const goodsCleaned = [];
|
|
1218
|
+
const goodsGaps = [];
|
|
1219
|
+
for (const w of goodsWords) {
|
|
1220
|
+
const { cleaned, removed, gaps } = stripGoodsReservedWords(w);
|
|
1221
|
+
if (removed.length) goodsRewritten.push({ term: String(w).trim(), removed, asked: cleaned || null });
|
|
1222
|
+
if (cleaned) { goodsCleaned.push(cleaned); goodsGaps.push(gaps); }
|
|
1223
|
+
}
|
|
1224
|
+
/** the gap list for each SENDABLE term, in the same order — the plan's statement of the distances. */
|
|
1225
|
+
const gapsFor = (terms) => terms.map((t) => goodsGaps[goodsCleaned.indexOf(t)] ?? []);
|
|
1226
|
+
const goodsSendable = caps?.goodsTextMultiWord
|
|
1227
|
+
? goodsCleaned
|
|
1228
|
+
: goodsCleaned.filter((w) => !/\s/.test(String(w).trim()));
|
|
1229
|
+
// WHAT WAS NOT ASKED, in one list, whichever way it came to be dropped: a whole term this register
|
|
1230
|
+
// cannot express, and a word removed from inside a term because the register reads it as an
|
|
1231
|
+
// operator. Both are the same fact to a reader — the model wrote it and the search did not carry it
|
|
1232
|
+
// — and both make the narrowing ask for LESS than the list it was given, never more.
|
|
1233
|
+
const goodsOmitted = [
|
|
1234
|
+
...goodsCleaned.filter((w) => !goodsSendable.includes(w)),
|
|
1235
|
+
...goodsRewritten.flatMap((r) => r.removed),
|
|
1236
|
+
];
|
|
1237
|
+
const goodsOmittedReason = goodsOmitted.length
|
|
1238
|
+
? `${goodsOmitted.length} of what the goods list asked for did not reach the register (${caps?.id ?? "unknown"}): `
|
|
1239
|
+
+ `a term it cannot express, or a word inside a term that it reads as a search operator and has no `
|
|
1240
|
+
+ `escape syntax for. What remained WAS asked — a term keeps its other words, and the adjacency is `
|
|
1241
|
+
+ `widened by the gap so the phrase still matches — and the class-wide sweep ran beside it. So the `
|
|
1242
|
+
+ `narrowing asked for LESS than the list it was given, never more. A disclosed gap, never a clean `
|
|
1243
|
+
+ `negative about those goods.`
|
|
1244
|
+
: null;
|
|
1245
|
+
const omittedStamp = goodsOmitted.length
|
|
1246
|
+
? { goods_text_omitted: goodsOmitted, goods_text_omitted_reason: goodsOmittedReason }
|
|
1247
|
+
: {};
|
|
1248
|
+
|
|
1249
|
+
if (goodsSendable.length && caps?.goodsTextSearch === true) {
|
|
1250
|
+
if (caps.goodsTextListOr === true) {
|
|
1251
|
+
// The register offers the list as alternatives in one clause: one question, one count — phrases
|
|
1252
|
+
// among them. A phrase compiles to an adjacency and the list to an OR, so the value reads
|
|
1253
|
+
// `a ADJ b OR c`, and that precedence is MEASURED rather than assumed: the mixed clause answers
|
|
1254
|
+
// the union of its alternatives, not the distributed reading `a ADJ (b OR c)`. Were it the
|
|
1255
|
+
// other way the clause would ask a different question and still answer 200.
|
|
1256
|
+
push({ axis: "primary-sweep", predicate: "default", term: manifest.dominant_element,
|
|
1257
|
+
expected_kind: "enumerate", provenance: "mark", goods_text: goodsSendable,
|
|
1258
|
+
goods_text_gaps: gapsFor(goodsSendable), qidSuffix: "+goods", ...omittedStamp });
|
|
1259
|
+
} else {
|
|
1260
|
+
// ── ONE ENTRY PER WORD, where the register has no OR on this field ──────────────────────────
|
|
1261
|
+
//
|
|
1262
|
+
// The alternative shapes were all worse. Joining the words into one value INTERSECTS them
|
|
1263
|
+
// there, so the answer narrows as the list grows and still returns 200 — a false clean that
|
|
1264
|
+
// gets quieter the more thorough the word list is. Merging the calls inside the connector is
|
|
1265
|
+
// not available either: the enumerate kernel drives its own paging and tests the ceiling off
|
|
1266
|
+
// each page's own total, so a fan-out below it would have to invent a total and a page
|
|
1267
|
+
// sequence across several independent streams.
|
|
1268
|
+
//
|
|
1269
|
+
// As separate PLAN entries each word is an ordinary dictated question: its own qid, its own
|
|
1270
|
+
// count, its own ledger row, its own crowd descriptor if it crowds, and the band merge folds
|
|
1271
|
+
// the records by id exactly as it already does across axes. Nothing new has to be trusted.
|
|
1272
|
+
//
|
|
1273
|
+
// The cost is real and it is the reason this is a ruling and not a default: a list of N words
|
|
1274
|
+
// is N questions on such a register, where a register with an OR asks one. The manifest's
|
|
1275
|
+
// 24-word ceiling is what bounds it.
|
|
1276
|
+
goodsSendable.forEach((word, i) => {
|
|
1277
|
+
push({ axis: "primary-sweep", predicate: "default", term: manifest.dominant_element,
|
|
1278
|
+
expected_kind: "enumerate", provenance: "mark", goods_text: [word],
|
|
1279
|
+
goods_text_gaps: gapsFor([word]),
|
|
1280
|
+
qidSuffix: `+goods-${termIdentity(word)}`, ...(i === 0 ? omittedStamp : {}) });
|
|
1281
|
+
});
|
|
1282
|
+
}
|
|
1283
|
+
}
|
|
1109
1284
|
// — EVERY seeded band, not just the dominant element's (see bandsFor). The wildcard fringe stays
|
|
1110
1285
|
// on the dominant band alone: it is crowd-gated on the dominant's own contains parent, and no other
|
|
1111
1286
|
// seed has one to gate against. The oracle does not ask for more — coverageGaps' family arm counts ANY
|
|
@@ -1168,55 +1343,29 @@ export function compileRegisterPlan({ manifest, job, form = null, skillVersion =
|
|
|
1168
1343
|
nice_classes: manifest.incumbent_classes.map(String), provenance: "model", qidSuffix: "+incumbent" });
|
|
1169
1344
|
}
|
|
1170
1345
|
|
|
1171
|
-
// ──
|
|
1172
|
-
//
|
|
1173
|
-
//
|
|
1174
|
-
//
|
|
1175
|
-
//
|
|
1176
|
-
//
|
|
1177
|
-
//
|
|
1178
|
-
//
|
|
1179
|
-
//
|
|
1180
|
-
//
|
|
1181
|
-
//
|
|
1182
|
-
//
|
|
1183
|
-
//
|
|
1184
|
-
//
|
|
1185
|
-
//
|
|
1186
|
-
//
|
|
1187
|
-
//
|
|
1188
|
-
//
|
|
1189
|
-
//
|
|
1190
|
-
//
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
// Dedup key is SCRIPT-PRESERVING (formKey, not norm): norm() strips everything outside [a-z0-9],
|
|
1195
|
-
// so a Chinese/Cyrillic dominant or distinctive element would key to "" and be silently DROPPED
|
|
1196
|
-
// from every owner's slices — undisclosed recall narrowing on the owner lane only (the primary
|
|
1197
|
-
// sweep still searches them), and a fully non-Latin manifest would compile zero slices. The
|
|
1198
|
-
// pipeline carries 8 transliteration scripts and a Chinese jx lane; non-Latin formatives are a
|
|
1199
|
-
// real shape, not an edge case.
|
|
1200
|
-
for (const t of [manifest.dominant_element, ...manifest.elements.filter((e) => e.kind === "distinctive").map((e) => e.value)]) {
|
|
1201
|
-
const k = formKey(t);
|
|
1202
|
-
if (!k || seenForm.has(k)) continue;
|
|
1203
|
-
seenForm.add(k);
|
|
1204
|
-
formatives.push(t);
|
|
1205
|
-
}
|
|
1206
|
-
for (const owner of manifest.watchlist_owners) {
|
|
1207
|
-
const sliceQids = formatives.map((t) =>
|
|
1208
|
-
push({ axis: "incumbent-class", predicate: "default", term: t, owner, expected_kind: "enumerate", provenance: "model", // — the SAME collapse rode this suffix and the issue did not name it: two different
|
|
1209
|
-
// non-Latin owner names both folded to `+owner-q`, so the incumbent-class rows for two separate
|
|
1210
|
-
// proprietors collided and were told apart by position, exactly as the terms were.
|
|
1211
|
-
qidSuffix: `+owner-${termIdentity(owner)}` }));
|
|
1212
|
-
// No formatives at all (defensive — the manifest model requires a dominant element, but honesty
|
|
1213
|
-
// beats assumption): the bare-owner count compiles WITHOUT covered_by. An empty covered_by array
|
|
1214
|
-
// is a plan the compiler's own parser refuses (register_plan_covered_by_invalid) — a deterministic
|
|
1215
|
-
// freeze kill, the exact class 2becf12 exists to prevent. Absent = honest: no slices exist.
|
|
1216
|
-
push({ axis: "incumbent-class", predicate: "owner", term: owner, expected_kind: "count", provenance: "model",
|
|
1217
|
-
qidSuffix: "+watch", ...(sliceQids.length ? { covered_by: sliceQids } : {}) });
|
|
1218
|
-
}
|
|
1219
|
-
}
|
|
1346
|
+
// ── THE GUESSED OWNER LANE IS GONE (2026-09-20) ────────────────────────────────────────────────
|
|
1347
|
+
//
|
|
1348
|
+
// What stood here compiled, for every name on the variants model's `watchlist_owners` list, one
|
|
1349
|
+
// owner × formative slice per formative plus one bare-owner count. On a measured run that was 54 of
|
|
1350
|
+
// the plan's 129 queries — 42% of everything the plan asked — and it returned 54 of the 2,146
|
|
1351
|
+
// records the run read, 2.5%.
|
|
1352
|
+
//
|
|
1353
|
+
// THE MEASUREMENT THAT ENDED IT, and it is not "the axis was quiet". Every conflict the run placed
|
|
1354
|
+
// rested on a record the close-form sweep found: 1,161 record references across every tier of the
|
|
1355
|
+
// report, all of them from the primary sweep. The owner records were not ineligible — all 2,146
|
|
1356
|
+
// records sat in the selection index the placement stage reads — they were simply never chosen.
|
|
1357
|
+
// And none of the eighteen watched owners appeared in a single live in-class close-form record;
|
|
1358
|
+
// four appeared anywhere in the run at all. The list was a guess about who mattered, made before
|
|
1359
|
+
// any record existed to check it against, and the records did not agree with it.
|
|
1360
|
+
//
|
|
1361
|
+
// WHAT WENT WITH IT, stated because it is a real loss and not a free win: 26 of those 54 queries
|
|
1362
|
+
// crowded out and were disclosed as coverage gaps, one of them explicitly as "never a clean
|
|
1363
|
+
// negative about this owner". Those disclosures go too. The judgment behind the change is that a
|
|
1364
|
+
// disclosure about a guessed owner is worth less than the queries it costs, and that owners are
|
|
1365
|
+
// better found by grouping the records the close forms actually return.
|
|
1366
|
+
//
|
|
1367
|
+
// The axis is NOT gone: the incumbent-class anchor above still compiles, so the coverage skeleton
|
|
1368
|
+
// still carries the axis and no clean is ever claimed over an axis that vanished.
|
|
1220
1369
|
|
|
1221
1370
|
// stable ordering: axis (REGISTER_AXES order) then insertion order within the axis
|
|
1222
1371
|
const axisRank = new Map(REGISTER_AXES.map((a, i) => [a, i]));
|
|
@@ -1252,6 +1401,23 @@ export function compileRegisterPlan({ manifest, job, form = null, skillVersion =
|
|
|
1252
1401
|
// DEFERRED coverage row for the ledger, never a dropped filter. Absent on a fully-covered plan, so
|
|
1253
1402
|
// corsearch plans stay byte-identical to the pre-phase-3 compiler.
|
|
1254
1403
|
...(deferredJurisdictions.length ? { deferred_coverage: deferredJurisdictions } : {}),
|
|
1404
|
+
// A GOODS TERM THE COMPILER DROPPED IS A DISCLOSURE, NOT A DETAIL. The words in this list are
|
|
1405
|
+
// OR-ed, so removing one makes the clause NARROWER: the search returns fewer records and a
|
|
1406
|
+
// conflict that term would have surfaced is simply never found. The model was asked for these
|
|
1407
|
+
// words and one of them did not reach the wire — if nothing says so, the run is quietly less
|
|
1408
|
+
// thorough than the list it was given, and it is our own compiler doing the dropping.
|
|
1409
|
+
//
|
|
1410
|
+
// It rides the plan rather than only the entry so a reader looking for what this search did NOT
|
|
1411
|
+
// ask does not have to walk the entries to find out. Absent when nothing was dropped, so every
|
|
1412
|
+
// plan that sends its whole list stays byte-identical.
|
|
1413
|
+
...(goodsOmitted.length ? { goods_text_not_asked: goodsOmitted, goods_text_not_asked_reason: goodsOmittedReason } : {}),
|
|
1414
|
+
// A term the compiler ASKED IN A DIFFERENT SHAPE than it was written. The register reads AND, OR,
|
|
1415
|
+
// NOT, ADJ and NEAR as operators inside the value and has no escape syntax, so the word comes out
|
|
1416
|
+
// and the rest of the item is still asked — "near field communication" goes as "field
|
|
1417
|
+
// communication". That is a narrower ask than the model wrote, and it must be visible: a reader
|
|
1418
|
+
// comparing the word list to what was searched would otherwise find a term that appears to have
|
|
1419
|
+
// been asked and was not, exactly.
|
|
1420
|
+
...(goodsRewritten.length ? { goods_text_rewritten: goodsRewritten } : {}),
|
|
1255
1421
|
...(caps ? { provider: caps.id } : {}),
|
|
1256
1422
|
entries: ordered,
|
|
1257
1423
|
};
|
|
@@ -1341,9 +1507,17 @@ export function extendRegisterPlan(prev, next) {
|
|
|
1341
1507
|
// The key is the question and nothing else — axis, predicate, the term or the OR-stack, and the owner
|
|
1342
1508
|
// that rides the incumbent-class suffix. Not classes or regions: those are run scope, identical across
|
|
1343
1509
|
// one compile, and folding them in would make a rescoped re-run duplicate every entry it already had.
|
|
1510
|
+
//
|
|
1511
|
+
// THE GOODS NARROWING IS PART OF THE QUESTION. The narrowed entry shares axis, predicate, term and
|
|
1512
|
+
// owner with the class-wide parent it sits beside — it differs only by asking what the filings
|
|
1513
|
+
// COVER. Left out of this key the extension reads it as a question already carried and appends
|
|
1514
|
+
// nothing, so a matter whose plan was frozen before the narrowing existed would never gain it: no
|
|
1515
|
+
// deferred row, no disclosure, just an entry never minted. A reader of that run could not tell
|
|
1516
|
+
// "the narrowing was never added" from "the narrowing found nothing".
|
|
1344
1517
|
const questionKey = (e) => [e.axis, e.predicate,
|
|
1345
1518
|
Array.isArray(e.terms) ? `terms:${e.terms.join("\u0000")}` : `term:${e.term ?? ""}`,
|
|
1346
|
-
e.owner ?? ""
|
|
1519
|
+
e.owner ?? "",
|
|
1520
|
+
goodsTermsList(e).join("\u0000")].join("\u0001");
|
|
1347
1521
|
// THE QUESTION ALONE DECIDES WHAT IS NEW, and the qid deliberately does NOT get a vote here. Keeping
|
|
1348
1522
|
// `!have.has(e.qid)` as an additional guard looks conservative and re-creates the whole defect: under
|
|
1349
1523
|
// the old scheme a fresh term and a stored one collide on `q#2` while asking DIFFERENT questions, and
|
|
@@ -115,7 +115,7 @@ band its plan entries call for is missing is refused as `named_band_missing`.
|
|
|
115
115
|
- **`saturation-probe`** — fires when any element is flagged `saturation: high`/`very-high`, plus a partial-phrase structural-prefix probe for slogan/descriptive-compound multi-word marks. Run count-only probes (`limit:1, fields:["uri"]`) capturing `total_hits`, and write **one `incomplete` crowd-descriptor block per probed element** (`fetched:0`, `reason:"crowd descriptor — saturated element <X>, count-only"`). It measures how crowded an element is — it **enumerates nothing and clears nothing**. Skip (not applicable) if all elements are distinctive/low-saturation. **A meaning-translation variant (`translit-*-meaning`) is NOT special-cased into a drop:** an everyday-word-scale count is just a crowd descriptor; whether the field-scoped named slice inside it matters is judgment's call — the funnel surfaces the count, and the **primary-sweep** axis runs the field-scoped named enumeration (below). Record `translit-too-generic` in the descriptor `reason` as a context note (NOT a drop / re-narrow / swap-to-phonetic signal). An implausibly LOW count on a translation that ought to have peers is the separate `translit-underretrieved` context note — also recorded in the descriptor `reason`, never acted on as a drop. Both are signals the lawyer reads.
|
|
116
116
|
- **`primary-sweep`** — always applicable. Run the playbook's named variant queries: the exact mark + each `phrase-substitution`, `visual-substitution`, family-pattern wildcard, and slang variant the manifest lists, **each via `register_enumerate`** (the tool guarantees the page loop). Owns the family-pattern / phrase-substitution / slang / visual-substitution variant queries. **Owns the dead-inclusive named enumeration** (its `register_enumerate` calls carry live AND dead records forward with their status — see step 6; there is no separate "dead-but-identical" or "D4" pass, because the funnel no longer filters dead records). **On a saturated unit, owns the [named-band enumeration of the dangerous category](#exact-in-class-live-floor-primary-sweep-unit-owns-it)** (the exact mark + the **distinctive** formative root / dominant token as the substring-in-scope-class-live named slice, per major+material jurisdiction, plus the phonetic fringe) — enumerated, not sampled. **The named enumeration covers a saturated `translit-*-meaning` meaning token too** (scoped to the filed in-scope Nice classes as the `nice_classes` × `regions` filter on the `register_enumerate` call, NOT goods-words ANDed into the search text) — not only the Latin dominant element / formative root. **But a COMMON component that is NOT the distinctive anchor** (a "stripped" common word the manifest marks common/descriptive — DAWN / LEGENDS / GREAT / OUTDOORS) **is count-only, owned by `saturation-probe`, and is NOT enumerated here**: `saturation-probe` already counts it and hands that count up; primary-sweep does **not** run the substring / per-major / phonetic enumeration on it and does **not** author a second crowd block for it (see the linked section — wholesale-enumerating a common component per-jurisdiction is the mega-crowd grind that SIGKILLs the stage).
|
|
117
117
|
- **`transliteration-numeric`** — fires when the manifest has `translit-*` variants or numeric-substitution variants. For each foreign-transliteration variant `register_enumerate` the transliterated form; for each numeric-substitution variant enumerate its query. Owns the foreign-transliteration + numeric-substitution variant queries. Skip if the manifest classifies the mark English-only / single-language. (Multi-script accumulation is no longer a window risk for the unit — `register_enumerate` owns the page loop and a crowd returns `incomplete`, so the unit does not deep-page raw rows itself.)
|
|
118
|
-
- **`incumbent-class`** — fires when the manifest has an `industry_incumbent_alert`. `register_enumerate` the named band scoped to the **UNION of the incumbent's primary classes AND the matter's in-scope Nice set** (pass that union as `nice_classes`) — **never an all-class incumbent sweep** (that is the unscoped crowd that timed this very axis out live). **Beyond the NAMED watchlist incumbents, the named enumeration is an UNNAMED-OWNER exact-in-class-live enumeration in those classes** — the watchlist seeds prioritise *attention* (judgment's use), they do **not** bound the funnel's enumeration. Every exact-token-in-class-live record crosses the firewall whether or not its owner was pre-listed; the watchlist is provably incomplete, so it never decides what the funnel enumerates. Skip only if there is no incumbent alert AND the frozen plan carries no entries on this axis —
|
|
118
|
+
- **`incumbent-class`** — fires when the manifest has an `industry_incumbent_alert`. `register_enumerate` the named band scoped to the **UNION of the incumbent's primary classes AND the matter's in-scope Nice set** (pass that union as `nice_classes`) — **never an all-class incumbent sweep** (that is the unscoped crowd that timed this very axis out live). **Beyond the NAMED watchlist incumbents, the named enumeration is an UNNAMED-OWNER exact-in-class-live enumeration in those classes** — the watchlist seeds prioritise *attention* (judgment's use), they do **not** bound the funnel's enumeration. Every exact-token-in-class-live record crosses the firewall whether or not its owner was pre-listed; the watchlist is provably incomplete, so it never decides what the funnel enumerates. Skip only if there is no incumbent alert AND the frozen plan carries no entries on this axis — the plan is the search authority. **The plan carries no owner lane from the manifest's watchlist.** Read owners from the records this axis and the primary sweep returned. Propose an owner slice only where an owner's other marks could change the assessment, for example when the records show a family of marks sharing the element. A bare-owner count is never written up as a reviewed portfolio or as "portfolio too large, noted".
|
|
119
119
|
|
|
120
120
|
### Exact-in-class-live floor — RENAMED to: the dangerous-category named enumeration (primary-sweep unit owns it) {#exact-in-class-live-floor-primary-sweep-unit-owns-it}
|
|
121
121
|
|
|
@@ -420,10 +420,16 @@ Watchlists drive cross-pollination triggers and owner-bound register sweeps (Ste
|
|
|
420
420
|
**Structured sibling:** mirror the register-relevant watchlist into the structured model's
|
|
421
421
|
`watchlist_owners` key — `aggressive_enforcers` ∪ `competitors` ∪ the matter frame's watchlist-owner
|
|
422
422
|
seeds (NOT `major_brand_owners` — too broad), client-excluded, at most 24 entries, real register
|
|
423
|
-
owner names only (never a sector or a description). The
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
423
|
+
owner names only (never a sector or a description). The list is recorded on the run and drives the
|
|
424
|
+
cross-pollination triggers. It compiles no register queries of its own: owners are read from the
|
|
425
|
+
records the searches return, not from a list written before any record exists. Omit the key when
|
|
426
|
+
Step 5 names none.
|
|
427
|
+
|
|
428
|
+
**Structured sibling:** **`goods_words`** — the words the register search is narrowed to.
|
|
429
|
+
1. Start with the client's goods and services wording: the order's wording if it has one, otherwise the product description in the company profile.
|
|
430
|
+
2. Add the words other filings use for the same goods that the client's wording does not already contain.
|
|
431
|
+
|
|
432
|
+
Single words or short phrases as a specification would write them, no wildcards, at most 24. Each word is matched against the goods and services description of registered marks, so use words a specification would contain. A word broader than the goods widens the search instead of narrowing it. The register cannot read the words and, or, not, adj or near inside an item. An item containing one is searched without it. Omit the key when the matter has no goods wording.
|
|
427
433
|
|
|
428
434
|
### Step 6 — Cross-mark themes (once, after all marks processed)
|
|
429
435
|
|
package/driver/stages.mjs
CHANGED
|
@@ -676,7 +676,7 @@ export const SUPPLEMENTAL_LANE_STEERING = "register_enumerate is NOT available t
|
|
|
676
676
|
// query, owner × term × class, on every provider whose contract declares ownerTermIntersection
|
|
677
677
|
// (a provider without it defers the slice loudly; it is never silently widened to mark-only).
|
|
678
678
|
// predicate:"owner" remains the BARE portfolio sweep (the term IS the owner name) — count context.
|
|
679
|
-
export const OWNER_SWEEP_STEERING = `OWNER / WATCHLIST COVERAGE on this lane: the owner×term slice is THE coverage instrument for a named owner — a mark-text proposal carrying the owner as a scope field, e.g. {predicate:"exact"|"default", term:"<the mark/element>", owner:"<the owner>", nice_classes:[…]}. That intersects the owner's portfolio with the dangerous band in ONE query and ENUMERATES record-by-record even where the bare portfolio is a many-thousand crowd — so a watchlist owner is answered by records, not by a number. The
|
|
679
|
+
export const OWNER_SWEEP_STEERING = `OWNER / WATCHLIST COVERAGE on this lane: the owner×term slice is THE coverage instrument for a named owner — a mark-text proposal carrying the owner as a scope field, e.g. {predicate:"exact"|"default", term:"<the mark/element>", owner:"<the owner>", nice_classes:[…]}. That intersects the owner's portfolio with the dangerous band in ONE query and ENUMERATES record-by-record even where the bare portfolio is a many-thousand crowd — so a watchlist owner is answered by records, not by a number. The plan does not search a guessed owner list. Owners come from the records the searches returned. Propose an owner slice only where an owner's other marks could change the assessment, for example when the records show a family of marks sharing the element. A bare predicate:"owner" proposal (the owner name as the term; nice_classes REQUIRED — an all-class owner sweep is refused) is CROWD CONTEXT, not coverage: an honest "incomplete" COUNT descriptor that sizes the portfolio and whose write-up POINTS AT the owner×term slice qids that actually cover it — never "portfolio too large, noted" as an ending in itself. register_search {owner, name} stays what it is: a COUNT-ONLY context probe (limit:1), a count, never a review pass. And you NEVER stand sampled register_search pages in for either — not for a slice, not for a sweep — nor write such a sample up as though the owner had been screened: ten of an owner's 432 hits presented as review is a recall hole with a clean face, not a search. "Portfolio too large, noted" is never a finding: size the crowd, then cover it with owner×term slices.`;
|
|
680
680
|
|
|
681
681
|
// PR-8 (reading layer) — the ONE band-reading contract every band-consuming judgment stage receives.
|
|
682
682
|
// The raw merged band is NOT named to the model any more (it stays a declared stage input for
|
package/driver/suite-census.json
CHANGED
|
@@ -201,6 +201,12 @@
|
|
|
201
201
|
"skips": 0,
|
|
202
202
|
"todos": 0
|
|
203
203
|
},
|
|
204
|
+
"a-connect-line-keeps-every-path-whole.test.mjs": {
|
|
205
|
+
"tests": 8,
|
|
206
|
+
"asserts": 20,
|
|
207
|
+
"skips": 0,
|
|
208
|
+
"todos": 0
|
|
209
|
+
},
|
|
204
210
|
"a-connect-line-says-where-it-runs.test.mjs": {
|
|
205
211
|
"tests": 11,
|
|
206
212
|
"asserts": 56,
|
|
@@ -225,6 +231,12 @@
|
|
|
225
231
|
"skips": 0,
|
|
226
232
|
"todos": 0
|
|
227
233
|
},
|
|
234
|
+
"a-crowded-sweep-is-narrowed-by-what-a-filing-covers.test.mjs": {
|
|
235
|
+
"tests": 22,
|
|
236
|
+
"asserts": 109,
|
|
237
|
+
"skips": 0,
|
|
238
|
+
"todos": 0
|
|
239
|
+
},
|
|
228
240
|
"a-crowded-zone-is-on-the-list-it-could-not-fill.test.mjs": {
|
|
229
241
|
"tests": 10,
|
|
230
242
|
"asserts": 23,
|
|
@@ -249,6 +261,12 @@
|
|
|
249
261
|
"skips": 0,
|
|
250
262
|
"todos": 0
|
|
251
263
|
},
|
|
264
|
+
"a-delivered-blocking-run-keeps-its-grounds-off-the-clients-conditions.test.mjs": {
|
|
265
|
+
"tests": 4,
|
|
266
|
+
"asserts": 9,
|
|
267
|
+
"skips": 0,
|
|
268
|
+
"todos": 0
|
|
269
|
+
},
|
|
252
270
|
"a-delivered-finding-is-not-reported-as-reaching-nothing.test.mjs": {
|
|
253
271
|
"tests": 4,
|
|
254
272
|
"asserts": 10,
|
|
@@ -1137,6 +1155,12 @@
|
|
|
1137
1155
|
"skips": 0,
|
|
1138
1156
|
"todos": 0
|
|
1139
1157
|
},
|
|
1158
|
+
"an-answer-shows-one-line-and-folds-the-rest.test.mjs": {
|
|
1159
|
+
"tests": 3,
|
|
1160
|
+
"asserts": 19,
|
|
1161
|
+
"skips": 0,
|
|
1162
|
+
"todos": 0
|
|
1163
|
+
},
|
|
1140
1164
|
"an-archived-run-says-whether-it-was-delivered.test.mjs": {
|
|
1141
1165
|
"tests": 8,
|
|
1142
1166
|
"asserts": 42,
|
|
@@ -3003,6 +3027,12 @@
|
|
|
3003
3027
|
"skips": 0,
|
|
3004
3028
|
"todos": 0
|
|
3005
3029
|
},
|
|
3030
|
+
"no-client-surface-speaks-the-delivery-gates-word.test.mjs": {
|
|
3031
|
+
"tests": 4,
|
|
3032
|
+
"asserts": 17,
|
|
3033
|
+
"skips": 0,
|
|
3034
|
+
"todos": 0
|
|
3035
|
+
},
|
|
3006
3036
|
"no-credential-rides-the-supplemental-mint.test.mjs": {
|
|
3007
3037
|
"tests": 6,
|
|
3008
3038
|
"asserts": 13,
|
|
@@ -3839,7 +3869,7 @@
|
|
|
3839
3869
|
},
|
|
3840
3870
|
"register-steering.test.mjs": {
|
|
3841
3871
|
"tests": 9,
|
|
3842
|
-
"asserts":
|
|
3872
|
+
"asserts": 70,
|
|
3843
3873
|
"skips": 0,
|
|
3844
3874
|
"todos": 0
|
|
3845
3875
|
},
|
|
@@ -4869,6 +4899,12 @@
|
|
|
4869
4899
|
"skips": 0,
|
|
4870
4900
|
"todos": 0
|
|
4871
4901
|
},
|
|
4902
|
+
"the-demo-says-what-it-is-first.test.mjs": {
|
|
4903
|
+
"tests": 3,
|
|
4904
|
+
"asserts": 16,
|
|
4905
|
+
"skips": 0,
|
|
4906
|
+
"todos": 0
|
|
4907
|
+
},
|
|
4872
4908
|
"the-demo-takes-its-folder-with-it.test.mjs": {
|
|
4873
4909
|
"tests": 6,
|
|
4874
4910
|
"asserts": 12,
|
|
@@ -5051,7 +5087,7 @@
|
|
|
5051
5087
|
},
|
|
5052
5088
|
"the-knockout-page-leads-with-the-read.test.mjs": {
|
|
5053
5089
|
"tests": 39,
|
|
5054
|
-
"asserts":
|
|
5090
|
+
"asserts": 125,
|
|
5055
5091
|
"skips": 0,
|
|
5056
5092
|
"todos": 0
|
|
5057
5093
|
},
|
|
@@ -5854,7 +5890,7 @@
|
|
|
5854
5890
|
},
|
|
5855
5891
|
"account-audit-chain.test.mjs": {
|
|
5856
5892
|
"tests": 10,
|
|
5857
|
-
"asserts":
|
|
5893
|
+
"asserts": 48,
|
|
5858
5894
|
"skips": 0,
|
|
5859
5895
|
"todos": 0
|
|
5860
5896
|
},
|
|
@@ -6565,6 +6601,12 @@
|
|
|
6565
6601
|
"skips": 0,
|
|
6566
6602
|
"todos": 0
|
|
6567
6603
|
},
|
|
6604
|
+
"theWaitIsNamed.test.ts": {
|
|
6605
|
+
"tests": 5,
|
|
6606
|
+
"asserts": 11,
|
|
6607
|
+
"skips": 0,
|
|
6608
|
+
"todos": 0
|
|
6609
|
+
},
|
|
6568
6610
|
"tone.test.ts": {
|
|
6569
6611
|
"tests": 7,
|
|
6570
6612
|
"asserts": 31,
|
|
@@ -45,13 +45,16 @@ export const VARIANT_CATEGORIES = ["core", "phonetic", "visual", "transliteratio
|
|
|
45
45
|
"exact-phrase", "exact-element", "plural-root", "formative-family"];
|
|
46
46
|
|
|
47
47
|
const short = (v) => String(v ?? "").replace(/\s+/g, " ").trim().slice(0, 40);
|
|
48
|
-
const MODEL_KEYS = ["schema_version", "mark", "dominant_element", "elements", "variants", "incumbent_classes", "watchlist_owners", "search_floor"];
|
|
48
|
+
const MODEL_KEYS = ["schema_version", "mark", "dominant_element", "elements", "variants", "incumbent_classes", "watchlist_owners", "goods_words", "search_floor"];
|
|
49
49
|
// The watchlist-owner seeds the register plan's OWNER LANE compiles from (F2, 2026-07-29): the
|
|
50
50
|
// Step-5 watchlists a run must COVER on the register (aggressive_enforcers ∪ competitors ∪ the
|
|
51
51
|
// matter-context watchlist-owner seeds; never the client). Bounded so a runaway list fails the
|
|
52
52
|
// stage loudly (the corrective ladder trims it) instead of compiling an unbounded query fan-out —
|
|
53
53
|
// the same 24 the supplemental lane caps an axis at; the postmortem run's fifteen owners fit with room.
|
|
54
54
|
export const WATCHLIST_OWNERS_MAX = 24;
|
|
55
|
+
// The goods-words bound. Wider than this is a list that has stopped describing these goods: every word
|
|
56
|
+
// is OR-ed on the wire, so an over-long list restores the very crowd the narrowing exists to cut.
|
|
57
|
+
export const GOODS_WORDS_MAX = 24;
|
|
55
58
|
const ELEMENT_KEYS = ["value", "kind"];
|
|
56
59
|
// `romanization` — the Latin-script form of a NON-LATIN `value`, and the whole reason the
|
|
57
60
|
// transliteration axis can be executed at all (see the doc block on variantRomanizationGaps below).
|
|
@@ -158,6 +161,40 @@ export function parseVariantManifestModel(raw) {
|
|
|
158
161
|
throw new Error(`variantmodel_watchlist_owners_invalid (${watchlist_owners.length} owners exceed the ${WATCHLIST_OWNERS_MAX}-owner bound — keep the NAMED watchlists only: aggressive enforcers, competitors, matter-context seeds)`);
|
|
159
162
|
}
|
|
160
163
|
|
|
164
|
+
// ── THE GOODS WORDS THE REGISTER SEARCH IS NARROWED TO ─────────────────────────────────────────
|
|
165
|
+
//
|
|
166
|
+
// The client's own goods and services wording plus the words another filing would use for the same
|
|
167
|
+
// goods. They are AND-ed with the mark on the register, so THIS LIST SETS WHAT THE NARROWED SWEEP
|
|
168
|
+
// CAN FIND: a word missing from it is a conflict the run will not see.
|
|
169
|
+
//
|
|
170
|
+
// NO WILDCARD, REFUSED HERE RATHER THAN ON THE WIRE: a mid-word wildcard is a hard 400 on the field
|
|
171
|
+
// the provider production runs on, and a value that cannot be sent is cheapest to catch at the door
|
|
172
|
+
// where the reason can name the word.
|
|
173
|
+
//
|
|
174
|
+
// A SHORT PHRASE IS ALLOWED, and what happens to it is the connector's business, not this parser's:
|
|
175
|
+
// one register matches it as an ordered phrase, another intersects its words, a third is unmeasured
|
|
176
|
+
// and refuses it. Each says so in its own capability contract, and the compiler declines to build an
|
|
177
|
+
// entry a register cannot express. Deciding it here would freeze one register's behaviour into a
|
|
178
|
+
// rule about every register.
|
|
179
|
+
let goods_words = [];
|
|
180
|
+
if (m.goods_words != null) {
|
|
181
|
+
if (!Array.isArray(m.goods_words) || !m.goods_words.every((w) => typeof w === "string"))
|
|
182
|
+
throw new Error("variantmodel_goods_words_invalid (an array of single-word strings, or omitted)");
|
|
183
|
+
const seen = new Set();
|
|
184
|
+
for (const raw of m.goods_words) {
|
|
185
|
+
const w = String(raw).replace(/\s+/g, " ").trim();
|
|
186
|
+
if (!w) throw new Error("variantmodel_goods_words_invalid (every goods word is a non-empty string)");
|
|
187
|
+
if (/[*?]/.test(w))
|
|
188
|
+
throw new Error(`variantmodel_goods_words_invalid ("${w}" carries a wildcard — a mid-word wildcard is refused by the register on this field)`);
|
|
189
|
+
const key = w.toLowerCase();
|
|
190
|
+
if (seen.has(key)) continue;
|
|
191
|
+
seen.add(key);
|
|
192
|
+
goods_words.push(w);
|
|
193
|
+
}
|
|
194
|
+
if (goods_words.length > GOODS_WORDS_MAX)
|
|
195
|
+
throw new Error(`variantmodel_goods_words_invalid (${goods_words.length} words exceed the ${GOODS_WORDS_MAX}-word bound — keep the client's own wording and the words a filing would use for the same goods)`);
|
|
196
|
+
}
|
|
197
|
+
|
|
161
198
|
// ── — THE SEARCH FLOOR, AS A TYPED DESIGNATION ────────────────────────────────────────────────
|
|
162
199
|
//
|
|
163
200
|
// The axes this mark's search floor obliges: work a `coverage-limited` row may not demote. It replaces
|
|
@@ -202,6 +239,7 @@ export function parseVariantManifestModel(raw) {
|
|
|
202
239
|
variants: outVariants,
|
|
203
240
|
incumbent_classes,
|
|
204
241
|
watchlist_owners,
|
|
242
|
+
goods_words,
|
|
205
243
|
search_floor,
|
|
206
244
|
};
|
|
207
245
|
}
|