clearotron 0.4.0-beta.2 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +5 -5
- package/build-info.json +2 -2
- package/driver/CHANGELOG.md +90 -0
- package/driver/ask-ledger.mjs +2 -2
- package/driver/coverage-form.mjs +6 -0
- package/driver/coverage-ledger.mjs +14 -2
- package/driver/driver.config.mjs +24 -7
- package/driver/engine/mcp/band-server.mjs +3 -3
- package/driver/engine/mcp/stdio-server.mjs +12 -1
- package/driver/gateway.mjs +27 -3
- package/driver/package.json +1 -1
- package/driver/pipeline-knockout.mjs +41 -10
- package/driver/pipeline.mjs +104 -12
- package/driver/progress.mjs +6 -1
- package/driver/provider-usage.mjs +16 -0
- package/driver/publish/index.mjs +11 -5
- package/driver/publish/knockout.mjs +16 -3
- package/driver/publish/render-knockout.mjs +38 -11
- package/driver/record-carry.mjs +17 -1
- package/driver/reference-strip-signatures.mjs +11 -1
- package/driver/register-plan.mjs +7 -3
- package/driver/register-served.mjs +91 -0
- package/driver/remedy-accounting.mjs +38 -9
- package/driver/reviewer-open-points.mjs +20 -23
- package/driver/score-redaction.mjs +416 -20
- package/driver/screen-gate.mjs +3 -3
- package/driver/skills/clearance-register/SKILL.md +0 -8
- package/driver/skills/clearance-register/digest.md +1 -1
- package/driver/skills/clearance-register/providers/corsearch.md +1 -1
- package/driver/skills/clearance-register/register-recipes.md +3 -9
- package/driver/skills/clearance-search/SKILL.md +2 -2
- package/driver/skills/clearance-search/phase2-execution.md +1 -1
- package/driver/skills/knockout-assess/SKILL.md +2 -0
- package/driver/stages-knockout.mjs +22 -0
- package/driver/suite-census.json +152 -26
- package/driver/unit-inventory.mjs +38 -43
- package/mcp-server/CHANGELOG.md +8 -0
- package/mcp-server/package.json +1 -1
- package/node_modules/brace-expansion/index.js +78 -22
- package/node_modules/brace-expansion/package.json +1 -1
- package/node_modules/readdir-glob/node_modules/brace-expansion/index.js +78 -22
- package/node_modules/readdir-glob/node_modules/brace-expansion/package.json +1 -1
- package/package.json +1 -1
- package/portal-ui/package.json +2 -2
- package/providers/_shared/term-shape.mjs +1 -1
- package/providers/jx/src/core.js +0 -1
- package/providers/jx/src/judge.js +0 -1
- package/providers/jx/src/nativeread.js +0 -1
- package/providers/oauth-mcp-bridge/CHANGELOG.md +8 -0
- package/providers/oauth-mcp-bridge/package.json +1 -1
- package/providers/signa/src/core.js +125 -17
- package/scripts/e2e-first-time.mjs +12 -4
- package/scripts/e2e-scenario-ops.mjs +25 -2
- package/scripts/e2e.mjs +246 -24
- package/scripts/mint-reference-strip-backlog.mjs +36 -2
- package/scripts/score.mjs +102 -29
- package/shared/identifier-scan.mjs +31 -1
package/scripts/score.mjs
CHANGED
|
@@ -64,7 +64,7 @@ import { recordQids } from "../driver/named-band.mjs";
|
|
|
64
64
|
import { previousRunDir, scenarioRefs } from "../driver/e2e-rounds.mjs";
|
|
65
65
|
// The names rule, kept out of this file so it is testable without a run directory — the same reason
|
|
66
66
|
// reference-score.mjs holds the scoring rules rather than this script.
|
|
67
|
-
import { protectedStrings, redactor, installRedaction, unclassifiedNotice, REDACTION_NOTICE } from "../driver/score-redaction.mjs";
|
|
67
|
+
import { protectedStrings, redactor, authoredRedactor, printAuthored, printPreRedacted, redactDeep, installRedaction, unclassifiedNotice, REDACTION_NOTICE } from "../driver/score-redaction.mjs";
|
|
68
68
|
import { readSettleStamp } from "../driver/settle-stamp.mjs";
|
|
69
69
|
import { envFrom } from "../shared/env-aliases.mjs"; // — resolves EITHER spelling; names the retired one because that is the live-writable half
|
|
70
70
|
|
|
@@ -517,8 +517,20 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
517
517
|
const B = s.buckets;
|
|
518
518
|
const scored = B.found.length + B.withheld.length + B.lost.length;
|
|
519
519
|
|
|
520
|
-
|
|
520
|
+
// A RULE IS A RULE. Nothing in it came from anywhere; routed so a party's ordinary long word cannot
|
|
521
|
+
// rewrite the page's own furniture.
|
|
522
|
+
printAuthored(`\n${"═".repeat(78)}`);
|
|
521
523
|
console.log(`${id} — ${ref.mark ?? "(mark unnamed in the reference)"}`);
|
|
524
|
+
// NOT ROUTED, AND THIS IS THE ONE THAT PROVED WHY. A path looks authored — the config repository's
|
|
525
|
+
// name and the working directory's name are furniture, and the derived layer was tokenising them, so
|
|
526
|
+
// a reader was shown a path they could not use. Routing it put a real client name on the page on the
|
|
527
|
+
// first run measured after the change: a run directory is NAMED AFTER THE MATTER, and that segment is
|
|
528
|
+
// a distinctive word of a party rather than a whole name, so it is exactly what the derived layer
|
|
529
|
+
// catches and exactly what the authored path drops.
|
|
530
|
+
//
|
|
531
|
+
// A path is not authored text. It is furniture with a data-derived segment in the middle of it, and
|
|
532
|
+
// the two cannot be separated by choosing an instrument for the whole line. Both stay on the full
|
|
533
|
+
// layer until the run directories themselves stop carrying matter.
|
|
522
534
|
console.log(`reference: ${refPath}`);
|
|
523
535
|
console.log(` ${ref.source}`);
|
|
524
536
|
console.log(`run: ${run.dir}`);
|
|
@@ -534,7 +546,9 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
534
546
|
// 28). The tool already had the honest pattern four lines down — `withheld` names what it could not
|
|
535
547
|
// read and declines to answer — and the delivery line invented a verdict from the same kind of
|
|
536
548
|
// absence. This is that asymmetry closed, in the direction of the honest half.
|
|
537
|
-
|
|
549
|
+
// `deliveryLine` composes authored words, a state token and a timestamp, read at its producer in
|
|
550
|
+
// reference-score.mjs — no part of it comes from the matter.
|
|
551
|
+
printAuthored(deliveryLine(run));
|
|
538
552
|
// — THE INSTRUMENT, BESIDE THE NUMBER. `--json` has carried `scorer_version` since this file
|
|
539
553
|
// shipped; the human path did not, and the human path is the one whose numbers get pasted into an
|
|
540
554
|
// issue. That body states 6/9 for a run that re-scores 5/2/2 today across two scorer changes
|
|
@@ -547,13 +561,17 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
547
561
|
console.log(`engine: ${run.engineCommit
|
|
548
562
|
? `${run.engineCommit.slice(0, 12)}${run.engineCommitFrom === "meta.json" ? " (from the pool's meta.json — this dir has no status.json)" : ""}`
|
|
549
563
|
: "(not recorded — no status.json engine stamp and no pool meta.json carrying one)"}`);
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
564
|
+
printAuthored(`scorer: v${SCORER_VERSION} (a score with no version predates this stamp and is not comparable to one)`);
|
|
565
|
+
printAuthored(`lane: ${run.lane}${run.hasDriver ? "" : " (no _driver/ — pool dir, not a workspace archive)"}`);
|
|
566
|
+
// Class numbers and territory codes. Neither is a name, and both were being broken up by the derived
|
|
567
|
+
// layer on references whose parties carry an ordinary long word.
|
|
568
|
+
printAuthored(`scope: classes ${run.scopeClasses.join(", ") || "(none recorded)"} territories ${run.scopeTerritories.join(", ") || "(none recorded)"}`);
|
|
553
569
|
// — WHICH SUBJECT MARKS THIS REFERENCE ANSWERS, on the line beside the classes and territories it
|
|
554
570
|
// already scopes by, because it is the same kind of fact. Never silent: a gold set that declares
|
|
555
571
|
// nothing says NOT DECLARED and names the marks the run searched.
|
|
556
|
-
|
|
572
|
+
// `coverage.why` is the SCORER's own sentence about what it could measure, not a sentence about
|
|
573
|
+
// anybody — the same reading that put `coverage` in the safe key set.
|
|
574
|
+
printAuthored(`coverage: ${s.coverage.state} — ${s.coverage.why}`);
|
|
557
575
|
// Never a bare "(unreadable)". An unread verdict is an absence, and an absence that prints as an empty
|
|
558
576
|
// parenthesis is the one a reader skims past — so it states the reason, and where a reading DID come
|
|
559
577
|
// from it names the artifact, because the two lanes answer this from different files.
|
|
@@ -616,7 +634,9 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
616
634
|
console.log(` entries are unreachable by construction. The counts axis above is this scenario's score.`);
|
|
617
635
|
console.log(` What follows folds over the run's OWN findings only.\n`);
|
|
618
636
|
}
|
|
619
|
-
|
|
637
|
+
// CLASS: THE TABLE'S OWN COLUMNS. "marks" is this tool's word for its own header, and on a reference
|
|
638
|
+
// whose parties carry it as a distinctive word the header read `of the «name N» the lawyer named`.
|
|
639
|
+
printAuthored(row("bucket", "n", "of the marks the lawyer named"));
|
|
620
640
|
console.log(row("found", B.found.length, `${pct(B.found.length, scored)} of ${scored} in-scope reference marks`));
|
|
621
641
|
if (s.registerOnly) {
|
|
622
642
|
// ONE source for the reason, so the summary and the rows can never name different causes. Missing
|
|
@@ -745,11 +765,11 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
745
765
|
for (const note of M.notes) console.log(` ${note}`);
|
|
746
766
|
|
|
747
767
|
console.log(`\n── axis B · field ${"─".repeat(59)}`);
|
|
748
|
-
if (!s.field.length)
|
|
768
|
+
if (!s.field.length) printAuthored(" the reference flags no entry as on-field — not scored, not passed");
|
|
749
769
|
for (const f of s.field) console.log(` ${String(f.state).padEnd(12)} ${f.mark} — ${f.detail}`);
|
|
750
770
|
|
|
751
771
|
console.log(`\n── axis C · sources ${"─".repeat(57)}`);
|
|
752
|
-
if (!s.sources.length)
|
|
772
|
+
if (!s.sources.length) printAuthored(" the reference names no channels — not scored, not passed");
|
|
753
773
|
for (const c of s.sources) console.log(` ${(c.searched ? "searched" : "ABSENT").padEnd(12)} ${c.channel}`);
|
|
754
774
|
|
|
755
775
|
console.log(`\n── axis D · gap discipline ${"─".repeat(50)}`);
|
|
@@ -761,9 +781,13 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
761
781
|
// instructed territory, or does the lane thin out as the count rises. Read `sub-query` first, then
|
|
762
782
|
// `returned` — a sub-query that ran and came back over the provider's ceiling is not depth holding.
|
|
763
783
|
const T = s.territories;
|
|
764
|
-
|
|
784
|
+
// AUTHORED, not data: this heading and the column ruler below it carry the words "territory" and
|
|
785
|
+
// "depth", which a party name can put into the protected set. Measured before this change: a reference
|
|
786
|
+
// whose proprietor was "Depth Charge" printed `per-territory «name 2»`, and one called "Territory
|
|
787
|
+
// Holdings" printed `per-«name 1» depth` — a reader cannot tell a redaction from a word.
|
|
788
|
+
printAuthored(`\n── axis E · per-territory depth ${"─".repeat(45)}`);
|
|
765
789
|
if (!T.subQueriesResolved) console.log(` sub-queries NOT MEASURABLE — ${T.why}`);
|
|
766
|
-
|
|
790
|
+
printAuthored(` ${"territory".padEnd(10)} ${"sub-query".padEnd(22)} ${"own".padEnd(4)} ${"grouped".padEnd(8)} ${"returned".padEnd(9)} ${"recall".padEnd(7)} reference entries`);
|
|
767
791
|
for (const r of T.rows) {
|
|
768
792
|
// `—` for recall is deliberate and is the whole design point of this row: a territory whose
|
|
769
793
|
// reference carries nothing to score prints NEITHER 0% (which reads as total failure) NOR 100%
|
|
@@ -899,15 +923,18 @@ function print(id, ref, run, s, delta, refPath) {
|
|
|
899
923
|
else for (const d of delta) console.log(` ${d.mark}: ${d.from} → ${d.to}`);
|
|
900
924
|
|
|
901
925
|
console.log(`\n${"═".repeat(78)}`);
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
926
|
+
// CLASS: THE CLOSING LEGEND. Eight lines, every one a literal of this file's own with nothing
|
|
927
|
+
// interpolated from a reference or a run — the class most obviously ours and the least defensible to
|
|
928
|
+
// leave on the data path, because a party's ordinary long word rewrites the page's own instructions.
|
|
929
|
+
printAuthored(`This is a measurement, not a verdict. There is no PASS here and the exit code is always 0.`);
|
|
930
|
+
printAuthored(`Reproducing the reference proves nothing — it is a regression tripwire, never a target.`);
|
|
931
|
+
printAuthored(`What to read: every WITHHELD row is a gather-to-judgment seam defect, not a recall one.`);
|
|
932
|
+
printAuthored(`Axis E: "own" counts sub-queries naming ONE territory and nothing else — the deep-dive itself.`);
|
|
933
|
+
printAuthored(`A territory with no reference entry prints "—", never 0% and never 100%. Both are conclusions.`);
|
|
907
934
|
// The instrument changed. A round comparing its noise against a round scored before `uncovered`
|
|
908
935
|
// existed is comparing two different measurements, and the drop will otherwise read as an improvement.
|
|
909
|
-
|
|
910
|
-
|
|
936
|
+
printAuthored(`"uncovered" is a finding of a mark this reference does not answer — a noise count from before`);
|
|
937
|
+
printAuthored(`that bucket existed is not comparable with one after it. The gold set must declare covers_marks.\n`);
|
|
911
938
|
}
|
|
912
939
|
|
|
913
940
|
// ── main ─────────────────────────────────────────────────────────────────────────────────────────────
|
|
@@ -977,17 +1004,40 @@ const delta = prev ? bucketDelta(scored.buckets, prev.buckets) : null;
|
|
|
977
1004
|
// that adds a print. Here `renderCarryThrough`'s lines and anything added after this was written are
|
|
978
1005
|
// covered without anyone remembering to.
|
|
979
1006
|
//
|
|
980
|
-
//
|
|
981
|
-
// lawyer never named is still somebody's name
|
|
1007
|
+
// THE INPUTS ARE THE REFERENCE, THE SCORED BUCKETS **AND THE RUN'S OWN FINDINGS**, and the third was
|
|
1008
|
+
// missing. A proprietor this run surfaced that the lawyer never named is still somebody's name; the
|
|
1009
|
+
// buckets carry the ones the scorer could join to a reference entry, and the FINDINGS carry every other.
|
|
1010
|
+
//
|
|
1011
|
+
// Measured on R18 `f5764a2f`: the `verdict:` line printed a proprietor in clear while tokenising the mark
|
|
1012
|
+
// beside it. The name sits at `findings[0].owner.name` — a key already in the protected set — and was
|
|
1013
|
+
// simply never walked, because the findings were not handed over. It also appears inside five prose
|
|
1014
|
+
// fields of the same document (`net`, `practical_position`, `read`, `condition`, `text`), so collecting it
|
|
1015
|
+
// as a name substitutes it in all of them rather than needing each one withheld whole. That matters:
|
|
1016
|
+
// `text` carries the reviewer's verdict word, and withholding it entire would blank a value the reader
|
|
1017
|
+
// came for.
|
|
982
1018
|
let unclassifiedKeys = [];
|
|
1019
|
+
// The two instruments the JSON payload needs, live only when names are withheld. With `--names` they
|
|
1020
|
+
// stay identity, which is the same "nothing to withhold" case `printAuthored` already has.
|
|
1021
|
+
let redactValue = (x) => x, redactKey = (x) => x;
|
|
983
1022
|
if (!opts.names) {
|
|
984
|
-
const { names, prose, unclassified } = protectedStrings({ reference: ref, scored });
|
|
1023
|
+
const { names, prose, derived, unclassified } = protectedStrings({ reference: ref, scored, run: run.findings ?? null });
|
|
985
1024
|
unclassifiedKeys = unclassified;
|
|
986
|
-
|
|
1025
|
+
// The second redactor is for lines this tool WROTE. It drops the derived-word layer
|
|
1026
|
+
// only, so a party's ordinary long word stops rewriting our own headings while its full name is still
|
|
1027
|
+
// taken out of them. A structural line nobody routed through `printAuthored` is redacted as before,
|
|
1028
|
+
// which is the old behaviour and the safe side.
|
|
1029
|
+
redactValue = redactor({ names, prose });
|
|
1030
|
+
redactKey = authoredRedactor({ names, prose, derived });
|
|
1031
|
+
installRedaction(redactValue, undefined, redactKey);
|
|
987
1032
|
}
|
|
988
1033
|
|
|
989
1034
|
if (opts.json) {
|
|
990
|
-
|
|
1035
|
+
// REDACTED AS A STRUCTURE, THEN WRITTEN RAW. Stringifying first and sending the text through the
|
|
1036
|
+
// boundary redacts a document with no prose in it: every string is either a field NAME this file
|
|
1037
|
+
// wrote or a VALUE out of the data, and they need opposite treatment. On one real score that turned
|
|
1038
|
+
// the keys `owner` and `additional` into tokens, because both are distinctive words of multi-word
|
|
1039
|
+
// parties in that reference — two fields no consumer could address, or discover the new name of.
|
|
1040
|
+
printPreRedacted(`${JSON.stringify(redactDeep({
|
|
991
1041
|
// — the instrument that produced these numbers, so a reader comparing two archived scores can
|
|
992
1042
|
// tell whether the comparison is valid. A score with no `scorer_version` predates this stamp.
|
|
993
1043
|
scorer_version: SCORER_VERSION,
|
|
@@ -1012,14 +1062,19 @@ if (opts.json) {
|
|
|
1012
1062
|
// so leaving it out would hide it from exactly the readers most likely to automate on it, and a
|
|
1013
1063
|
// consumer would have no way to tell an absent measure from a clean one.
|
|
1014
1064
|
carry_through: (() => { const ct = carryThrough(run.dir); return { ...ct, coverage: coverageConflicts(run.dir, ct.lost) }; })(),
|
|
1015
|
-
}, null, 2));
|
|
1065
|
+
}, { redactValue, redactKey }), null, 2)}\n`);
|
|
1016
1066
|
} else {
|
|
1017
1067
|
// Said before the first number, not after the last: a reader who stops at the top screen must know
|
|
1018
1068
|
// which of the two readings they are holding.
|
|
1019
|
-
|
|
1069
|
+
// THE NOTICE THAT EXPLAINS THE TOKENS HAD TOKENS IN IT. A party's distinctive word collided with
|
|
1070
|
+
// "marks", so the sentence defining «name N» was itself redacted — useless exactly where it matters.
|
|
1071
|
+
if (!opts.names) printAuthored(`\n ${REDACTION_NOTICE}`);
|
|
1020
1072
|
// A reference that has grown a field this module does not classify is unprotected in exactly that
|
|
1021
1073
|
// field, so the reader learns it before anything below, not after.
|
|
1022
|
-
|
|
1074
|
+
// AND THE WARNING PRINTED ONE OF ITS OWN KEY NAMES AS A TOKEN, so it could not name the key it
|
|
1075
|
+
// exists to name. It fires when a new reference field is unclassified, which is the one moment it has
|
|
1076
|
+
// to be readable.
|
|
1077
|
+
if (unclassifiedKeys.length) printAuthored(` ${unclassifiedNotice(unclassifiedKeys)}`);
|
|
1023
1078
|
print(String(id).toUpperCase(), ref, run, scored, delta, refPath);
|
|
1024
1079
|
// — THE LAWYER'S OWN STATEMENTS OF WHAT THE RUN MUST DEMONSTRATE. The buckets cannot carry
|
|
1025
1080
|
// these: an assertion says WHY a mark matters, and that reasoning is what tells a reader which lane to
|
|
@@ -1028,11 +1083,29 @@ if (opts.json) {
|
|
|
1028
1083
|
if (statements.length) {
|
|
1029
1084
|
console.log(`\n THE REFERENCE'S OWN ASSERTIONS AND CONTROLS (${statements.length}) — the scorer does not read English, so`);
|
|
1030
1085
|
console.log(" these are the run's own facts about the marks each one names, never a verdict on the sentence:");
|
|
1086
|
+
// A MARK LIFTED OUT OF A WITHHELD SENTENCE IS STILL A MARK.
|
|
1087
|
+
//
|
|
1088
|
+
// `st.text` is a reference sentence, so it is in the prose set and prints as withheld. These rows are
|
|
1089
|
+
// the SAME sentence, parsed: `scoreStatements` runs the mark extractor over it and reports each mark
|
|
1090
|
+
// it names with the bucket that mark landed in. A mark that appears nowhere but inside that sentence
|
|
1091
|
+
// was never a value of a name field, so it is in no name set, and the boundary redactor has nothing
|
|
1092
|
+
// to match — it printed in clear, beside a mark that was correctly withheld, on a real score.
|
|
1093
|
+
//
|
|
1094
|
+
// WITHHELD AT THE PRINT SITE RATHER THAN BY WIDENING THE PROTECTED SET. Collecting what the extractor
|
|
1095
|
+
// finds into the protected set was tried first and measured: that extractor is a floor, not a sound
|
|
1096
|
+
// extraction — it reads capitals, lawyers emphasise in capitals, and the redactor is case-insensitive,
|
|
1097
|
+
// so an emphasised `REFERENCE` in one gold sentence tokenised every lowercase "reference" on the
|
|
1098
|
+
// page. 34 more tokens, 17 from that one word, and a bucket table reading `88% of 8 in-scope «name»
|
|
1099
|
+
// marks`. The cure was worse than the leak, so it is not there; the state is the finding, and the
|
|
1100
|
+
// mark's identity is the detail that belongs behind the flag.
|
|
1101
|
+
const markShown = (mark) => (opts.names ? mark : "[withheld — run again with --names to read it]");
|
|
1031
1102
|
for (const st of statements) {
|
|
1032
1103
|
console.log(`\n [${st.kind}] ${st.text}`);
|
|
1033
1104
|
if (st.halves.length)
|
|
1034
|
-
for (const h of st.halves) console.log(` ${h.mark} — ${h.state}`);
|
|
1035
|
-
|
|
1105
|
+
for (const h of st.halves) console.log(` ${markShown(h.mark)} — ${h.state}`);
|
|
1106
|
+
// `why` names the same marks in a sentence of its own, so it is withheld the same way or it is a
|
|
1107
|
+
// second door onto the first defect.
|
|
1108
|
+
if (st.why) console.log(` UNEVALUATED: ${opts.names ? st.why : st.why.replace(/names .*?, none of which/, "names marks this reference holds, none of which")}`);
|
|
1036
1109
|
}
|
|
1037
1110
|
} else if (ref) {
|
|
1038
1111
|
// Absence, stated. A reference with no assertions and a reference the scorer failed to read are
|
|
@@ -129,7 +129,37 @@ export const ALLOWED_CONTEXT =
|
|
|
129
129
|
* The escaped form (`C:\\`) is not a raw path, so an escape beside a path in a string is still read.
|
|
130
130
|
*/
|
|
131
131
|
const BACKSLASH_ESCAPE = /\\(?:[\\nrtfv0]|u[0-9a-fA-F]{4}|x[0-9a-fA-F]{2})/g;
|
|
132
|
-
|
|
132
|
+
|
|
133
|
+
// A RAW PATH IS NOT ONLY AN ABSOLUTE ONE. The drive-letter form was the only one recognised, so the same
|
|
134
|
+
// misreading came back on the relative forms: `.\node_modules\.bin` and `<prefix>\node_modules` were read
|
|
135
|
+
// as holding an escaped newline, which ate the `n` and left the three letters after it standing at the
|
|
136
|
+
// front of a word. Three occurrences on the public tip, in a CI workflow and a test.
|
|
137
|
+
//
|
|
138
|
+
// The anchors below are what makes a backslash a separator rather than an escape, and each is a shape a
|
|
139
|
+
// path is written in and a string literal is not. A backslash after a LETTER is deliberately not an anchor:
|
|
140
|
+
// that is the escape the rule exists for, and the arms pin it.
|
|
141
|
+
//
|
|
142
|
+
// EACH ANCHOR MATCHES ONLY THE SHAPE ITS COMMENT CLAIMS, and the first version of this did not. `[>%]` alone
|
|
143
|
+
// read ANY closing angle or percent before an escape as a path — `rose to 50%\nNorvale`, `cmd >\nNorvale` —
|
|
144
|
+
// and `\.{1,2}` read the last two dots of an ellipsis as `..\`. In each of those the escape was left
|
|
145
|
+
// unescaped, the `n` stayed glued to the name, and A REAL NAME WENT UNFOUND. That is the worse of this
|
|
146
|
+
// rule's two failure directions, and it is the one a widened anchor buys.
|
|
147
|
+
//
|
|
148
|
+
// THE RESIDUE, said here so it is not a surprise: `a<b>\nNorvale` still reads as a path, because `<b>` is
|
|
149
|
+
// indistinguishable by shape from `<root>`. A closing HTML tag immediately before an escape, with a name
|
|
150
|
+
// straight after it, is not caught. A minimum length inside the angles would be an arbitrary line through
|
|
151
|
+
// real stand-ins, so this is recorded rather than fixed.
|
|
152
|
+
const PATH_TAIL = "[^\"'`<>|?*\\r\\n]*";
|
|
153
|
+
const RAW_WINDOWS_PATH = new RegExp([
|
|
154
|
+
// C:\… — a drive letter
|
|
155
|
+
"(?<![A-Za-z0-9])[A-Za-z]:\\\\(?!\\\\)",
|
|
156
|
+
// .\… and ..\… — relative. The dot in the lookbehind is what stops an ellipsis reading as `..\`.
|
|
157
|
+
"(?<![A-Za-z0-9.])\\.{1,2}\\\\(?!\\\\)",
|
|
158
|
+
// <prefix>\… — a stand-in for the first segment, as a comment or a doc writes it
|
|
159
|
+
"(?<=<[A-Za-z_][A-Za-z0-9_]*>)\\\\(?!\\\\)",
|
|
160
|
+
// %TEMP%\… — an environment variable, as a script writes it
|
|
161
|
+
"(?<=%[A-Za-z_][A-Za-z0-9_]*%)\\\\(?!\\\\)",
|
|
162
|
+
].map((anchor) => anchor + PATH_TAIL).join("|"), "g");
|
|
133
163
|
|
|
134
164
|
export const unescapeBoundaries = (line) => {
|
|
135
165
|
let out = "", at = 0;
|