polytypo 1.2.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/README.md +48 -0
  2. package/dist/chunk-442V6TKI.cjs +31 -0
  3. package/dist/chunk-442V6TKI.cjs.map +1 -0
  4. package/dist/chunk-72GQS2N7.cjs +147 -0
  5. package/dist/chunk-72GQS2N7.cjs.map +1 -0
  6. package/dist/chunk-7S5PZ7T5.cjs +205 -0
  7. package/dist/chunk-7S5PZ7T5.cjs.map +1 -0
  8. package/dist/chunk-7ZJ7YYTQ.js +264 -0
  9. package/dist/chunk-7ZJ7YYTQ.js.map +1 -0
  10. package/dist/chunk-A3VCFDVL.cjs +264 -0
  11. package/dist/chunk-A3VCFDVL.cjs.map +1 -0
  12. package/dist/{chunk-KAO74OJ6.js → chunk-B3I7VL5E.js} +2 -2
  13. package/dist/chunk-DAUTLRFG.js +33 -0
  14. package/dist/chunk-DAUTLRFG.js.map +1 -0
  15. package/dist/chunk-GLQQOA6T.js +5778 -0
  16. package/dist/chunk-GLQQOA6T.js.map +1 -0
  17. package/dist/{chunk-POGDWVG5.cjs → chunk-KRY73ZPE.cjs} +40 -16
  18. package/dist/chunk-KRY73ZPE.cjs.map +1 -0
  19. package/dist/{chunk-YYHKX5HE.cjs → chunk-M7YWQ47K.cjs} +3 -3
  20. package/dist/{chunk-YYHKX5HE.cjs.map → chunk-M7YWQ47K.cjs.map} +1 -1
  21. package/dist/{chunk-VARU43JT.js → chunk-NT2RADO3.js} +4 -154
  22. package/dist/chunk-NT2RADO3.js.map +1 -0
  23. package/dist/chunk-PLWW5KPC.js +205 -0
  24. package/dist/chunk-PLWW5KPC.js.map +1 -0
  25. package/dist/{chunk-6NTDU63P.js → chunk-PNX4XW5I.js} +35 -11
  26. package/dist/chunk-PNX4XW5I.js.map +1 -0
  27. package/dist/chunk-QLV5EWGF.cjs +33 -0
  28. package/dist/chunk-QLV5EWGF.cjs.map +1 -0
  29. package/dist/chunk-W3XMTX7Q.js +31 -0
  30. package/dist/chunk-W3XMTX7Q.js.map +1 -0
  31. package/dist/chunk-YX6QTWOV.cjs +5778 -0
  32. package/dist/chunk-YX6QTWOV.cjs.map +1 -0
  33. package/dist/{errors-joZChNws.d.ts → errors-CGFOVO3h.d.cts} +34 -4
  34. package/dist/{errors-joZChNws.d.cts → errors-CGFOVO3h.d.ts} +34 -4
  35. package/dist/html.cjs +15 -7
  36. package/dist/html.cjs.map +1 -1
  37. package/dist/html.d.cts +9 -3
  38. package/dist/html.d.ts +9 -3
  39. package/dist/html.js +12 -4
  40. package/dist/html.js.map +1 -1
  41. package/dist/index.cjs +33 -12
  42. package/dist/index.cjs.map +1 -1
  43. package/dist/index.d.cts +14 -3
  44. package/dist/index.d.ts +14 -3
  45. package/dist/index.js +28 -7
  46. package/dist/index.js.map +1 -1
  47. package/dist/markdown.cjs +15 -7
  48. package/dist/markdown.cjs.map +1 -1
  49. package/dist/markdown.d.cts +9 -3
  50. package/dist/markdown.d.ts +9 -3
  51. package/dist/markdown.js +12 -4
  52. package/dist/markdown.js.map +1 -1
  53. package/dist/text.cjs +13 -6
  54. package/dist/text.cjs.map +1 -1
  55. package/dist/text.d.cts +8 -3
  56. package/dist/text.d.ts +8 -3
  57. package/dist/text.js +10 -3
  58. package/dist/text.js.map +1 -1
  59. package/dist/yaml.cjs +29 -0
  60. package/dist/yaml.cjs.map +1 -0
  61. package/dist/yaml.d.cts +27 -0
  62. package/dist/yaml.d.ts +27 -0
  63. package/dist/yaml.js +29 -0
  64. package/dist/yaml.js.map +1 -0
  65. package/package.json +12 -2
  66. package/dist/chunk-44A6YP7S.js +0 -21
  67. package/dist/chunk-44A6YP7S.js.map +0 -1
  68. package/dist/chunk-5VBDCDS6.cjs +0 -19
  69. package/dist/chunk-5VBDCDS6.cjs.map +0 -1
  70. package/dist/chunk-6NTDU63P.js.map +0 -1
  71. package/dist/chunk-B4O4R6HI.cjs +0 -5633
  72. package/dist/chunk-B4O4R6HI.cjs.map +0 -1
  73. package/dist/chunk-HGUA2NVX.cjs +0 -21
  74. package/dist/chunk-HGUA2NVX.cjs.map +0 -1
  75. package/dist/chunk-LS57O4KJ.cjs +0 -297
  76. package/dist/chunk-LS57O4KJ.cjs.map +0 -1
  77. package/dist/chunk-OHF7OFWY.js +0 -5633
  78. package/dist/chunk-OHF7OFWY.js.map +0 -1
  79. package/dist/chunk-POGDWVG5.cjs.map +0 -1
  80. package/dist/chunk-RC3UKUYZ.js +0 -19
  81. package/dist/chunk-RC3UKUYZ.js.map +0 -1
  82. package/dist/chunk-VARU43JT.js.map +0 -1
  83. /package/dist/{chunk-KAO74OJ6.js.map → chunk-B3I7VL5E.js.map} +0 -0
package/README.md CHANGED
@@ -57,6 +57,7 @@ the parsers that mode needs:
57
57
  | `polytypo/text` | none |
58
58
  | `polytypo/html` | parse5 |
59
59
  | `polytypo/markdown` | parse5, micromark and its GFM, frontmatter and MDX extensions |
60
+ | `polytypo/yaml` | none |
60
61
  | `polytypo` | all of the above |
61
62
 
62
63
  HTML and Markdown are first-class modes, not an afterthought — tags, attributes and fenced code
@@ -69,6 +70,53 @@ transform(`<a title="test... wait">Wait... she said "go on."</a>`, { locale: "en
69
70
  // <a title="test... wait">Wait… she said “go on.”</a>
70
71
  ```
71
72
 
73
+ `yaml` mode is the one that asks something of you, and it asks for a reason. YAML is a data format
74
+ with prose in some of it, so you name the keys whose values are prose; there is no default and no
75
+ guess:
76
+
77
+ ```ts
78
+ import { transform } from "polytypo/yaml";
79
+
80
+ transform("summary: Rates -- all of them...\nrun: git diff -- a--b\n", {
81
+ locale: "en-US",
82
+ keys: ["summary"],
83
+ });
84
+ // summary: Rates—all of them…
85
+ // run: git diff -- a--b
86
+ ```
87
+
88
+ Nothing in YAML's syntax separates a sentence from a shell script: `description` holds one and
89
+ `run` holds the other, spelled identically. Without `keys` the same document comes back with
90
+ `if !` rewritten as `if!`. Quoting, indentation, anchors and a block scalar's chomping indicator
91
+ are never decoded and rewritten — the file is located, not re-emitted — so the trailing newlines
92
+ of a `|+` block come back exactly as you wrote them.
93
+
94
+ In `markdown` mode the skipped regions are the dialect's own structural ones, not whatever looks
95
+ like code. CommonMark counts an indented block as code at **four** spaces; at two it is an ordinary
96
+ paragraph, so a JSON sample indented by two comes back with curly quotes — valid JSON in, invalid
97
+ JSON out, and nothing in the return value says so. MDX has no indented code blocks at all, so there
98
+ even four spaces is prose and a brace in it is a JSX expression: the same sample throws
99
+ `POLYTYPO_MALFORMED_INPUT` instead. A fence or a code span is the one marker that means _code_ in
100
+ both dialects.
101
+
102
+ `analyze()` runs the same pipeline and reports what it would do instead of doing it — one record
103
+ per edit, each with the rule that made it and code-point offsets into the input you passed (into
104
+ the **document**, in `html`, `markdown` and `yaml` mode, not into a span):
105
+
106
+ ```ts
107
+ import { analyze } from "polytypo";
108
+
109
+ analyze(`Wait... "really"?`, { locale: "en-US" });
110
+ // [ { ruleId: "ellipsis", start: 4, end: 7, before: "...", after: "…" },
111
+ // { ruleId: "quotes", start: 8, end: 9, before: `"`, after: "“" }, … ]
112
+ ```
113
+
114
+ It is a report, not a patch. The list is empty exactly when `transform` would return the input
115
+ unchanged, and every `ruleId` is a rule that was enabled for that call — but two rules may touch
116
+ the same original range (French `spaces` deletes the space before `:` and `nbsp` puts a no-break
117
+ one back), so replaying the list is not guaranteed to reproduce the output. Call `transform` for
118
+ the text. Full contract: `spec/rules/analyze.md`.
119
+
72
120
  The aggregate `polytypo` entry defaults `mode` to `"text"` and is the one to use when the mode is
73
121
  chosen at runtime. `locale` has no default anywhere and must always be passed explicitly — there is
74
122
  no silent fallback to English.
@@ -0,0 +1,31 @@
1
+ "use strict";Object.defineProperty(exports, "__esModule", {value: true});
2
+
3
+
4
+
5
+
6
+
7
+
8
+
9
+ var _chunkYX6QTWOVcjs = require('./chunk-YX6QTWOV.cjs');
10
+
11
+ // src/engine/text-pipeline.ts
12
+ function runTextPipeline(input, options) {
13
+ const narrowTarget = _chunkYX6QTWOVcjs.resolveNarrowTarget.call(void 0, options.narrowNbsp);
14
+ const planned = _chunkYX6QTWOVcjs.planRules.call(void 0, options.rules);
15
+ const locale = _chunkYX6QTWOVcjs.getLocaleData.call(void 0, options.locale);
16
+ return _chunkYX6QTWOVcjs.fromCodePoints.call(void 0, _chunkYX6QTWOVcjs.runRules.call(void 0, _chunkYX6QTWOVcjs.toCodePoints.call(void 0, input), planned, locale, "text", narrowTarget));
17
+ }
18
+ function analyzeTextPipeline(input, options) {
19
+ const narrowTarget = _chunkYX6QTWOVcjs.resolveNarrowTarget.call(void 0, options.narrowNbsp);
20
+ const planned = _chunkYX6QTWOVcjs.planRules.call(void 0, options.rules);
21
+ const locale = _chunkYX6QTWOVcjs.getLocaleData.call(void 0, options.locale);
22
+ const cp = _chunkYX6QTWOVcjs.toCodePoints.call(void 0, input);
23
+ const origin = cp.map((_value, index) => index);
24
+ return _chunkYX6QTWOVcjs.runRulesRecording.call(void 0, cp, planned, locale, "text", narrowTarget, origin, cp.length);
25
+ }
26
+
27
+
28
+
29
+
30
+ exports.runTextPipeline = runTextPipeline; exports.analyzeTextPipeline = analyzeTextPipeline;
31
+ //# sourceMappingURL=chunk-442V6TKI.cjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["/home/runner/work/polytypo-js/polytypo-js/dist/chunk-442V6TKI.cjs","../src/engine/text-pipeline.ts"],"names":[],"mappings":"AAAA;AACE;AACA;AACA;AACA;AACA;AACA;AACA;AACF,wDAA6B;AAC7B;AACA;ACQO,SAAS,eAAA,CAAgB,KAAA,EAAe,OAAA,EAAmC;AAChF,EAAA,MAAM,aAAA,EAAe,mDAAA,OAAoB,CAAQ,UAAU,CAAA;AAC3D,EAAA,MAAM,QAAA,EAAU,yCAAA,OAAU,CAAQ,KAAK,CAAA;AACvC,EAAA,MAAM,OAAA,EAAS,6CAAA,OAAc,CAAQ,MAAM,CAAA;AAC3C,EAAA,OAAO,8CAAA,wCAAe,4CAAS,KAAkB,CAAA,EAAG,OAAA,EAAS,MAAA,EAAQ,MAAA,EAAQ,YAAY,CAAC,CAAA;AAC5F;AAGO,SAAS,mBAAA,CAAoB,KAAA,EAAe,OAAA,EAAqC;AACtF,EAAA,MAAM,aAAA,EAAe,mDAAA,OAAoB,CAAQ,UAAU,CAAA;AAC3D,EAAA,MAAM,QAAA,EAAU,yCAAA,OAAU,CAAQ,KAAK,CAAA;AACvC,EAAA,MAAM,OAAA,EAAS,6CAAA,OAAc,CAAQ,MAAM,CAAA;AAC3C,EAAA,MAAM,GAAA,EAAK,4CAAA,KAAkB,CAAA;AAC7B,EAAA,MAAM,OAAA,EAAS,EAAA,CAAG,GAAA,CAAI,CAAC,MAAA,EAAQ,KAAA,EAAA,GAAU,KAAK,CAAA;AAC9C,EAAA,OAAO,iDAAA,EAAkB,EAAI,OAAA,EAAS,MAAA,EAAQ,MAAA,EAAQ,YAAA,EAAc,MAAA,EAAQ,EAAA,CAAG,MAAM,CAAA;AACvF;ADRA;AACA;AACE;AACA;AACF,6FAAC","file":"/home/runner/work/polytypo-js/polytypo-js/dist/chunk-442V6TKI.cjs","sourcesContent":[null,"import type { Options } from \"../types.js\";\nimport { fromCodePoints, toCodePoints } from \"./codepoints.js\";\nimport type { Change } from \"./origin.js\";\nimport { getLocaleData } from \"./locale.js\";\nimport { resolveNarrowTarget } from \"./narrow-target.js\";\nimport { planRules, runRules, runRulesRecording } from \"./rule-runner.js\";\n\n/**\n * `text` mode only. Deliberately imports nothing from `../modes/html.js` or\n * `../modes/markdown.js` (or their parser dependencies) — this is what makes `polytypo/text`'s\n * module graph exclude `parse5` and the Micromark stack (AUDIT_REMEDIATION_AND_RELEASE_PLAN.md\n * 5.1).\n *\n * Validation order — `planRules` before `getLocaleData` — mirrors the pre-Stage-5 aggregate\n * `runPipeline` exactly (an unknown-rule error must win over an unknown-locale error when both\n * are present, since that is public, tested behaviour, not an implementation detail this\n * refactor was authorised to change).\n */\nexport function runTextPipeline(input: string, options: Partial<Options>): string {\n const narrowTarget = resolveNarrowTarget(options.narrowNbsp);\n const planned = planRules(options.rules);\n const locale = getLocaleData(options.locale);\n return fromCodePoints(runRules(toCodePoints(input), planned, locale, \"text\", narrowTarget));\n}\n\n/** analyze.md §1: the same pipeline as `runTextPipeline`, reporting instead of applying. */\nexport function analyzeTextPipeline(input: string, options: Partial<Options>): Change[] {\n const narrowTarget = resolveNarrowTarget(options.narrowNbsp);\n const planned = planRules(options.rules);\n const locale = getLocaleData(options.locale);\n const cp = toCodePoints(input);\n const origin = cp.map((_value, index) => index);\n return runRulesRecording(cp, planned, locale, \"text\", narrowTarget, origin, cp.length);\n}\n"]}
@@ -0,0 +1,147 @@
1
+ "use strict";Object.defineProperty(exports, "__esModule", {value: true});
2
+
3
+ var _chunkYX6QTWOVcjs = require('./chunk-YX6QTWOV.cjs');
4
+
5
+ // src/modes/html.ts
6
+ var _parse5 = require('parse5');
7
+
8
+ // src/modes/parse-error.ts
9
+ function describe(error) {
10
+ if (typeof error !== "object" || error === null) return String(error);
11
+ const positioned = error;
12
+ const reason = typeof positioned.reason === "string" ? positioned.reason : error.message;
13
+ const place = positioned.place;
14
+ if (place !== null && place !== void 0 && typeof place.line === "number" && typeof place.column === "number") {
15
+ return `${reason} (line ${place.line}, column ${place.column})`;
16
+ }
17
+ return reason;
18
+ }
19
+ function wrapParserErrors(what, run) {
20
+ try {
21
+ return run();
22
+ } catch (error) {
23
+ if (error instanceof _chunkYX6QTWOVcjs.PolytypoError) throw error;
24
+ throw new (0, _chunkYX6QTWOVcjs.PolytypoError)(
25
+ "POLYTYPO_MALFORMED_INPUT",
26
+ `Input does not parse as ${what}: ${describe(error)}`
27
+ );
28
+ }
29
+ }
30
+
31
+ // src/modes/html.ts
32
+ var SKIPPED_ELEMENTS = /* @__PURE__ */ new Set([
33
+ "code",
34
+ "pre",
35
+ "kbd",
36
+ "samp",
37
+ "var",
38
+ "script",
39
+ "style",
40
+ "textarea",
41
+ "svg",
42
+ "math"
43
+ ]);
44
+ function isSkippedElement(tagName) {
45
+ return SKIPPED_ELEMENTS.has(tagName);
46
+ }
47
+ var AMPERSAND = 38;
48
+ var SEMICOLON = 59;
49
+ var HASH = 35;
50
+ var LOWER_X = 120;
51
+ var UPPER_X = 88;
52
+ var MAX_REFERENCE_BODY = 34;
53
+ function isAsciiDigit(unit) {
54
+ return unit >= 48 && unit <= 57;
55
+ }
56
+ function isAsciiHexDigit(unit) {
57
+ return isAsciiDigit(unit) || unit >= 97 && unit <= 102 || unit >= 65 && unit <= 70;
58
+ }
59
+ function isAsciiAlphanumeric(unit) {
60
+ return isAsciiDigit(unit) || unit >= 97 && unit <= 122 || unit >= 65 && unit <= 90;
61
+ }
62
+ function characterReferenceEnd(source, at, limit) {
63
+ if (source.charCodeAt(at) !== AMPERSAND) return -1;
64
+ let i = at + 1;
65
+ if (i >= limit) return -1;
66
+ if (source.charCodeAt(i) === HASH) {
67
+ i += 1;
68
+ const unit = i < limit ? source.charCodeAt(i) : -1;
69
+ const hex = unit === LOWER_X || unit === UPPER_X;
70
+ if (hex) i += 1;
71
+ const digitsAt = i;
72
+ while (i < limit && (hex ? isAsciiHexDigit(source.charCodeAt(i)) : isAsciiDigit(source.charCodeAt(i)))) {
73
+ i += 1;
74
+ }
75
+ if (i === digitsAt) return -1;
76
+ return i < limit && source.charCodeAt(i) === SEMICOLON ? i + 1 : -1;
77
+ }
78
+ const nameAt = i;
79
+ while (i < limit && i - nameAt < MAX_REFERENCE_BODY && isAsciiAlphanumeric(source.charCodeAt(i))) {
80
+ i += 1;
81
+ }
82
+ if (i === nameAt) return -1;
83
+ return i < limit && source.charCodeAt(i) === SEMICOLON ? i + 1 : -1;
84
+ }
85
+ function splitCharacterReferences(source, start, end) {
86
+ const spans = [];
87
+ let cursor = start;
88
+ let i = start;
89
+ while (i < end) {
90
+ if (source.charCodeAt(i) === AMPERSAND) {
91
+ const stop = characterReferenceEnd(source, i, end);
92
+ if (stop > 0) {
93
+ if (i > cursor) spans.push({ start: cursor, end: i });
94
+ cursor = stop;
95
+ i = stop;
96
+ continue;
97
+ }
98
+ }
99
+ i += 1;
100
+ }
101
+ if (end > cursor) spans.push({ start: cursor, end });
102
+ return spans;
103
+ }
104
+ function isParent(node) {
105
+ return "childNodes" in node;
106
+ }
107
+ function collect(node, source, spans) {
108
+ if (node.nodeName === "#text") {
109
+ const location = node.sourceCodeLocation;
110
+ if (location === void 0 || location === null) return;
111
+ for (const span of splitCharacterReferences(source, location.startOffset, location.endOffset)) {
112
+ spans.push(span);
113
+ }
114
+ return;
115
+ }
116
+ if (node.nodeName === "#comment" || node.nodeName === "#documentType") return;
117
+ if ("tagName" in node && isSkippedElement(node.tagName)) return;
118
+ if (node.nodeName === "template" && "content" in node) {
119
+ collect(node.content, source, spans);
120
+ return;
121
+ }
122
+ if (!isParent(node)) return;
123
+ for (const child of node.childNodes) collect(child, source, spans);
124
+ }
125
+ function htmlSpans(source) {
126
+ const document = wrapParserErrors("HTML", () => _parse5.parse.call(void 0, source, { sourceCodeLocationInfo: true }));
127
+ const spans = [];
128
+ collect(document, source, spans);
129
+ return spans;
130
+ }
131
+ function htmlFragmentSpans(source, offset) {
132
+ const fragment = wrapParserErrors(
133
+ "HTML",
134
+ () => _parse5.parseFragment.call(void 0, source, { sourceCodeLocationInfo: true })
135
+ );
136
+ const spans = [];
137
+ collect(fragment, source, spans);
138
+ return offset === 0 ? spans : spans.map((span) => ({ start: span.start + offset, end: span.end + offset }));
139
+ }
140
+
141
+
142
+
143
+
144
+
145
+
146
+ exports.wrapParserErrors = wrapParserErrors; exports.isSkippedElement = isSkippedElement; exports.htmlSpans = htmlSpans; exports.htmlFragmentSpans = htmlFragmentSpans;
147
+ //# sourceMappingURL=chunk-72GQS2N7.cjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["/home/runner/work/polytypo-js/polytypo-js/dist/chunk-72GQS2N7.cjs","../src/modes/html.ts","../src/modes/parse-error.ts"],"names":[],"mappings":"AAAA;AACE;AACF,wDAA6B;AAC7B;AACA;ACJA,gCAAqC;ADMrC;AACA;AEAA,SAAS,QAAA,CAAS,KAAA,EAAwB;AACxC,EAAA,GAAA,CAAI,OAAO,MAAA,IAAU,SAAA,GAAY,MAAA,IAAU,IAAA,EAAM,OAAO,MAAA,CAAO,KAAK,CAAA;AACpE,EAAA,MAAM,WAAA,EAAa,KAAA;AACnB,EAAA,MAAM,OAAA,EACJ,OAAO,UAAA,CAAW,OAAA,IAAW,SAAA,EAAW,UAAA,CAAW,OAAA,EAAU,KAAA,CAAgB,OAAA;AAC/E,EAAA,MAAM,MAAA,EAAQ,UAAA,CAAW,KAAA;AACzB,EAAA,GAAA,CACE,MAAA,IAAU,KAAA,GACV,MAAA,IAAU,KAAA,EAAA,GACV,OAAO,KAAA,CAAM,KAAA,IAAS,SAAA,GACtB,OAAO,KAAA,CAAM,OAAA,IAAW,QAAA,EACxB;AACA,IAAA,OAAO,CAAA,EAAA;AACT,EAAA;AACO,EAAA;AACT;AAcgB;AACV,EAAA;AACK,IAAA;AACA,EAAA;AACH,IAAA;AACE,IAAA;AACJ,MAAA;AACA,MAAA;AACF,IAAA;AACF,EAAA;AACF;AFjBY;AACA;ACfN;AACJ,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACA,EAAA;AACD;AAEe;AACP,EAAA;AACT;AAEM;AACA;AACO;AACP;AACA;AAGA;AAEG;AACA,EAAA;AACT;AAES;AACA,EAAA;AACT;AAES;AACA,EAAA;AACT;AAQS;AACH,EAAA;AACI,EAAA;AACC,EAAA;AAEL,EAAA;AACG,IAAA;AACC,IAAA;AACA,IAAA;AACG,IAAA;AACH,IAAA;AAEJ,IAAA;AAGK,MAAA;AACP,IAAA;AACI,IAAA;AACG,IAAA;AACT,EAAA;AAEM,EAAA;AAEJ,EAAA;AAIK,IAAA;AACP,EAAA;AACU,EAAA;AACH,EAAA;AACT;AAOgB;AACR,EAAA;AACF,EAAA;AACI,EAAA;AACD,EAAA;AACD,IAAA;AACI,MAAA;AACF,MAAA;AACE,QAAA;AACJ,QAAA;AACI,QAAA;AACJ,QAAA;AACF,MAAA;AACF,IAAA;AACK,IAAA;AACP,EAAA;AACU,EAAA;AACH,EAAA;AACT;AAES;AACA,EAAA;AACT;AAES;AACE,EAAA;AACD,IAAA;AACF,IAAA;AACJ,IAAA;AACQ,MAAA;AACR,IAAA;AACA,IAAA;AACF,EAAA;AAGS,EAAA;AAEL,EAAA;AAGK,EAAA;AACC,IAAA;AACR,IAAA;AACF,EAAA;AAEK,EAAA;AACL,EAAA;AACF;AAQgB;AACR,EAAA;AACA,EAAA;AACE,EAAA;AACD,EAAA;AACT;AAGgB;AACR,EAAA;AAA4B,IAAA;AAChC,IAAA;AACF,EAAA;AACM,EAAA;AACE,EAAA;AACD,EAAA;AAGT;AD9BY;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/polytypo-js/polytypo-js/dist/chunk-72GQS2N7.cjs","sourcesContent":[null,"import { parse, parseFragment } from \"parse5\";\nimport type { DefaultTreeAdapterMap } from \"parse5\";\nimport { wrapParserErrors } from \"./parse-error.js\";\nimport type { Span } from \"./spans.js\";\n\ntype Node = DefaultTreeAdapterMap[\"node\"];\ntype ParentNode = DefaultTreeAdapterMap[\"parentNode\"];\n\n/**\n * spec/rules/modes.md 3.6, exhaustive and **closed**: extending it is a spec change, not an\n * implementation decision. `svg` and `math` are here because in MathML a quotation mark, a\n * hyphen and a prime are operators and identifiers — substituting a curly glyph changes what\n * the expression means. Everything not listed is processable, **including unknown and custom\n * elements**: guessing from a tag name is exactly the heuristic that diverges across runtimes.\n */\nconst SKIPPED_ELEMENTS: ReadonlySet<string> = new Set([\n \"code\",\n \"pre\",\n \"kbd\",\n \"samp\",\n \"var\",\n \"script\",\n \"style\",\n \"textarea\",\n \"svg\",\n \"math\",\n]);\n\nexport function isSkippedElement(tagName: string): boolean {\n return SKIPPED_ELEMENTS.has(tagName);\n}\n\nconst AMPERSAND = 0x26;\nconst SEMICOLON = 0x3b;\nconst HASH = 0x23;\nconst LOWER_X = 0x78;\nconst UPPER_X = 0x58;\n\n/** Longest named reference in the WHATWG table is 32 characters; the guard is a bound, not a rule. */\nconst MAX_REFERENCE_BODY = 34;\n\nfunction isAsciiDigit(unit: number): boolean {\n return unit >= 0x30 && unit <= 0x39;\n}\n\nfunction isAsciiHexDigit(unit: number): boolean {\n return isAsciiDigit(unit) || (unit >= 0x61 && unit <= 0x66) || (unit >= 0x41 && unit <= 0x46);\n}\n\nfunction isAsciiAlphanumeric(unit: number): boolean {\n return isAsciiDigit(unit) || (unit >= 0x61 && unit <= 0x7a) || (unit >= 0x41 && unit <= 0x5a);\n}\n\n/**\n * The end offset of a well-formed character reference starting at `at`, or -1. A bare `&` that a\n * parser would repair is deliberately **not** treated as a reference: it stays inside the span,\n * where no rule can emit or delete it, rather than manufacturing a boundary that suppresses\n * conversions around it.\n */\nfunction characterReferenceEnd(source: string, at: number, limit: number): number {\n if (source.charCodeAt(at) !== AMPERSAND) return -1;\n let i = at + 1;\n if (i >= limit) return -1;\n\n if (source.charCodeAt(i) === HASH) {\n i += 1;\n const unit = i < limit ? source.charCodeAt(i) : -1;\n const hex = unit === LOWER_X || unit === UPPER_X;\n if (hex) i += 1;\n const digitsAt = i;\n while (\n i < limit &&\n (hex ? isAsciiHexDigit(source.charCodeAt(i)) : isAsciiDigit(source.charCodeAt(i)))\n ) {\n i += 1;\n }\n if (i === digitsAt) return -1;\n return i < limit && source.charCodeAt(i) === SEMICOLON ? i + 1 : -1;\n }\n\n const nameAt = i;\n while (\n i < limit &&\n i - nameAt < MAX_REFERENCE_BODY &&\n isAsciiAlphanumeric(source.charCodeAt(i))\n ) {\n i += 1;\n }\n if (i === nameAt) return -1;\n return i < limit && source.charCodeAt(i) === SEMICOLON ? i + 1 : -1;\n}\n\n/**\n * modes.md 3.6: every character reference is skipped as an **opaque unit**, and a text node\n * containing one is split into spans around it. That is the only way `&nbsp;` survives as\n * `&nbsp;` instead of becoming a literal U+00A0 on the way out.\n */\nexport function splitCharacterReferences(source: string, start: number, end: number): Span[] {\n const spans: Span[] = [];\n let cursor = start;\n let i = start;\n while (i < end) {\n if (source.charCodeAt(i) === AMPERSAND) {\n const stop = characterReferenceEnd(source, i, end);\n if (stop > 0) {\n if (i > cursor) spans.push({ start: cursor, end: i });\n cursor = stop;\n i = stop;\n continue;\n }\n }\n i += 1;\n }\n if (end > cursor) spans.push({ start: cursor, end });\n return spans;\n}\n\nfunction isParent(node: Node): node is ParentNode {\n return \"childNodes\" in node;\n}\n\nfunction collect(node: Node, source: string, spans: Span[]): void {\n if (node.nodeName === \"#text\") {\n const location = node.sourceCodeLocation;\n if (location === undefined || location === null) return;\n for (const span of splitCharacterReferences(source, location.startOffset, location.endOffset)) {\n spans.push(span);\n }\n return;\n }\n\n // Comments, doctype and processing instructions never contribute a span (modes.md 3.6).\n if (node.nodeName === \"#comment\" || node.nodeName === \"#documentType\") return;\n\n if (\"tagName\" in node && isSkippedElement(node.tagName)) return;\n\n // A `template`'s children live in a separate document fragment; `template` is not skipped.\n if (node.nodeName === \"template\" && \"content\" in node) {\n collect(node.content, source, spans);\n return;\n }\n\n if (!isParent(node)) return;\n for (const child of node.childNodes) collect(child, source, spans);\n}\n\n/**\n * Locate the processable spans of an HTML document. The tree is used only to find offsets and is\n * then discarded — modes.md 4 forbids reserialisation, which is what makes attribute quoting,\n * self-closing forms, entity spelling, tag case and malformed-input recovery preserved by\n * construction rather than by effort.\n */\nexport function htmlSpans(source: string): Span[] {\n const document = wrapParserErrors(\"HTML\", () => parse(source, { sourceCodeLocationInfo: true }));\n const spans: Span[] = [];\n collect(document, source, spans);\n return spans;\n}\n\n/** The same, for a fragment: used for HTML blocks embedded in markdown (modes.md 3.7). */\nexport function htmlFragmentSpans(source: string, offset: number): Span[] {\n const fragment = wrapParserErrors(\"HTML\", () =>\n parseFragment(source, { sourceCodeLocationInfo: true }),\n );\n const spans: Span[] = [];\n collect(fragment, source, spans);\n return offset === 0\n ? spans\n : spans.map((span) => ({ start: span.start + offset, end: span.end + offset }));\n}\n","import { PolytypoError } from \"../errors.js\";\n\ninterface PositionedError {\n readonly reason?: unknown;\n readonly place?: { readonly line?: unknown; readonly column?: unknown } | null;\n}\n\nfunction describe(error: unknown): string {\n if (typeof error !== \"object\" || error === null) return String(error);\n const positioned = error as PositionedError;\n const reason =\n typeof positioned.reason === \"string\" ? positioned.reason : (error as Error).message;\n const place = positioned.place;\n if (\n place !== null &&\n place !== undefined &&\n typeof place.line === \"number\" &&\n typeof place.column === \"number\"\n ) {\n return `${reason} (line ${place.line}, column ${place.column})`;\n }\n return reason;\n}\n\n/**\n * spec/rules/modes.md 3.7.2. **The parser's own error type must never escape.** A\n * `VFileMessage`, a `Nokogiri::SyntaxError` or a Python exception on the public surface puts a\n * dependency's type into the contract and is unreproducible in the other four runtimes; only the\n * code is contractual (ARCHITECTURE.md 4.6), while the message and the source position are\n * useful and are not.\n *\n * Reachable in exactly one place: `dialect: \"mdx\"`, because MDX embeds JavaScript. HTML parsing\n * is specified with total error recovery and every byte sequence is valid CommonMark, so neither\n * of the other two languages can fail. The wrapper is applied to all of them anyway — the\n * guarantee is that no parser type escapes, not that this particular parser is trusted.\n */\nexport function wrapParserErrors<T>(what: string, run: () => T): T {\n try {\n return run();\n } catch (error) {\n if (error instanceof PolytypoError) throw error;\n throw new PolytypoError(\n \"POLYTYPO_MALFORMED_INPUT\",\n `Input does not parse as ${what}: ${describe(error)}`,\n );\n }\n}\n"]}
@@ -0,0 +1,205 @@
1
+ "use strict";Object.defineProperty(exports, "__esModule", {value: true});
2
+
3
+
4
+
5
+
6
+
7
+
8
+
9
+
10
+
11
+
12
+ var _chunkYX6QTWOVcjs = require('./chunk-YX6QTWOV.cjs');
13
+
14
+ // src/modes/spans.ts
15
+ var SPACE = 32;
16
+ var LINE_TERMINATORS = /* @__PURE__ */ new Set([
17
+ 10,
18
+ 13,
19
+ 11,
20
+ 12,
21
+ 133,
22
+ 8232,
23
+ 8233
24
+ ]);
25
+ function gapIsLineBoundary(source, from, to) {
26
+ for (let i = from; i < to; i += 1) {
27
+ if (LINE_TERMINATORS.has(source.charCodeAt(i))) return true;
28
+ }
29
+ return false;
30
+ }
31
+ function normalizeSpans(spans) {
32
+ const sorted = spans.filter((s) => s.end > s.start).sort((a, b) => a.start - b.start);
33
+ const out = [];
34
+ for (const span of sorted) {
35
+ const last = out[out.length - 1];
36
+ if (last === void 0) {
37
+ out.push(span);
38
+ continue;
39
+ }
40
+ if (span.start < last.end) {
41
+ throw new (0, _chunkYX6QTWOVcjs.PolytypoError)(
42
+ "POLYTYPO_RULE_CONTRACT",
43
+ `mode extractor produced overlapping spans (${last.start}, ${last.end}) and (${span.start}, ${span.end})`
44
+ );
45
+ }
46
+ if (span.start === last.end) {
47
+ out[out.length - 1] = { start: last.start, end: span.end };
48
+ continue;
49
+ }
50
+ out.push(span);
51
+ }
52
+ return out;
53
+ }
54
+ function concatenateSpans(source, spans) {
55
+ const cp = [];
56
+ let previous;
57
+ for (const span of spans) {
58
+ if (previous !== void 0) {
59
+ cp.push(gapIsLineBoundary(source, previous.end, span.start) ? _chunkYX6QTWOVcjs.LINE_MARKER : _chunkYX6QTWOVcjs.MARKER);
60
+ }
61
+ for (const value of _chunkYX6QTWOVcjs.toCodePoints.call(void 0, source.slice(span.start, span.end))) cp.push(value);
62
+ previous = span;
63
+ }
64
+ return cp;
65
+ }
66
+ function spanRangesOf(cp) {
67
+ const ranges = [];
68
+ let first = 0;
69
+ for (let i = 0; i < cp.length; i += 1) {
70
+ if (_chunkYX6QTWOVcjs.isMarker.call(void 0, cp[i])) {
71
+ ranges.push({ first, last: i - 1 });
72
+ first = i + 1;
73
+ }
74
+ }
75
+ ranges.push({ first, last: cp.length - 1 });
76
+ return ranges;
77
+ }
78
+ function spanContaining(ranges, p) {
79
+ for (const range of ranges) {
80
+ if (p >= range.first && p <= range.last + 1) return range;
81
+ }
82
+ return void 0;
83
+ }
84
+ function filterBoundaryEdits(cp, edits, ranges) {
85
+ const out = [];
86
+ for (const edit of edits) {
87
+ let containsMarker = false;
88
+ for (let i = edit.start; i < edit.end; i += 1) {
89
+ if (_chunkYX6QTWOVcjs.isMarker.call(void 0, cp[i])) {
90
+ containsMarker = true;
91
+ break;
92
+ }
93
+ }
94
+ if (containsMarker) continue;
95
+ const p = edit.start;
96
+ const q = edit.end - 1;
97
+ const d = edit.end - edit.start;
98
+ const r = edit.replacement.length;
99
+ const span = spanContaining(ranges, p);
100
+ if (span !== void 0 && (p === span.first || q === span.last)) {
101
+ if (r > d) continue;
102
+ const first = edit.replacement[0];
103
+ const last = edit.replacement[r - 1];
104
+ if (r > 0 && p === span.first && first === SPACE && cp[p] !== SPACE) continue;
105
+ if (r > 0 && q === span.last && last === SPACE && cp[q] !== SPACE) continue;
106
+ }
107
+ out.push(edit);
108
+ }
109
+ return out;
110
+ }
111
+ function splitOnMarker(cp, expected) {
112
+ const pieces = [[]];
113
+ for (const value of cp) {
114
+ if (_chunkYX6QTWOVcjs.isMarker.call(void 0, value)) {
115
+ pieces.push([]);
116
+ continue;
117
+ }
118
+ pieces[pieces.length - 1].push(value);
119
+ }
120
+ if (pieces.length !== expected) {
121
+ throw new (0, _chunkYX6QTWOVcjs.PolytypoError)(
122
+ "POLYTYPO_RULE_CONTRACT",
123
+ `boundary markers did not survive the pipeline: expected ${expected} spans, found ${pieces.length}`
124
+ );
125
+ }
126
+ return pieces;
127
+ }
128
+ function originOfSpans(source, spans) {
129
+ const codePointIndexOf = new Array(source.length + 1);
130
+ let cpIndex = 0;
131
+ for (let i = 0; i < source.length; ) {
132
+ codePointIndexOf[i] = cpIndex;
133
+ const code = source.codePointAt(i);
134
+ const width = code > 65535 ? 2 : 1;
135
+ if (width === 2) codePointIndexOf[i + 1] = cpIndex;
136
+ i += width;
137
+ cpIndex += 1;
138
+ }
139
+ codePointIndexOf[source.length] = cpIndex;
140
+ const origin = [];
141
+ let previous;
142
+ for (const span of spans) {
143
+ if (previous !== void 0) origin.push(_chunkYX6QTWOVcjs.NO_ORIGIN);
144
+ const base = codePointIndexOf[span.start];
145
+ const length = _chunkYX6QTWOVcjs.toCodePoints.call(void 0, source.slice(span.start, span.end)).length;
146
+ for (let k = 0; k < length; k += 1) origin.push(base + k);
147
+ previous = span;
148
+ }
149
+ return origin;
150
+ }
151
+
152
+ // src/engine/span-runner.ts
153
+ function runRulesOverSpans(cp, planned, locale, mode, narrowTarget) {
154
+ let current = cp;
155
+ for (const id of planned) {
156
+ const rule = _chunkYX6QTWOVcjs.RULES[id];
157
+ if (rule === void 0) continue;
158
+ const edits = rule.apply({ cp: current, locale, mode, narrowTarget });
159
+ current = _chunkYX6QTWOVcjs.applyEdits.call(void 0, current, filterBoundaryEdits(current, edits, spanRangesOf(current)), id);
160
+ }
161
+ return current;
162
+ }
163
+ function runOverSpans(source, spans, planned, locale, mode, narrowTarget) {
164
+ const normalized = normalizeSpans(spans);
165
+ if (normalized.length === 0) return source;
166
+ const transformed = runRulesOverSpans(
167
+ concatenateSpans(source, normalized),
168
+ planned,
169
+ locale,
170
+ mode,
171
+ narrowTarget
172
+ );
173
+ const pieces = splitOnMarker(transformed, normalized.length);
174
+ let out = "";
175
+ let cursor = 0;
176
+ for (let i = 0; i < normalized.length; i += 1) {
177
+ const span = normalized[i];
178
+ const replacement = _chunkYX6QTWOVcjs.fromCodePoints.call(void 0, pieces[i]);
179
+ const original = source.slice(span.start, span.end);
180
+ out += source.slice(cursor, span.start);
181
+ out += replacement === original ? original : replacement;
182
+ cursor = span.end;
183
+ }
184
+ return out + source.slice(cursor);
185
+ }
186
+ function analyzeOverSpans(source, spans, planned, locale, mode, narrowTarget) {
187
+ const normalized = normalizeSpans(spans);
188
+ if (normalized.length === 0) return [];
189
+ return _chunkYX6QTWOVcjs.runRulesRecording.call(void 0,
190
+ concatenateSpans(source, normalized),
191
+ planned,
192
+ locale,
193
+ mode,
194
+ narrowTarget,
195
+ originOfSpans(source, normalized),
196
+ _chunkYX6QTWOVcjs.toCodePoints.call(void 0, source).length,
197
+ (current, edits) => filterBoundaryEdits(current, edits, spanRangesOf(current))
198
+ );
199
+ }
200
+
201
+
202
+
203
+
204
+ exports.runOverSpans = runOverSpans; exports.analyzeOverSpans = analyzeOverSpans;
205
+ //# sourceMappingURL=chunk-7S5PZ7T5.cjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["/home/runner/work/polytypo-js/polytypo-js/dist/chunk-7S5PZ7T5.cjs","../src/modes/spans.ts","../src/engine/span-runner.ts"],"names":[],"mappings":"AAAA;AACE;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACF,wDAA6B;AAC7B;AACA;ACMA,IAAM,MAAA,EAAQ,EAAA;AAEd,IAAM,iBAAA,kBAAwC,IAAI,GAAA,CAAI;AAAA,EACpD,EAAA;AAAA,EAAM,EAAA;AAAA,EAAM,EAAA;AAAA,EAAM,EAAA;AAAA,EAAM,GAAA;AAAA,EAAM,IAAA;AAAA,EAAQ;AACxC,CAAC,CAAA;AAED,SAAS,iBAAA,CAAkB,MAAA,EAAgB,IAAA,EAAc,EAAA,EAAqB;AAC5E,EAAA,IAAA,CAAA,IAAS,EAAA,EAAI,IAAA,EAAM,EAAA,EAAI,EAAA,EAAI,EAAA,GAAK,CAAA,EAAG;AACjC,IAAA,GAAA,CAAI,gBAAA,CAAiB,GAAA,CAAI,MAAA,CAAO,UAAA,CAAW,CAAC,CAAC,CAAA,EAAG,OAAO,IAAA;AAAA,EACzD;AACA,EAAA,OAAO,KAAA;AACT;AAuBO,SAAS,cAAA,CAAe,KAAA,EAAgC;AAC7D,EAAA,MAAM,OAAA,EAAS,KAAA,CAAM,MAAA,CAAO,CAAC,CAAA,EAAA,GAAM,CAAA,CAAE,IAAA,EAAM,CAAA,CAAE,KAAK,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,EAAA,GAAM,CAAA,CAAE,MAAA,EAAQ,CAAA,CAAE,KAAK,CAAA;AACpF,EAAA,MAAM,IAAA,EAAc,CAAC,CAAA;AACrB,EAAA,IAAA,CAAA,MAAW,KAAA,GAAQ,MAAA,EAAQ;AACzB,IAAA,MAAM,KAAA,EAAO,GAAA,CAAI,GAAA,CAAI,OAAA,EAAS,CAAC,CAAA;AAC/B,IAAA,GAAA,CAAI,KAAA,IAAS,KAAA,CAAA,EAAW;AACtB,MAAA,GAAA,CAAI,IAAA,CAAK,IAAI,CAAA;AACb,MAAA,QAAA;AAAA,IACF;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,MAAA,EAAQ,IAAA,CAAK,GAAA,EAAK;AACzB,MAAA,MAAM,IAAI,oCAAA;AAAA,QACR,wBAAA;AAAA,QACA,CAAA,2CAAA,EAA8C,IAAA,CAAK,KAAK,CAAA,EAAA,EAAK,IAAA,CAAK,GAAG,CAAA,OAAA,EAAU,IAAA,CAAK,KAAK,CAAA,EAAA,EAAK,IAAA,CAAK,GAAG,CAAA,CAAA;AAAA,MACxG,CAAA;AAAA,IACF;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,MAAA,IAAU,IAAA,CAAK,GAAA,EAAK;AAC3B,MAAA,GAAA,CAAI,GAAA,CAAI,OAAA,EAAS,CAAC,EAAA,EAAI,EAAE,KAAA,EAAO,IAAA,CAAK,KAAA,EAAO,GAAA,EAAK,IAAA,CAAK,IAAI,CAAA;AACzD,MAAA,QAAA;AAAA,IACF;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,IAAI,CAAA;AAAA,EACf;AACA,EAAA,OAAO,GAAA;AACT;AAGO,SAAS,gBAAA,CAAiB,MAAA,EAAgB,KAAA,EAAkC;AACjF,EAAA,MAAM,GAAA,EAAe,CAAC,CAAA;AACtB,EAAA,IAAI,QAAA;AACJ,EAAA,IAAA,CAAA,MAAW,KAAA,GAAQ,KAAA,EAAO;AACxB,IAAA,GAAA,CAAI,SAAA,IAAa,KAAA,CAAA,EAAW;AAC1B,MAAA,EAAA,CAAG,IAAA,CAAK,iBAAA,CAAkB,MAAA,EAAQ,QAAA,CAAS,GAAA,EAAK,IAAA,CAAK,KAAK,EAAA,EAAI,8BAAA,EAAc,wBAAM,CAAA;AAAA,IACpF;AACA,IAAA,IAAA,CAAA,MAAW,MAAA,GAAS,4CAAA,MAAa,CAAO,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,IAAA,CAAK,GAAG,CAAC,CAAA,EAAG,EAAA,CAAG,IAAA,CAAK,KAAK,CAAA;AACnF,IAAA,SAAA,EAAW,IAAA;AAAA,EACb;AACA,EAAA,OAAO,EAAA;AACT;AAOO,SAAS,YAAA,CAAa,EAAA,EAAoC;AAC/D,EAAA,MAAM,OAAA,EAAsB,CAAC,CAAA;AAC7B,EAAA,IAAI,MAAA,EAAQ,CAAA;AACZ,EAAA,IAAA,CAAA,IAAS,EAAA,EAAI,CAAA,EAAG,EAAA,EAAI,EAAA,CAAG,MAAA,EAAQ,EAAA,GAAK,CAAA,EAAG;AACrC,IAAA,GAAA,CAAI,wCAAA,EAAS,CAAG,CAAC,CAAW,CAAA,EAAG;AAC7B,MAAA,MAAA,CAAO,IAAA,CAAK,EAAE,KAAA,EAAO,IAAA,EAAM,EAAA,EAAI,EAAE,CAAC,CAAA;AAClC,MAAA,MAAA,EAAQ,EAAA,EAAI,CAAA;AAAA,IACd;AAAA,EACF;AACA,EAAA,MAAA,CAAO,IAAA,CAAK,EAAE,KAAA,EAAO,IAAA,EAAM,EAAA,CAAG,OAAA,EAAS,EAAE,CAAC,CAAA;AAC1C,EAAA,OAAO,MAAA;AACT;AAEA,SAAS,cAAA,CAAe,MAAA,EAA8B,CAAA,EAAkC;AACtF,EAAA,IAAA,CAAA,MAAW,MAAA,GAAS,MAAA,EAAQ;AAC1B,IAAA,GAAA,CAAI,EAAA,GAAK,KAAA,CAAM,MAAA,GAAS,EAAA,GAAK,KAAA,CAAM,KAAA,EAAO,CAAA,EAAG,OAAO,KAAA;AAAA,EACtD;AACA,EAAA,OAAO,KAAA,CAAA;AACT;AA4BO,SAAS,mBAAA,CACd,EAAA,EACA,KAAA,EACA,MAAA,EACQ;AACR,EAAA,MAAM,IAAA,EAAc,CAAC,CAAA;AACrB,EAAA,IAAA,CAAA,MAAW,KAAA,GAAQ,KAAA,EAAO;AACxB,IAAA,IAAI,eAAA,EAAiB,KAAA;AACrB,IAAA,IAAA,CAAA,IAAS,EAAA,EAAI,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,IAAA,CAAK,GAAA,EAAK,EAAA,GAAK,CAAA,EAAG;AAC7C,MAAA,GAAA,CAAI,wCAAA,EAAS,CAAG,CAAC,CAAW,CAAA,EAAG;AAC7B,QAAA,eAAA,EAAiB,IAAA;AACjB,QAAA,KAAA;AAAA,MACF;AAAA,IACF;AACA,IAAA,GAAA,CAAI,cAAA,EAAgB,QAAA;AAIpB,IAAA,MAAM,EAAA,EAAI,IAAA,CAAK,KAAA;AACf,IAAA,MAAM,EAAA,EAAI,IAAA,CAAK,IAAA,EAAM,CAAA;AACrB,IAAA,MAAM,EAAA,EAAI,IAAA,CAAK,IAAA,EAAM,IAAA,CAAK,KAAA;AAC1B,IAAA,MAAM,EAAA,EAAI,IAAA,CAAK,WAAA,CAAY,MAAA;AAC3B,IAAA,MAAM,KAAA,EAAO,cAAA,CAAe,MAAA,EAAQ,CAAC,CAAA;AACrC,IAAA,GAAA,CAAI,KAAA,IAAS,KAAA,EAAA,GAAA,CAAc,EAAA,IAAM,IAAA,CAAK,MAAA,GAAS,EAAA,IAAM,IAAA,CAAK,IAAA,CAAA,EAAO;AAC/D,MAAA,GAAA,CAAI,EAAA,EAAI,CAAA,EAAG,QAAA;AAMX,MAAA,MAAM,MAAA,EAAQ,IAAA,CAAK,WAAA,CAAY,CAAC,CAAA;AAChC,MAAA,MAAM,KAAA,EAAO,IAAA,CAAK,WAAA,CAAY,EAAA,EAAI,CAAC,CAAA;AACnC,MAAA,GAAA,CAAI,EAAA,EAAI,EAAA,GAAK,EAAA,IAAM,IAAA,CAAK,MAAA,GAAS,MAAA,IAAU,MAAA,GAAS,EAAA,CAAG,CAAC,EAAA,IAAM,KAAA,EAAO,QAAA;AACrE,MAAA,GAAA,CAAI,EAAA,EAAI,EAAA,GAAK,EAAA,IAAM,IAAA,CAAK,KAAA,GAAQ,KAAA,IAAS,MAAA,GAAS,EAAA,CAAG,CAAC,EAAA,IAAM,KAAA,EAAO,QAAA;AAAA,IACrE;AAEA,IAAA,GAAA,CAAI,IAAA,CAAK,IAAI,CAAA;AAAA,EACf;AACA,EAAA,OAAO,GAAA;AACT;AAGO,SAAS,aAAA,CAAc,EAAA,EAAuB,QAAA,EAA8B;AACjF,EAAA,MAAM,OAAA,EAAqB,CAAC,CAAC,CAAC,CAAA;AAC9B,EAAA,IAAA,CAAA,MAAW,MAAA,GAAS,EAAA,EAAI;AACtB,IAAA,GAAA,CAAI,wCAAA,KAAc,CAAA,EAAG;AACnB,MAAA,MAAA,CAAO,IAAA,CAAK,CAAC,CAAC,CAAA;AACd,MAAA,QAAA;AAAA,IACF;AACA,IAAC,MAAA,CAAO,MAAA,CAAO,OAAA,EAAS,CAAC,CAAA,CAAe,IAAA,CAAK,KAAK,CAAA;AAAA,EACpD;AACA,EAAA,GAAA,CAAI,MAAA,CAAO,OAAA,IAAW,QAAA,EAAU;AAC9B,IAAA,MAAM,IAAI,oCAAA;AAAA,MACR,wBAAA;AAAA,MACA,CAAA,wDAAA,EAA2D,QAAQ,CAAA,cAAA,EAAiB,MAAA,CAAO,MAAM,CAAA;AAAA,IAAA;AACnG,EAAA;AAEF,EAAA;AACF;AASO;AACL,EAAA;AACA,EAAA;AACA,EAAA;AACE,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AAAW,EAAA;AAEb,EAAA;AAEA,EAAA;AACA,EAAA;AACA,EAAA;AACE,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AAAW,EAAA;AAEb,EAAA;AACF;ADlFA;AACA;AEjIA;AAOE,EAAA;AACA,EAAA;AACE,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AAA4F,EAAA;AAE9F,EAAA;AACF;AAYO;AAQL,EAAA;AACA,EAAA;AAEA,EAAA;AAAoB,IAAA;AACiB,IAAA;AACnC,IAAA;AACA,IAAA;AACA,IAAA;AACA,EAAA;AAEF,EAAA;AAEA,EAAA;AACA,EAAA;AACA,EAAA;AACE,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AAAc,EAAA;AAEhB,EAAA;AACF;AAOO;AAQL,EAAA;AACA,EAAA;AACA,EAAA;AAAO,IAAA;AAC8B,IAAA;AACnC,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACgC,IAAA;AACX,IAAA;AACwD,EAAA;AAEjF;AF4FA;AACA;AACA;AACA;AACA","file":"/home/runner/work/polytypo-js/polytypo-js/dist/chunk-7S5PZ7T5.cjs","sourcesContent":[null,"import { toCodePoints } from \"../engine/codepoints.js\";\nimport { isMarker, LINE_MARKER, MARKER } from \"../engine/sentinels.js\";\nimport { PolytypoError } from \"../errors.js\";\nimport type { Edit } from \"../types.js\";\nimport { NO_ORIGIN } from \"../engine/origin.js\";\n\n/**\n * The boundary markers are defined in the engine (src/engine/sentinels.ts), which is where all\n * three non-code-point sentinels live and where their disjointness is maintained. They are\n * re-exported here because the mode layer is what writes them into the array.\n *\n * Which marker separates two spans is decided by the **raw source bytes of the gap between\n * them**, so it is decidable without asking the parser anything and is identical in five\n * runtimes: a gap containing a line terminator gives `LINE_MARKER`, anything else `MARKER`.\n */\nexport { LINE_MARKER, MARKER, isMarker } from \"../engine/sentinels.js\";\n\n/** `BREAK` as the rules define it (`spaces.md` 3.1), tested against the gap's raw source. */\n/** U+0020, the one emitted code point whose meaning is positional (modes.md 3.4, 5 item 2). */\nconst SPACE = 0x20;\n\nconst LINE_TERMINATORS: ReadonlySet<number> = new Set([\n 0x0a, 0x0d, 0x0b, 0x0c, 0x85, 0x2028, 0x2029,\n]);\n\nfunction gapIsLineBoundary(source: string, from: number, to: number): boolean {\n for (let i = from; i < to; i += 1) {\n if (LINE_TERMINATORS.has(source.charCodeAt(i))) return true;\n }\n return false;\n}\n\n/**\n * A processable span, identified by its offsets in the **original source**. Offsets are UTF-16\n * indices into the JS source string, which is what every JS parser reports; they are an\n * implementation detail of this runtime and never reach the rules, which index code points.\n */\nexport interface Span {\n readonly start: number;\n readonly end: number;\n}\n\n/** A span's extent in the concatenated code-point array: `s₀` and `s₁` of modes.md 3.4. */\nexport interface SpanRange {\n readonly first: number;\n readonly last: number;\n}\n\n/**\n * Sort, drop empties, and coalesce spans separated by nothing in the source (modes.md 7.5:\n * a parser that reports one text run as two adjacent nodes must not manufacture a boundary).\n * Overlapping spans are an extractor bug and are rejected rather than silently merged.\n */\nexport function normalizeSpans(spans: readonly Span[]): Span[] {\n const sorted = spans.filter((s) => s.end > s.start).sort((a, b) => a.start - b.start);\n const out: Span[] = [];\n for (const span of sorted) {\n const last = out[out.length - 1];\n if (last === undefined) {\n out.push(span);\n continue;\n }\n if (span.start < last.end) {\n throw new PolytypoError(\n \"POLYTYPO_RULE_CONTRACT\",\n `mode extractor produced overlapping spans (${last.start}, ${last.end}) and (${span.start}, ${span.end})`,\n );\n }\n if (span.start === last.end) {\n out[out.length - 1] = { start: last.start, end: span.end };\n continue;\n }\n out.push(span);\n }\n return out;\n}\n\n/** `S₁ ⌢ [marker] ⌢ S₂ ⌢ … ⌢ Sₘ` (modes.md 3.5 step 2). */\nexport function concatenateSpans(source: string, spans: readonly Span[]): number[] {\n const cp: number[] = [];\n let previous: Span | undefined;\n for (const span of spans) {\n if (previous !== undefined) {\n cp.push(gapIsLineBoundary(source, previous.end, span.start) ? LINE_MARKER : MARKER);\n }\n for (const value of toCodePoints(source.slice(span.start, span.end))) cp.push(value);\n previous = span;\n }\n return cp;\n}\n\n/**\n * The span extents of the array as it stands. Recomputed after every rule, because applying\n * edits shifts every index after the first one — the markers themselves are guaranteed to\n * survive, since no edit may contain one.\n */\nexport function spanRangesOf(cp: readonly number[]): SpanRange[] {\n const ranges: SpanRange[] = [];\n let first = 0;\n for (let i = 0; i < cp.length; i += 1) {\n if (isMarker(cp[i] as number)) {\n ranges.push({ first, last: i - 1 });\n first = i + 1;\n }\n }\n ranges.push({ first, last: cp.length - 1 });\n return ranges;\n}\n\nfunction spanContaining(ranges: readonly SpanRange[], p: number): SpanRange | undefined {\n for (const range of ranges) {\n if (p >= range.first && p <= range.last + 1) return range;\n }\n return undefined;\n}\n\n/**\n * modes.md 3.4, two safety nets, both pure functions of `(p, q, r, s₀, s₁)` — so the verdict is\n * identical on every run, which is what makes redistribution deterministic (modes.md 5, point 3).\n *\n * 1. **No edit may contain a marker.** No rule can produce one; one that does is a bug, and the\n * edit is discarded rather than treated as a redistribution question.\n * 2. **The edge-growth rule.** An edit is discarded if it would place code points at an\n * extremity of its span that were not there before: with `d = q - p + 1` the replaced length\n * and `r` the replacement length, discard when `p = s₀ and r > d`, or `q = s₁ and r > d`.\n *\n * The length test is the whole rule and needs no knowledge of Markdown or HTML syntax. It\n * separates exactly the cases that matter: `\"` → `“` at an edge is 1 → 1 and applies; `--` →\n * `␣–␣` at an edge is 2 → 3 and is discarded, while the same edit interior to a span applies;\n * `(c)` → `©` is 3 → 1 and applies, because shrinking is always safe. An insertion has `d = 0`,\n * so it is discarded exactly when its position coincides with a span edge — the rule this one\n * generalises.\n *\n * **Deletion at an edge is not restricted here**, and must not be: `r > d` is false for a\n * deletion, so this filter never sees one. Deletion at an edge is governed by modes.md 3.3's\n * *Edge tests* clause instead — where a rule asks \"am I at the edge of the text I am allowed to\n * modify\" rather than \"what character is here\", the marker behaves as `NONE`. That clause is the\n * one place a rule may treat a span edge as the end of the text, it exists because `spaces` is\n * the only rule that deletes, and 3.3 states the division: **a rule that deletes must treat a\n * span edge as the end of the text; a rule that replaces or inserts must not, and is governed by\n * this filter.** The single test that claims the clause is `spaces.md` 3.2 step 4.\n */\nexport function filterBoundaryEdits(\n cp: readonly number[],\n edits: readonly Edit[],\n ranges: readonly SpanRange[],\n): Edit[] {\n const out: Edit[] = [];\n for (const edit of edits) {\n let containsMarker = false;\n for (let i = edit.start; i < edit.end; i += 1) {\n if (isMarker(cp[i] as number)) {\n containsMarker = true;\n break;\n }\n }\n if (containsMarker) continue;\n\n // `[start, end)` in the engine's half-open form is `cp[p … q]` with p = start, q = end - 1;\n // an insertion is `q = p - 1`, which falls out of the same expression.\n const p = edit.start;\n const q = edit.end - 1;\n const d = edit.end - edit.start;\n const r = edit.replacement.length;\n const span = spanContaining(ranges, p);\n if (span !== undefined && (p === span.first || q === span.last)) {\n if (r > d) continue;\n // The character clause. `r > d` is not the rule, only a formalisation of it that misses\n // `r === d`: `dashes` P3 admits a run of THREE dashes, so `---` -> `␣–␣` is 3 -> 3 and the\n // length test sees nothing while U+0020 lands on both extremities anyway. Testing the\n // character catches it, and only U+0020 needs testing — it is the one code point any rule\n // emits whose meaning comes from its position rather than from itself (modes.md 5 item 2).\n const first = edit.replacement[0];\n const last = edit.replacement[r - 1];\n if (r > 0 && p === span.first && first === SPACE && cp[p] !== SPACE) continue;\n if (r > 0 && q === span.last && last === SPACE && cp[q] !== SPACE) continue;\n }\n\n out.push(edit);\n }\n return out;\n}\n\n/** Redistribute the transformed array back to one piece per span (modes.md 3.5 step 4). */\nexport function splitOnMarker(cp: readonly number[], expected: number): number[][] {\n const pieces: number[][] = [[]];\n for (const value of cp) {\n if (isMarker(value)) {\n pieces.push([]);\n continue;\n }\n (pieces[pieces.length - 1] as number[]).push(value);\n }\n if (pieces.length !== expected) {\n throw new PolytypoError(\n \"POLYTYPO_RULE_CONTRACT\",\n `boundary markers did not survive the pipeline: expected ${expected} spans, found ${pieces.length}`,\n );\n }\n return pieces;\n}\n\n/**\n * The origin map for `concatenateSpans` (analyze.md §2): for every code point of the joined\n * array, the code-point offset of the character it came from IN THE DOCUMENT, and NO_ORIGIN for\n * the markers, which came from nowhere. `Span` bounds are native string indices, so the\n * conversion to code-point offsets happens here and not in the caller — this is the one place\n * that knows both coordinate systems.\n */\nexport function originOfSpans(source: string, spans: readonly Span[]): number[] {\n const codePointIndexOf = new Array<number>(source.length + 1);\n let cpIndex = 0;\n for (let i = 0; i < source.length;) {\n codePointIndexOf[i] = cpIndex;\n const code = source.codePointAt(i) as number;\n const width = code > 0xffff ? 2 : 1;\n if (width === 2) codePointIndexOf[i + 1] = cpIndex;\n i += width;\n cpIndex += 1;\n }\n codePointIndexOf[source.length] = cpIndex;\n\n const origin: number[] = [];\n let previous: Span | undefined;\n for (const span of spans) {\n if (previous !== undefined) origin.push(NO_ORIGIN);\n const base = codePointIndexOf[span.start] as number;\n const length = toCodePoints(source.slice(span.start, span.end)).length;\n for (let k = 0; k < length; k += 1) origin.push(base + k);\n previous = span;\n }\n return origin;\n}\n","import {\n concatenateSpans,\n filterBoundaryEdits,\n normalizeSpans,\n originOfSpans,\n spanRangesOf,\n splitOnMarker,\n type Span,\n} from \"../modes/spans.js\";\nimport { RULES } from \"../rules/registry.js\";\nimport type { LocaleData, Mode, RuleId } from \"../types.js\";\nimport { applyEdits } from \"./edits.js\";\nimport { fromCodePoints, toCodePoints } from \"./codepoints.js\";\nimport type { Change } from \"./origin.js\";\nimport { runRulesRecording } from \"./rule-runner.js\";\n\n/**\n * The same sequence as `runRules` (`./rule-runner.js`), with the two boundary filters of\n * modes.md 3.4 interposed. The span extents are recomputed after every rule, because applying\n * an edit shifts every index after it; the markers themselves always survive, since no edit may\n * contain one.\n */\nfunction runRulesOverSpans(\n cp: readonly number[],\n planned: readonly RuleId[],\n locale: LocaleData,\n mode: Mode,\n narrowTarget: number,\n): readonly number[] {\n let current = cp;\n for (const id of planned) {\n const rule = RULES[id];\n if (rule === undefined) continue;\n const edits = rule.apply({ cp: current, locale, mode, narrowTarget });\n current = applyEdits(current, filterBoundaryEdits(current, edits, spanRangesOf(current)), id);\n }\n return current;\n}\n\n/**\n * modes.md 3.5. The pipeline runs **once**, over the marker-separated concatenation of every\n * processable span — not per span, which would pair quotation marks in isolation, and not over a\n * naive concatenation, which would manufacture adjacencies the document does not have.\n *\n * The output is the input with a set of disjoint substring replacements applied and nothing else\n * (modes.md 4). A span whose content the rules did not change contributes no replacement, so a\n * document needing no changes comes back byte-identical; the parser located the spans and was\n * then discarded, and the document is never serialised.\n */\nexport function runOverSpans(\n source: string,\n spans: readonly Span[],\n planned: readonly RuleId[],\n locale: LocaleData,\n mode: Mode,\n narrowTarget: number,\n): string {\n const normalized = normalizeSpans(spans);\n if (normalized.length === 0) return source;\n\n const transformed = runRulesOverSpans(\n concatenateSpans(source, normalized),\n planned,\n locale,\n mode,\n narrowTarget,\n );\n const pieces = splitOnMarker(transformed, normalized.length);\n\n let out = \"\";\n let cursor = 0;\n for (let i = 0; i < normalized.length; i += 1) {\n const span = normalized[i] as Span;\n const replacement = fromCodePoints(pieces[i] as number[]);\n const original = source.slice(span.start, span.end);\n out += source.slice(cursor, span.start);\n out += replacement === original ? original : replacement;\n cursor = span.end;\n }\n return out + source.slice(cursor);\n}\n\n/**\n * `runOverSpans`, reporting instead of applying (analyze.md §1). The span table supplies the\n * origin map, so every change comes back in DOCUMENT coordinates — analyze.md §6 names a\n * runtime that reports span-local offsets here as the mistake that passes every text-mode test.\n */\nexport function analyzeOverSpans(\n source: string,\n spans: readonly Span[],\n planned: readonly RuleId[],\n locale: LocaleData,\n mode: Mode,\n narrowTarget: number,\n): Change[] {\n const normalized = normalizeSpans(spans);\n if (normalized.length === 0) return [];\n return runRulesRecording(\n concatenateSpans(source, normalized),\n planned,\n locale,\n mode,\n narrowTarget,\n originOfSpans(source, normalized),\n toCodePoints(source).length,\n (current, edits) => filterBoundaryEdits(current, edits, spanRangesOf(current)),\n );\n}\n"]}