@chaffjs/lang-en 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +1 -1
  2. package/dist/citation.d.ts +9 -0
  3. package/dist/citation.d.ts.map +1 -1
  4. package/dist/citation.js +40 -4
  5. package/dist/citation.js.map +1 -1
  6. package/dist/dates.d.ts +3 -0
  7. package/dist/dates.d.ts.map +1 -0
  8. package/dist/dates.js +84 -0
  9. package/dist/dates.js.map +1 -0
  10. package/dist/definitions.d.ts +5 -0
  11. package/dist/definitions.d.ts.map +1 -0
  12. package/dist/definitions.js +33 -0
  13. package/dist/definitions.js.map +1 -0
  14. package/dist/depth.d.ts +4 -0
  15. package/dist/depth.d.ts.map +1 -0
  16. package/dist/depth.js +4 -0
  17. package/dist/depth.js.map +1 -0
  18. package/dist/index.d.ts.map +1 -1
  19. package/dist/index.js +2 -4
  20. package/dist/index.js.map +1 -1
  21. package/dist/reference-list.d.ts +7 -0
  22. package/dist/reference-list.d.ts.map +1 -0
  23. package/dist/reference-list.js +56 -0
  24. package/dist/reference-list.js.map +1 -0
  25. package/dist/roman.d.ts +3 -0
  26. package/dist/roman.d.ts.map +1 -0
  27. package/dist/roman.js +10 -0
  28. package/dist/roman.js.map +1 -0
  29. package/dist/sentence-split.d.ts +6 -0
  30. package/dist/sentence-split.d.ts.map +1 -0
  31. package/dist/sentence-split.js +114 -0
  32. package/dist/sentence-split.js.map +1 -0
  33. package/dist/structure.d.ts +0 -2
  34. package/dist/structure.d.ts.map +1 -1
  35. package/dist/structure.js +73 -90
  36. package/dist/structure.js.map +1 -1
  37. package/lexicons/total-label.yaml +9 -0
  38. package/lexicons/weekday.yaml +11 -0
  39. package/package.json +1 -1
  40. package/src/citation.ts +48 -4
  41. package/src/dates.ts +93 -0
  42. package/src/definitions.ts +43 -0
  43. package/src/depth.ts +3 -0
  44. package/src/index.ts +2 -4
  45. package/src/reference-list.ts +66 -0
  46. package/src/roman.ts +9 -0
  47. package/src/sentence-split.ts +131 -0
  48. package/src/structure.ts +76 -104
package/src/roman.ts ADDED
@@ -0,0 +1,9 @@
1
+ const ROMAN: Readonly<Record<string, number>> = { i: 1, v: 5, x: 10, l: 50, c: 100 };
2
+
3
+ /** "IV" → 4, "xii" → 12. Undefined for anything that is not a roman numeral. */
4
+ export const parseRoman = (text: string): number | undefined => {
5
+ const values = [...text.toLowerCase()].map((char) => ROMAN[char]);
6
+ if (values.length === 0 || values.some((value) => value === undefined)) return undefined;
7
+ const known = values.filter((value) => value !== undefined);
8
+ return known.reduce((total, value, index) => ((known[index + 1] ?? 0) > value ? total - value : total + value), 0);
9
+ };
@@ -0,0 +1,131 @@
1
+ import { DefaultAbbrMarkerOptions, split, SentenceSplitterSyntax } from "sentence-splitter";
2
+ import type { Span } from "chaffjs/plugin";
3
+
4
+ /**
5
+ * sentence-splitter に長い段落を一度に渡すと、文の数の 2 乗で遅くなる。
6
+ * 文を閉じるたびに、それまでに閉じた括弧を全部並べ直すため。
7
+ *
8
+ * そこで分割器の状態が空に戻る所(文が閉じ、括弧も開いていない所)で切り、切れ目ごとに渡す。
9
+ * 切ってよいかは分割器の規則を写して決める。写し方が狂うと文が変わるので、迷う所では切らない。
10
+ */
11
+
12
+ // PairMaker と同じ組。開き括弧を鍵にする。
13
+ const PAIRS: readonly (readonly [string, string])[] = [
14
+ ['"', '"'],
15
+ ["[", "]"],
16
+ ["(", ")"],
17
+ ["{", "}"],
18
+ ["「", "」"],
19
+ ["(", ")"],
20
+ ["『", "』"],
21
+ ["{", "}"],
22
+ ["[", "]"],
23
+ ["〚", "〛"],
24
+ ["【", "】"],
25
+ ["《", "》"],
26
+ ];
27
+ const KEY_OF = new Map<string, string>(
28
+ PAIRS.flatMap(([open, close]): [string, string][] => [
29
+ [close, open],
30
+ [open, open],
31
+ ]),
32
+ );
33
+ const CLOSE_OF = new Map<string, string>(PAIRS.map(([open, close]): [string, string] => [open, close]));
34
+
35
+ // AbbrMarker の語の区切り。分割器は UTF-16 の 1 単位ずつ読むので、こちらも 1 単位で判定する。
36
+ const CJK = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u;
37
+ const isCJK = (unit: string | undefined): boolean => unit !== undefined && CJK.test(unit);
38
+ const isSpace = (unit: string | undefined): boolean => unit !== undefined && /\s/.test(unit);
39
+
40
+ // 和文の句点の並び。直前が仮名・漢字なら、略語の判定がこの句点に掛かることはない。
41
+ const JAPANESE_STOPS = /[。.?!]+/g;
42
+ // 英語は「空白の後の英字だけの語 + ピリオド + 空白」。略語の一覧に無い語だけを切れ目にする。
43
+ const ENGLISH_STOP = /(?<=^|\s)[A-Za-z]{2,}\.(?=\s)/g;
44
+ const { language } = DefaultAbbrMarkerOptions;
45
+ const ABBREVIATIONS = new Set(
46
+ [...language.ABBREVIATIONS, ...language.PREPOSITIVE_ABBREVIATIONS, ...language.EXCLAMATION_WORDS].map((word) => word.toLowerCase()),
47
+ );
48
+ // 「J. Smith」のような語は前の語を見て略語か決まる。前の語が切れ目の向こうにあると判定が変わる。
49
+ const CAPITAL_DOT = /\p{Lu}\./gu;
50
+
51
+ /** 各位置の手前で、開いたままの括弧があるか。PairMaker と同じく種類ごとに 1 つまで開く。 */
52
+ const pairOpenBefore = (text: string): boolean[] => {
53
+ const open = new Set<string>();
54
+ const before = [false];
55
+ text.split("").forEach((unit) => {
56
+ const key = KEY_OF.get(unit);
57
+ if (key !== undefined && !open.has(key) && unit === key) open.add(key);
58
+ else if (key !== undefined && open.has(key) && unit === CLOSE_OF.get(key)) open.delete(key);
59
+ before.push(open.size > 0);
60
+ });
61
+ return before;
62
+ };
63
+
64
+ /** sorted の中で from 以上の最初の値。無ければ fallback。 */
65
+ const firstAtOrAfter = (sorted: readonly number[], from: number, fallback: number): number => {
66
+ const search = (low: number, high: number): number => {
67
+ if (low >= high) return sorted[low] ?? fallback;
68
+ const middle = Math.floor((low + high) / 2);
69
+ return (sorted[middle] ?? fallback) >= from ? search(low, middle) : search(middle + 1, high);
70
+ };
71
+ return search(0, sorted.length);
72
+ };
73
+
74
+ const indexesOf = (text: string, pattern: RegExp): number[] => [...text.matchAll(pattern)].map((match) => match.index);
75
+
76
+ /** 句点の並びの直後。和文は直前が仮名・漢字の句点、英文は略語でない語のピリオド。 */
77
+ const stopEnds = (text: string): number[] => {
78
+ const japanese = [...text.matchAll(JAPANESE_STOPS)].filter((match) => isCJK(text[match.index - 1]));
79
+ const english = [...text.matchAll(ENGLISH_STOP)].filter((match) => !ABBREVIATIONS.has(match[0].toLowerCase()));
80
+ return [...japanese, ...english].map((match) => match.index + match[0].length).sort((a, b) => a - b);
81
+ };
82
+
83
+ const skipSpaces = (text: string, from: number): number => {
84
+ const spaces = /\s*/y;
85
+ spaces.lastIndex = from;
86
+ return from + (spaces.exec(text)?.[0].length ?? 0);
87
+ };
88
+
89
+ /**
90
+ * 句点の直後から次の文の頭へ。空白を挟むか、仮名・漢字が続くときだけ切れる。
91
+ * それ以外の文字が続くと、分割器はその文字を前の語の続きとして読む。
92
+ */
93
+ const chunkStart = (text: string, stopEnd: number): number | undefined => {
94
+ const next = text[stopEnd];
95
+ if (isSpace(next)) return skipSpaces(text, stopEnd);
96
+ return isCJK(next) ? stopEnd : undefined;
97
+ };
98
+
99
+ type Landmarks = { readonly open: readonly boolean[]; readonly spaces: readonly number[]; readonly capitalDots: readonly number[] };
100
+
101
+ /**
102
+ * 切れ目の後ろ、最初の 2 つの語の塊に「J.」の形が無いこと。
103
+ * 分割器はその形の語だけ、空白を越えて前の語を読み返す。
104
+ */
105
+ const noLookBack = (text: string, start: number, marks: Landmarks): boolean => {
106
+ const secondRun = skipSpaces(text, firstAtOrAfter(marks.spaces, start, text.length));
107
+ return firstAtOrAfter(marks.capitalDots, start, Infinity) > secondRun;
108
+ };
109
+
110
+ const cutPoints = (text: string): number[] => {
111
+ const marks: Landmarks = { open: pairOpenBefore(text), spaces: indexesOf(text, /\s/g), capitalDots: indexesOf(text, CAPITAL_DOT) };
112
+ return stopEnds(text).flatMap((stopEnd) => {
113
+ if (marks.open[stopEnd] === true) return [];
114
+ const start = chunkStart(text, stopEnd);
115
+ return start !== undefined && start < text.length && noLookBack(text, start, marks) ? [start] : [];
116
+ });
117
+ };
118
+
119
+ const spansOf = (text: string, offset: number): Span[] =>
120
+ split(text)
121
+ .filter((node) => node.type === SentenceSplitterSyntax.Sentence)
122
+ .map((node) => ({ start: offset + node.range[0], end: offset + node.range[1] }));
123
+
124
+ /** 分割器に別々に渡してよい区間。つなぐと text 全体になる。 */
125
+ export const chunksOf = (text: string): Span[] => {
126
+ const starts = [0, ...cutPoints(text)];
127
+ return starts.map((start, index) => ({ start, end: starts[index + 1] ?? text.length }));
128
+ };
129
+
130
+ /** split(text) の文の span と同じものを返す。区間ごとに分けて渡すだけ。 */
131
+ export const sentenceSpans = (text: string): Span[] => chunksOf(text).flatMap((chunk) => spansOf(text.slice(chunk.start, chunk.end), chunk.start));
package/src/structure.ts CHANGED
@@ -1,19 +1,14 @@
1
1
  import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
2
2
  import { citedDocumentAfter } from "./citation.ts";
3
+ import { membersAfter } from "./reference-list.ts";
4
+ import { parseRoman } from "./roman.ts";
5
+ import { dates } from "./dates.ts";
6
+ import { definitionScopeDepth, definitions, opensDefinitionScope } from "./definitions.ts";
7
+ import { CHAPTER_DEPTH, PART_DEPTH } from "./depth.ts";
3
8
 
4
9
  // Contracts, specifications and statutes in English. core nests what this reads; it does not know
5
10
  // how English numbers its articles.
6
11
 
7
- const ROMAN: Readonly<Record<string, number>> = { i: 1, v: 5, x: 10, l: 50, c: 100 };
8
-
9
- /** "IV" → 4, "xii" → 12. Undefined for anything that is not a roman numeral. */
10
- export const parseRoman = (text: string): number | undefined => {
11
- const values = [...text.toLowerCase()].map((char) => ROMAN[char]);
12
- if (values.length === 0 || values.some((value) => value === undefined)) return undefined;
13
- const known = values.filter((value) => value !== undefined);
14
- return known.reduce((total, value, index) => ((known[index + 1] ?? 0) > value ? total - value : total + value), 0);
15
- };
16
-
17
12
  const numberOf = (text: string | undefined): string | undefined => {
18
13
  if (text === undefined) return undefined;
19
14
  if (/^\d{1,3}(?:\.\d{1,3}){0,5}$/u.test(text)) return text;
@@ -33,10 +28,16 @@ const titleOf = (rest: string): string | undefined => {
33
28
 
34
29
  const ARTICLE = /^\s{0,3}(?:ARTICLE|Article)\s+(?<n>\d{1,3}|[IVXLC]{1,7})\b(?<rest>.*)$/u;
35
30
  const SECTION = /^\s{0,3}(?:SECTION|Section|§)\s*(?<n>\d{1,3}(?:\.\d{1,3}){0,5})\b(?<rest>.*)$/u;
36
- const LETTERED = /^\s{0,6}\((?<n>[a-z]{1,4}|\d{1,3})\)\s+(?<rest>\S.*)$/u;
31
+ /**
32
+ * An amendment inserts a subsection between two others and numbers it "(A1)" or "(2A)". It is a subsection, written
33
+ * outside the sequence: it has no ordinal, so "(1)" after "(A1)" is still the first.
34
+ */
35
+ const INSERTED = "\\d{1,3}[A-Z]{1,2}|[A-Z]{1,2}\\d{1,3}";
36
+ const LETTERED = new RegExp(`^\\s{0,6}\\((?<n>[a-z]{1,4}|\\d{1,3}|${INSERTED})\\)\\s+(?<rest>\\S.*)$`, "u");
37
+ const IS_INSERTED = new RegExp(`^(?:${INSERTED})$`, "u");
37
38
  const MULTI_ROMAN = /^(?:ii|iii|iv|vi|vii|viii|ix)$/u;
38
39
 
39
- const headed = (pattern: RegExp, line: string, label: (n: string) => string): NumberedLine | undefined => {
40
+ const headed = (pattern: RegExp, line: string, numbering: string, label: (n: string) => string): NumberedLine | undefined => {
40
41
  const groups = pattern.exec(line)?.groups;
41
42
  const number = numberOf(groups?.["n"]);
42
43
  const heading = titleOf(groups?.["rest"] ?? "");
@@ -51,6 +52,7 @@ const headed = (pattern: RegExp, line: string, label: (n: string) => string): Nu
51
52
  heading,
52
53
  rest: heading,
53
54
  ordinal: Number(parts.at(-1)),
55
+ numbering,
54
56
  };
55
57
  };
56
58
 
@@ -61,9 +63,9 @@ type Style = "letter" | "roman" | "digit";
61
63
  * 開いたときの読みを覚えておく代わりに、一つ上に英字が開いていたかで決め直す。
62
64
  */
63
65
  const styleOfLabel = (open: NumberedLine, index: number, all: readonly NumberedLine[]): Style | undefined => {
64
- const inner = /^\((?<n>[a-z0-9]{1,4})\)$/u.exec(open.label)?.groups?.["n"];
66
+ const inner = /^\((?<n>[A-Za-z0-9]{1,5})\)$/u.exec(open.label)?.groups?.["n"];
65
67
  if (inner === undefined) return undefined;
66
- if (/^\d+$/u.test(inner)) return "digit";
68
+ if (/^\d+$/u.test(inner) || IS_INSERTED.test(inner)) return "digit";
67
69
  if (MULTI_ROMAN.test(inner)) return "roman";
68
70
  const above = all[index - 1];
69
71
  return /^[ivx]$/u.test(inner) && above !== undefined && styleOfLabel(above, index - 1, all) === "letter" && above.depth < open.depth ? "roman" : "letter";
@@ -77,7 +79,7 @@ const followsLetter = (raw: string, context: NumberingContext): boolean =>
77
79
 
78
80
  /** "(i)" is a roman numeral right under "(a)", or when a roman list is already open; the letter i otherwise. */
79
81
  const styleOf = (raw: string, context: NumberingContext): Style => {
80
- if (/^\d+$/u.test(raw)) return "digit";
82
+ if (/^\d+$/u.test(raw) || IS_INSERTED.test(raw)) return "digit";
81
83
  if (MULTI_ROMAN.test(raw)) return "roman";
82
84
  if (!/^[ivx]$/u.test(raw) || followsLetter(raw, context)) return "letter";
83
85
  const open = styles(context);
@@ -98,7 +100,7 @@ const LETTER_BEFORE_A = "a".charCodeAt(0) - 1;
98
100
 
99
101
  /** "(b)" は 2 番目、"(ii)" も 2 番目。二文字以上の英字("(aa)")は並びが決まらないので付けない。 */
100
102
  const ordinalOf = (raw: string, style: Style): number | undefined => {
101
- if (style === "digit") return Number(raw);
103
+ if (style === "digit") return IS_INSERTED.test(raw) ? undefined : Number(raw);
102
104
  if (style === "roman") return parseRoman(raw);
103
105
  return raw.length === 1 ? raw.charCodeAt(0) - LETTER_BEFORE_A : undefined;
104
106
  };
@@ -135,33 +137,17 @@ const chapter = (pattern: RegExp, line: string, depth: number, prefix: string, w
135
137
  };
136
138
 
137
139
  const numbered = (line: string, context: NumberingContext): NumberedLine | undefined =>
138
- chapter(PART, line, -2, "pt", "Part") ??
139
- chapter(CHAPTER, line, -1, "ch", "Chapter") ??
140
- headed(ARTICLE, line, (n) => `Article ${n}`) ??
141
- headed(SECTION, line, (n) => `Section ${n}`) ??
140
+ chapter(PART, line, PART_DEPTH, "pt", "Part") ??
141
+ chapter(CHAPTER, line, CHAPTER_DEPTH, "ch", "Chapter") ??
142
+ headed(ARTICLE, line, "article", (n) => `Article ${n}`) ??
143
+ headed(SECTION, line, "section", (n) => `Section ${n}`) ??
142
144
  lettered(line, context);
143
145
 
144
- const mentions = (
145
- pattern: RegExp,
146
- text: string,
147
- attrs: (groups: Readonly<Record<string, string | undefined>>, whole: string) => Mention["attrs"] | undefined,
148
- ): Mention[] =>
149
- [...text.matchAll(pattern)].flatMap((match) => {
150
- const found = attrs(match.groups ?? {}, match[0]);
151
- return found === undefined ? [] : [{ start: match.index, end: match.index + match[0].length, attrs: found }];
152
- });
153
-
154
- const DEFINITIONS = [
155
- /["“](?<term>[^"”\n]{1,60})["”] (?:means|shall mean|refers to|has the meaning)\b/gu,
156
- /\((?:the |hereinafter )?["“](?<term>[^"”\n]{1,60})["”]\)/gu,
157
- /\(hereinafter referred to as ["“](?<term>[^"”\n]{1,60})["”]\)/gu,
158
- ];
159
-
160
- const definitions = (text: string): Mention[] =>
161
- DEFINITIONS.flatMap((pattern) => mentions(pattern, text, (groups) => (groups["term"] === undefined ? undefined : { term: groups["term"] })));
162
-
163
- const REFERENCE = /\b(?:Sections?|Articles?|§) ?(?<n>\d{1,3}(?:\.\d{1,3}){0,5}|[IVXLC]{1,7})\b/gu;
164
- const SUBDIVISION = /^\((?<p>[a-z0-9]{1,4})\)/u;
146
+ const REFERENCE = /(?<word>\b[Ss]ections?|\b[Aa]rticles?|§) ?(?<n>\d{1,3}(?:\.\d{1,3}){0,5}|[IVXLC]{1,7})\b/gu;
147
+ /** "(a)", "(ii)", "(3)", and an inserted "(A1)" or "(2A)": the same labels the tree reads. */
148
+ const SUBDIVISION = new RegExp(`^\\((?<p>[a-z0-9]{1,4}|${INSERTED})\\)`, "u");
149
+ /** The longest label, with its parentheses: "(ZZ999)". */
150
+ const MAX_SUBDIVISION_LENGTH = 7;
165
151
 
166
152
  /** "(a)(ii)(3)" is as deep as a reference goes; more parentheses are text, not a deeper address. */
167
153
  const MAX_SUBDIVISIONS = 4;
@@ -174,7 +160,7 @@ const subdivisions = (text: string, from: number): { readonly parts: readonly st
174
160
  const parts: string[] = [];
175
161
  let end = from;
176
162
  while (parts.length < MAX_SUBDIVISIONS) {
177
- const part = SUBDIVISION.exec(text.slice(end, end + 6))?.groups?.["p"];
163
+ const part = SUBDIVISION.exec(text.slice(end, end + MAX_SUBDIVISION_LENGTH))?.groups?.["p"];
178
164
  if (part === undefined) break;
179
165
  parts.push(part);
180
166
  end += part.length + 2;
@@ -182,19 +168,51 @@ const subdivisions = (text: string, from: number): { readonly parts: readonly st
182
168
  return { parts, end };
183
169
  };
184
170
 
171
+ /** Parentheses opened and not yet closed in `between`. */
172
+ const PAREN_STEP: Readonly<Record<string, number>> = { "(": 1, ")": -1 };
173
+
174
+ /**
175
+ * "section 120(3) of the Communications Act 2003 (conditions under section 120 …)": a gloss in parentheses after
176
+ * a reference into another document describes that document, so the references in it are into it too — until the
177
+ * parenthesis the reference stood in closes. One pass over the line, however many references it holds.
178
+ */
179
+ type Gloss = { depth: number; scanned: number; readonly anchors: { readonly document: string; readonly depth: number }[] };
180
+
181
+ const advance = (gloss: Gloss, text: string, to: number): void => {
182
+ for (let index = gloss.scanned; index < to; index += 1) {
183
+ gloss.depth += PAREN_STEP[text[index] ?? ""] ?? 0;
184
+ while ((gloss.anchors.at(-1)?.depth ?? -Infinity) > gloss.depth) gloss.anchors.pop();
185
+ }
186
+ gloss.scanned = Math.max(gloss.scanned, to);
187
+ };
188
+
189
+ const glossedDocument = (gloss: Gloss): string | undefined => {
190
+ const anchor = gloss.anchors.at(-1);
191
+ return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
192
+ };
193
+
185
194
  /**
186
195
  * "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
187
196
  * "Section 9 of the Master Agreement" carries the other document's name, and is not looked up in this tree.
188
197
  */
189
- const references = (text: string): Mention[] =>
190
- [...text.matchAll(REFERENCE)].flatMap((match) => {
198
+ const references = (text: string): Mention[] => {
199
+ const gloss: Gloss = { depth: 0, scanned: 0, anchors: [] };
200
+ return [...text.matchAll(REFERENCE)].flatMap((match) => {
191
201
  const main = numberOf(match.groups?.["n"]);
192
202
  if (main === undefined) return [];
193
203
  const { parts, end } = subdivisions(text, match.index + match[0].length);
194
- const document = citedDocumentAfter(text, end);
195
- const attrs = { target: [main, ...parts].join("."), label: text.slice(match.index, end), ...(document === undefined ? {} : { document }) };
196
- return [{ start: match.index, end, attrs }];
204
+ advance(gloss, text, match.index);
205
+ const cited = citedDocumentAfter(text, end);
206
+ const document = cited ?? glossedDocument(gloss);
207
+ gloss.scanned = Math.max(gloss.scanned, end);
208
+ if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
209
+ const numbering = /^[Aa]/u.test(match.groups?.["word"] ?? "") ? "article" : "section";
210
+ const shared = { numbering, ...(document === undefined ? {} : { document }) };
211
+ const first = { start: match.index, end, attrs: { target: [main, ...parts].join("."), label: text.slice(match.index, end), ...shared } };
212
+ const [plural, roman] = [/s$/u.test(match.groups?.["word"] ?? ""), /^[IVXLC]+$/u.test(match.groups?.["n"] ?? "")];
213
+ return [first, ...membersAfter(text, end, [main, ...parts], shared, plural, roman)];
197
214
  });
215
+ };
198
216
 
199
217
  /** Longest first, and never inside a word: "shall not" is not also "shall", "mayor" is not "may". */
200
218
  const MARKERS: readonly (readonly [string, "must" | "must-not" | "may"])[] = [
@@ -304,63 +322,17 @@ const quantities = (text: string): Mention[] =>
304
322
  return unit === undefined || Number.isNaN(value) || isWordChar(text[match.index - 1]) ? [] : [{ start: match.index, end, attrs: { value, unit } }];
305
323
  });
306
324
 
307
- const MONTHS = ["january", "february", "march", "april", "may", "june", "july", "august", "september", "october", "november", "december"];
308
- const MONTH_WORD = /\b(?<month>[A-Z][a-z]{2,8})\b/gu;
309
- const ISO_DATE = /\b(?<y>\d{4})-(?<m>\d{2})-(?<d>\d{2})\b/gu;
310
- const DAY_BEFORE = /(?<d>\d{1,2})(?:st|nd|rd|th)? $/u;
311
- const DAY_YEAR_AFTER = /^ (?<d>\d{1,2})(?:st|nd|rd|th)?,? (?<y>\d{4})\b/u;
312
- const YEAR_AFTER = /^,? (?<y>\d{4})\b/u;
313
-
314
- const pad = (value: string): string => value.padStart(2, "0");
315
-
316
- /** "1 April 2024": the day written before the month. */
317
- const dayBefore = (text: string, at: number): { readonly day: string; readonly start: number } | undefined => {
318
- const found = DAY_BEFORE.exec(text.slice(Math.max(0, at - 6), at));
319
- const day = found?.groups?.["d"];
320
- return found === null || day === undefined ? undefined : { day, start: at - found[0].length };
321
- };
322
-
323
- /** "April 1, 2024" → 2024-04-01. */
324
- const monthDayYear = (text: string, at: number, end: number, month: number): Mention | undefined => {
325
- const found = DAY_YEAR_AFTER.exec(text.slice(end, end + 16));
326
- if (found?.groups === undefined) return undefined;
327
- const value = `${found.groups["y"] ?? ""}-${pad(String(month))}-${pad(found.groups["d"] ?? "")}`;
328
- return { start: at, end: end + found[0].length, attrs: { value } };
329
- };
330
-
331
- /** "1 April 2024" → 2024-04-01, "April 2024" → 2024-04. */
332
- const monthYear = (text: string, at: number, end: number, month: number): Mention | undefined => {
333
- const found = YEAR_AFTER.exec(text.slice(end, end + 8));
334
- const year = found?.groups?.["y"];
335
- if (found === null || year === undefined) return undefined;
336
- const before = dayBefore(text, at);
337
- const value = [year, pad(String(month)), ...(before === undefined ? [] : [pad(before.day)])].join("-");
338
- return { start: before?.start ?? at, end: end + found[0].length, attrs: { value } };
339
- };
340
-
341
- /**
342
- * A month name alone is not a date: "May" is also the modal verb, so it counts only with a year beside it.
343
- * The month is found first and its neighbours read with anchored patterns, never one long alternation.
344
- */
345
- const namedDate = (text: string, match: RegExpExecArray): Mention | undefined => {
346
- const month = MONTHS.indexOf((match.groups?.["month"] ?? "").toLowerCase()) + 1;
347
- if (month === 0) return undefined;
348
- const end = match.index + match[0].length;
349
- return monthDayYear(text, match.index, end, month) ?? monthYear(text, match.index, end, month);
350
- };
351
-
352
- const isoDate = (match: RegExpExecArray): Mention => ({
353
- start: match.index,
354
- end: match.index + match[0].length,
355
- attrs: { value: `${match.groups?.["y"] ?? ""}-${match.groups?.["m"] ?? ""}-${match.groups?.["d"] ?? ""}` },
356
- });
357
-
358
- const dates = (text: string): Mention[] =>
359
- [...[...text.matchAll(MONTH_WORD)].flatMap((match) => namedDate(text, match) ?? []), ...[...text.matchAll(ISO_DATE)].map(isoDate)].sort(
360
- (left, right) => left.start - right.start,
361
- );
362
-
363
325
  /** "2.5 days" and "1.5 times" are amounts, not section 2.5 titled "days". */
364
326
  const countedAfter = (_number: string, rest: string): boolean => unitAfter(` ${rest}`, 0) !== undefined;
365
327
 
366
- export const structure: StructurePatterns = { numbered, definitions, references, obligations, quantities, dates, countedAfter };
328
+ export const structure: StructurePatterns = {
329
+ numbered,
330
+ definitions,
331
+ references,
332
+ obligations,
333
+ quantities,
334
+ dates,
335
+ countedAfter,
336
+ opensDefinitionScope,
337
+ definitionScopeDepth,
338
+ };