@chaffjs/lang-en 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,393 @@
1
+ import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
2
+ import { citedDocumentAfter } from "./citation.ts";
3
+ import { membersAfter } from "./reference-list.ts";
4
+ import { parseRoman } from "./roman.ts";
5
+ import { definitionScopeDepth, definitions, opensDefinitionScope } from "./definitions.ts";
6
+ import { CHAPTER_DEPTH, PART_DEPTH } from "./depth.ts";
7
+
8
+ // Contracts, specifications and statutes in English. core nests what this reads; it does not know
9
+ // how English numbers its articles.
10
+
11
+ const numberOf = (text: string | undefined): string | undefined => {
12
+ if (text === undefined) return undefined;
13
+ if (/^\d{1,3}(?:\.\d{1,3}){0,5}$/u.test(text)) return text;
14
+ const roman = parseRoman(text);
15
+ return roman === undefined ? undefined : String(roman);
16
+ };
17
+
18
+ /**
19
+ * A heading is the number followed by nothing, punctuation, or a capitalised title.
20
+ * "Article 3 shall apply to…" at the start of a line is a sentence, not the heading of Article 3.
21
+ */
22
+ const titleOf = (rest: string): string | undefined => {
23
+ const trimmed = rest.replace(/^[.:\-–—\s]+/u, "").trim();
24
+ if (trimmed !== "" && /^\p{Ll}/u.test(trimmed)) return undefined;
25
+ return trimmed;
26
+ };
27
+
28
+ const ARTICLE = /^\s{0,3}(?:ARTICLE|Article)\s+(?<n>\d{1,3}|[IVXLC]{1,7})\b(?<rest>.*)$/u;
29
+ const SECTION = /^\s{0,3}(?:SECTION|Section|§)\s*(?<n>\d{1,3}(?:\.\d{1,3}){0,5})\b(?<rest>.*)$/u;
30
+ /**
31
+ * An amendment inserts a subsection between two others and numbers it "(A1)" or "(2A)". It is a subsection, written
32
+ * outside the sequence: it has no ordinal, so "(1)" after "(A1)" is still the first.
33
+ */
34
+ const INSERTED = "\\d{1,3}[A-Z]{1,2}|[A-Z]{1,2}\\d{1,3}";
35
+ const LETTERED = new RegExp(`^\\s{0,6}\\((?<n>[a-z]{1,4}|\\d{1,3}|${INSERTED})\\)\\s+(?<rest>\\S.*)$`, "u");
36
+ const IS_INSERTED = new RegExp(`^(?:${INSERTED})$`, "u");
37
+ const MULTI_ROMAN = /^(?:ii|iii|iv|vi|vii|viii|ix)$/u;
38
+
39
+ const headed = (pattern: RegExp, line: string, numbering: string, label: (n: string) => string): NumberedLine | undefined => {
40
+ const groups = pattern.exec(line)?.groups;
41
+ const number = numberOf(groups?.["n"]);
42
+ const heading = titleOf(groups?.["rest"] ?? "");
43
+ if (groups === undefined || number === undefined || heading === undefined) return undefined;
44
+ const parts = number.split(".");
45
+ return {
46
+ kind: "article",
47
+ depth: parts.length,
48
+ number,
49
+ absolute: true,
50
+ label: label(groups["n"] ?? ""),
51
+ heading,
52
+ rest: heading,
53
+ ordinal: Number(parts.at(-1)),
54
+ numbering,
55
+ };
56
+ };
57
+
58
+ type Style = "letter" | "roman" | "digit";
59
+
60
+ /**
61
+ * 開いている項目の書き方。"(ii)" はローマ数字、"(b)" は英字。一文字の "(i)" はどちらにも読めるので、
62
+ * 開いたときの読みを覚えておく代わりに、一つ上に英字が開いていたかで決め直す。
63
+ */
64
+ const styleOfLabel = (open: NumberedLine, index: number, all: readonly NumberedLine[]): Style | undefined => {
65
+ const inner = /^\((?<n>[A-Za-z0-9]{1,5})\)$/u.exec(open.label)?.groups?.["n"];
66
+ if (inner === undefined) return undefined;
67
+ if (/^\d+$/u.test(inner) || IS_INSERTED.test(inner)) return "digit";
68
+ if (MULTI_ROMAN.test(inner)) return "roman";
69
+ const above = all[index - 1];
70
+ return /^[ivx]$/u.test(inner) && above !== undefined && styleOfLabel(above, index - 1, all) === "letter" && above.depth < open.depth ? "roman" : "letter";
71
+ };
72
+
73
+ const styles = (context: NumberingContext): (Style | undefined)[] => context.open.map((open, index, all) => styleOfLabel(open, index, all));
74
+
75
+ /** "(h)" の次の "(i)" は英字。開いている英字の次の文字なら、ローマ数字とは読まない。 */
76
+ const followsLetter = (raw: string, context: NumberingContext): boolean =>
77
+ context.open.some((open, index) => styles(context)[index] === "letter" && open.number.charCodeAt(0) + 1 === raw.charCodeAt(0));
78
+
79
+ /** "(i)" is a roman numeral right under "(a)", or when a roman list is already open; the letter i otherwise. */
80
+ const styleOf = (raw: string, context: NumberingContext): Style => {
81
+ if (/^\d+$/u.test(raw) || IS_INSERTED.test(raw)) return "digit";
82
+ if (MULTI_ROMAN.test(raw)) return "roman";
83
+ if (!/^[ivx]$/u.test(raw) || followsLetter(raw, context)) return "letter";
84
+ const open = styles(context);
85
+ return open.includes("roman") || open.at(-1) === "letter" ? "roman" : "letter";
86
+ };
87
+
88
+ /**
89
+ * A sibling has the depth of the open item written the same way: "(b)" closes "(i)" and sits beside "(a)".
90
+ * A new way of numbering goes one deeper than whatever is open.
91
+ */
92
+ const depthFor = (style: Style, context: NumberingContext): number => {
93
+ const open = styles(context);
94
+ const sibling = [...context.open].reverse().find((_item, reversed) => open[context.open.length - 1 - reversed] === style);
95
+ return sibling?.depth ?? (context.open.at(-1)?.depth ?? 0) + 1;
96
+ };
97
+
98
+ const LETTER_BEFORE_A = "a".charCodeAt(0) - 1;
99
+
100
+ /** "(b)" は 2 番目、"(ii)" も 2 番目。二文字以上の英字("(aa)")は並びが決まらないので付けない。 */
101
+ const ordinalOf = (raw: string, style: Style): number | undefined => {
102
+ if (style === "digit") return IS_INSERTED.test(raw) ? undefined : Number(raw);
103
+ if (style === "roman") return parseRoman(raw);
104
+ return raw.length === 1 ? raw.charCodeAt(0) - LETTER_BEFORE_A : undefined;
105
+ };
106
+
107
+ const lettered = (line: string, context: NumberingContext): NumberedLine | undefined => {
108
+ const groups = LETTERED.exec(line)?.groups;
109
+ const raw = groups?.["n"];
110
+ if (groups === undefined || raw === undefined) return undefined;
111
+ const style = styleOf(raw, context);
112
+ // 番地は書かれたままの "ii" を使う。参照「Section 4.2(a)(ii)」も同じ形で書かれるので、そのまま引ける。
113
+ return {
114
+ kind: "item",
115
+ depth: depthFor(style, context),
116
+ ordinal: ordinalOf(raw, style),
117
+ number: raw,
118
+ absolute: false,
119
+ label: `(${raw})`,
120
+ heading: "",
121
+ rest: groups["rest"]?.trim() ?? "",
122
+ };
123
+ };
124
+
125
+ /** Chapters restart in each part, so a chapter continues its part's address: PART II, CHAPTER 1 → pt2.ch1. */
126
+ const PART = /^\s{0,3}(?:PART|Part)\s+(?<n>\d{1,3}|[IVXLC]{1,7})\b(?<rest>.*)$/u;
127
+ const CHAPTER = /^\s{0,3}(?:CHAPTER|Chapter)\s+(?<n>\d{1,3}|[IVXLC]{1,7})\b(?<rest>.*)$/u;
128
+
129
+ const chapter = (pattern: RegExp, line: string, depth: number, prefix: string, word: string): NumberedLine | undefined => {
130
+ const groups = pattern.exec(line)?.groups;
131
+ const number = numberOf(groups?.["n"]);
132
+ const heading = titleOf(groups?.["rest"] ?? "");
133
+ if (groups === undefined || number === undefined || heading === undefined) return undefined;
134
+ const label = `${word} ${groups["n"] ?? ""}`;
135
+ return { kind: "chapter", depth, number: prefix + number, absolute: false, label, heading, rest: heading, ordinal: Number(number) };
136
+ };
137
+
138
+ const numbered = (line: string, context: NumberingContext): NumberedLine | undefined =>
139
+ chapter(PART, line, PART_DEPTH, "pt", "Part") ??
140
+ chapter(CHAPTER, line, CHAPTER_DEPTH, "ch", "Chapter") ??
141
+ headed(ARTICLE, line, "article", (n) => `Article ${n}`) ??
142
+ headed(SECTION, line, "section", (n) => `Section ${n}`) ??
143
+ lettered(line, context);
144
+
145
+ const REFERENCE = /(?<word>\b[Ss]ections?|\b[Aa]rticles?|§) ?(?<n>\d{1,3}(?:\.\d{1,3}){0,5}|[IVXLC]{1,7})\b/gu;
146
+ /** "(a)", "(ii)", "(3)", and an inserted "(A1)" or "(2A)": the same labels the tree reads. */
147
+ const SUBDIVISION = new RegExp(`^\\((?<p>[a-z0-9]{1,4}|${INSERTED})\\)`, "u");
148
+ /** The longest label, with its parentheses: "(ZZ999)". */
149
+ const MAX_SUBDIVISION_LENGTH = 7;
150
+
151
+ /** "(a)(ii)(3)" is as deep as a reference goes; more parentheses are text, not a deeper address. */
152
+ const MAX_SUBDIVISIONS = 4;
153
+
154
+ /**
155
+ * "(a)(ii)" のような続きの括弧を、正規表現を複雑にせずに一つずつ読む。
156
+ * 再帰にしないのは、括弧が延々と続く行でスタックを使い切らないため。
157
+ */
158
+ const subdivisions = (text: string, from: number): { readonly parts: readonly string[]; readonly end: number } => {
159
+ const parts: string[] = [];
160
+ let end = from;
161
+ while (parts.length < MAX_SUBDIVISIONS) {
162
+ const part = SUBDIVISION.exec(text.slice(end, end + MAX_SUBDIVISION_LENGTH))?.groups?.["p"];
163
+ if (part === undefined) break;
164
+ parts.push(part);
165
+ end += part.length + 2;
166
+ }
167
+ return { parts, end };
168
+ };
169
+
170
+ /** Parentheses opened and not yet closed in `between`. */
171
+ const PAREN_STEP: Readonly<Record<string, number>> = { "(": 1, ")": -1 };
172
+
173
+ /**
174
+ * "section 120(3) of the Communications Act 2003 (conditions under section 120 …)": a gloss in parentheses after
175
+ * a reference into another document describes that document, so the references in it are into it too — until the
176
+ * parenthesis the reference stood in closes. One pass over the line, however many references it holds.
177
+ */
178
+ type Gloss = { depth: number; scanned: number; readonly anchors: { readonly document: string; readonly depth: number }[] };
179
+
180
+ const advance = (gloss: Gloss, text: string, to: number): void => {
181
+ for (let index = gloss.scanned; index < to; index += 1) {
182
+ gloss.depth += PAREN_STEP[text[index] ?? ""] ?? 0;
183
+ while ((gloss.anchors.at(-1)?.depth ?? -Infinity) > gloss.depth) gloss.anchors.pop();
184
+ }
185
+ gloss.scanned = Math.max(gloss.scanned, to);
186
+ };
187
+
188
+ const glossedDocument = (gloss: Gloss): string | undefined => {
189
+ const anchor = gloss.anchors.at(-1);
190
+ return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
191
+ };
192
+
193
+ /**
194
+ * "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
195
+ * "Section 9 of the Master Agreement" carries the other document's name, and is not looked up in this tree.
196
+ */
197
+ const references = (text: string): Mention[] => {
198
+ const gloss: Gloss = { depth: 0, scanned: 0, anchors: [] };
199
+ return [...text.matchAll(REFERENCE)].flatMap((match) => {
200
+ const main = numberOf(match.groups?.["n"]);
201
+ if (main === undefined) return [];
202
+ const { parts, end } = subdivisions(text, match.index + match[0].length);
203
+ advance(gloss, text, match.index);
204
+ const cited = citedDocumentAfter(text, end);
205
+ const document = cited ?? glossedDocument(gloss);
206
+ gloss.scanned = Math.max(gloss.scanned, end);
207
+ if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
208
+ const numbering = /^[Aa]/u.test(match.groups?.["word"] ?? "") ? "article" : "section";
209
+ const shared = { numbering, ...(document === undefined ? {} : { document }) };
210
+ const first = { start: match.index, end, attrs: { target: [main, ...parts].join("."), label: text.slice(match.index, end), ...shared } };
211
+ const [plural, roman] = [/s$/u.test(match.groups?.["word"] ?? ""), /^[IVXLC]+$/u.test(match.groups?.["n"] ?? "")];
212
+ return [first, ...membersAfter(text, end, [main, ...parts], shared, plural, roman)];
213
+ });
214
+ };
215
+
216
+ /** Longest first, and never inside a word: "shall not" is not also "shall", "mayor" is not "may". */
217
+ const MARKERS: readonly (readonly [string, "must" | "must-not" | "may"])[] = [
218
+ ["is required to", "must"],
219
+ ["shall not", "must-not"],
220
+ ["must not", "must-not"],
221
+ ["may not", "must-not"],
222
+ ["agrees to", "must"],
223
+ ["shall", "must"],
224
+ ["must", "must"],
225
+ ["may", "may"],
226
+ ];
227
+
228
+ const isWordChar = (char: string | undefined): boolean => char !== undefined && /[\p{L}\p{N}_]/u.test(char);
229
+
230
+ /** Every whole-word occurrence. A loop, not recursion: a line with thousands of "may" must not exhaust the stack. */
231
+ const wordAt = (lower: string, word: string): number[] => {
232
+ const found: number[] = [];
233
+ for (let at = lower.indexOf(word); at !== -1; at = lower.indexOf(word, at + 1)) {
234
+ if (!isWordChar(lower[at - 1]) && !isWordChar(lower[at + word.length])) found.push(at);
235
+ }
236
+ return found;
237
+ };
238
+
239
+ const SENTENCE_OPENERS = new Set([".", "!", "?", ":", ";", '"', "“", "("]);
240
+
241
+ /** "May" in the middle of a sentence is the month: "published in May 2023". At the start it can be the modal. */
242
+ const isMonthName = (text: string, at: number): boolean => {
243
+ if (!text.startsWith("May", at)) return false;
244
+ const before = text.slice(Math.max(0, at - 4), at).trimEnd();
245
+ return before !== "" && !SENTENCE_OPENERS.has(before.at(-1) ?? "");
246
+ };
247
+
248
+ /**
249
+ * Longest marker first; a hit that overlaps one already kept is dropped. Overlap is checked with a mark
250
+ * per character, not by comparing every pair, which would go quadratic on a line with thousands of markers.
251
+ */
252
+ const obligations = (text: string): Mention[] => {
253
+ const lower = text.toLowerCase();
254
+ const taken = new Uint8Array(text.length);
255
+ const kept: Mention[] = [];
256
+ MARKERS.forEach(([marker, type]) => {
257
+ wordAt(lower, marker).forEach((start) => {
258
+ const end = start + marker.length;
259
+ if (isMonthName(text, start)) return;
260
+ if (taken.subarray(start, end).some((mark) => mark === 1)) return;
261
+ taken.fill(1, start, end);
262
+ kept.push({ start, end, attrs: { marker, type } });
263
+ });
264
+ });
265
+ return kept.sort((left, right) => left.start - right.start);
266
+ };
267
+
268
+ /** Find the number first, then look at what is right before (a currency) and right after (a unit). */
269
+ const NUMBER_RUN = /\d[\d,.]{0,15}/gu;
270
+ const UNITS = [
271
+ "business days",
272
+ "business day",
273
+ "calendar days",
274
+ "calendar day",
275
+ "days",
276
+ "day",
277
+ "weeks",
278
+ "week",
279
+ "months",
280
+ "month",
281
+ "years",
282
+ "year",
283
+ "hours",
284
+ "hour",
285
+ "minutes",
286
+ "minute",
287
+ "percent",
288
+ "times",
289
+ "%",
290
+ ];
291
+ const CURRENCIES = ["USD", "EUR", "$", "€", "£"];
292
+
293
+ /** "30." の終わりの点は文の終わり。数の一部にしない。 */
294
+ const withoutTrailingPunctuation = (run: string): string => {
295
+ let end = run.length;
296
+ while (end > 0 && (run[end - 1] === "." || run[end - 1] === ",")) end -= 1;
297
+ return run.slice(0, end);
298
+ };
299
+
300
+ /** One space or tab may sit between a number and its unit or currency. */
301
+ const isGap = (char: string | undefined): boolean => char === " " || char === "\t";
302
+
303
+ const unitAfter = (text: string, from: number): string | undefined => {
304
+ const gap = isGap(text[from]) ? 1 : 0;
305
+ const unit = UNITS.find((candidate) => text.startsWith(candidate, from + gap) && !isWordChar(text[from + gap + candidate.length]));
306
+ return unit;
307
+ };
308
+
309
+ /** The currency just before the number, allowing one space. Looks back a few characters, never at the whole line. */
310
+ const currencyBefore = (text: string, at: number): string | undefined => {
311
+ const end = isGap(text[at - 1]) ? at - 1 : at;
312
+ return CURRENCIES.find((currency) => text.startsWith(currency, end - currency.length));
313
+ };
314
+
315
+ const quantities = (text: string): Mention[] =>
316
+ [...text.matchAll(NUMBER_RUN)].flatMap((match) => {
317
+ const digits = withoutTrailingPunctuation(match[0]);
318
+ const value = Number(digits.replace(/,/gu, ""));
319
+ const end = match.index + digits.length;
320
+ const unit = unitAfter(text, end) ?? currencyBefore(text, match.index);
321
+ return unit === undefined || Number.isNaN(value) || isWordChar(text[match.index - 1]) ? [] : [{ start: match.index, end, attrs: { value, unit } }];
322
+ });
323
+
324
+ const MONTHS = ["january", "february", "march", "april", "may", "june", "july", "august", "september", "october", "november", "december"];
325
+ const MONTH_WORD = /\b(?<month>[A-Z][a-z]{2,8})\b/gu;
326
+ const ISO_DATE = /\b(?<y>\d{4})-(?<m>\d{2})-(?<d>\d{2})\b/gu;
327
+ const DAY_BEFORE = /(?<d>\d{1,2})(?:st|nd|rd|th)? $/u;
328
+ const DAY_YEAR_AFTER = /^ (?<d>\d{1,2})(?:st|nd|rd|th)?,? (?<y>\d{4})\b/u;
329
+ const YEAR_AFTER = /^,? (?<y>\d{4})\b/u;
330
+
331
+ const pad = (value: string): string => value.padStart(2, "0");
332
+
333
+ /** "1 April 2024": the day written before the month. */
334
+ const dayBefore = (text: string, at: number): { readonly day: string; readonly start: number } | undefined => {
335
+ const found = DAY_BEFORE.exec(text.slice(Math.max(0, at - 6), at));
336
+ const day = found?.groups?.["d"];
337
+ return found === null || day === undefined ? undefined : { day, start: at - found[0].length };
338
+ };
339
+
340
+ /** "April 1, 2024" → 2024-04-01. */
341
+ const monthDayYear = (text: string, at: number, end: number, month: number): Mention | undefined => {
342
+ const found = DAY_YEAR_AFTER.exec(text.slice(end, end + 16));
343
+ if (found?.groups === undefined) return undefined;
344
+ const value = `${found.groups["y"] ?? ""}-${pad(String(month))}-${pad(found.groups["d"] ?? "")}`;
345
+ return { start: at, end: end + found[0].length, attrs: { value } };
346
+ };
347
+
348
+ /** "1 April 2024" → 2024-04-01, "April 2024" → 2024-04. */
349
+ const monthYear = (text: string, at: number, end: number, month: number): Mention | undefined => {
350
+ const found = YEAR_AFTER.exec(text.slice(end, end + 8));
351
+ const year = found?.groups?.["y"];
352
+ if (found === null || year === undefined) return undefined;
353
+ const before = dayBefore(text, at);
354
+ const value = [year, pad(String(month)), ...(before === undefined ? [] : [pad(before.day)])].join("-");
355
+ return { start: before?.start ?? at, end: end + found[0].length, attrs: { value } };
356
+ };
357
+
358
+ /**
359
+ * A month name alone is not a date: "May" is also the modal verb, so it counts only with a year beside it.
360
+ * The month is found first and its neighbours read with anchored patterns, never one long alternation.
361
+ */
362
+ const namedDate = (text: string, match: RegExpExecArray): Mention | undefined => {
363
+ const month = MONTHS.indexOf((match.groups?.["month"] ?? "").toLowerCase()) + 1;
364
+ if (month === 0) return undefined;
365
+ const end = match.index + match[0].length;
366
+ return monthDayYear(text, match.index, end, month) ?? monthYear(text, match.index, end, month);
367
+ };
368
+
369
+ const isoDate = (match: RegExpExecArray): Mention => ({
370
+ start: match.index,
371
+ end: match.index + match[0].length,
372
+ attrs: { value: `${match.groups?.["y"] ?? ""}-${match.groups?.["m"] ?? ""}-${match.groups?.["d"] ?? ""}` },
373
+ });
374
+
375
+ const dates = (text: string): Mention[] =>
376
+ [...[...text.matchAll(MONTH_WORD)].flatMap((match) => namedDate(text, match) ?? []), ...[...text.matchAll(ISO_DATE)].map(isoDate)].sort(
377
+ (left, right) => left.start - right.start,
378
+ );
379
+
380
+ /** "2.5 days" and "1.5 times" are amounts, not section 2.5 titled "days". */
381
+ const countedAfter = (_number: string, rest: string): boolean => unitAfter(` ${rest}`, 0) !== undefined;
382
+
383
+ export const structure: StructurePatterns = {
384
+ numbered,
385
+ definitions,
386
+ references,
387
+ obligations,
388
+ quantities,
389
+ dates,
390
+ countedAfter,
391
+ opensDefinitionScope,
392
+ definitionScopeDepth,
393
+ };