@chaffjs/lang-en 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/citation.d.ts +2 -0
- package/dist/citation.d.ts.map +1 -1
- package/dist/citation.js +37 -12
- package/dist/citation.js.map +1 -1
- package/dist/closing-quote.d.ts +6 -0
- package/dist/closing-quote.d.ts.map +1 -0
- package/dist/closing-quote.js +50 -0
- package/dist/closing-quote.js.map +1 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -2
- package/dist/index.js.map +1 -1
- package/dist/item-style.d.ts +30 -0
- package/dist/item-style.d.ts.map +1 -0
- package/dist/item-style.js +124 -0
- package/dist/item-style.js.map +1 -0
- package/dist/long-runs.d.ts +6 -0
- package/dist/long-runs.d.ts.map +1 -0
- package/dist/long-runs.js +6 -0
- package/dist/long-runs.js.map +1 -0
- package/dist/pos.d.ts.map +1 -1
- package/dist/pos.js +32 -5
- package/dist/pos.js.map +1 -1
- package/dist/proper-noun.d.ts +21 -0
- package/dist/proper-noun.d.ts.map +1 -1
- package/dist/proper-noun.js +37 -0
- package/dist/proper-noun.js.map +1 -1
- package/dist/quoted-stop.d.ts +13 -0
- package/dist/quoted-stop.d.ts.map +1 -0
- package/dist/quoted-stop.js +45 -0
- package/dist/quoted-stop.js.map +1 -0
- package/dist/sentence-split.d.ts +2 -0
- package/dist/sentence-split.d.ts.map +1 -1
- package/dist/sentence-split.js +3 -1
- package/dist/sentence-split.js.map +1 -1
- package/dist/structure.d.ts.map +1 -1
- package/dist/structure.js +24 -65
- package/dist/structure.js.map +1 -1
- package/lexicons/common-acronym.yaml +4 -0
- package/lexicons/date-time-unit.yaml +6 -0
- package/lexicons/emphasis-word.yaml +1 -0
- package/lexicons/fixed-phrase.yaml +14 -0
- package/lexicons/honorific.yaml +15 -0
- package/lexicons/http-method.yaml +14 -0
- package/lexicons/quantity-noun.yaml +6 -0
- package/package.json +1 -1
- package/src/citation.ts +41 -13
- package/src/closing-quote.ts +52 -0
- package/src/index.ts +5 -2
- package/src/item-style.ts +130 -0
- package/src/long-runs.ts +6 -0
- package/src/pos.ts +40 -5
- package/src/proper-noun.ts +43 -0
- package/src/quoted-stop.ts +52 -0
- package/src/sentence-split.ts +3 -1
- package/src/structure.ts +23 -67
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import type { Span } from "chaffjs/plugin";
|
|
2
|
+
import { NEXT_SENTENCE_HEAD } from "./quoted-stop.ts";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* sentence-splitter は疑問符・感嘆符の後の曲がった閉じ引用符(“Is it done?” Nobody …)と直線の一重引用符の手前で文を切る。
|
|
6
|
+
* 次の文が「”」で始まり、前の文は引用符が開いたまま終わる。
|
|
7
|
+
*
|
|
8
|
+
* 前の文が句点で終わり、次の文がそこから間を置かずに閉じ引用符(閉じ括弧も)で始まるとき、それを前の文の末尾へ戻す。
|
|
9
|
+
* 戻した後に空白と大文字が続けば文は二つのまま、何も続かなければ前の文だけ、
|
|
10
|
+
* 小文字・数字・句読点が続けば引用は文の途中にあるので一つの文につなぐ(直線の二重引用符で分割器がそうするのと同じ)。
|
|
11
|
+
* 開き引用符(“ ‘)は閉じの分類に入らない。語頭のアポストロフィ(’Tis ’90s)は後ろに字が続くので閉じと読まない。
|
|
12
|
+
*/
|
|
13
|
+
const CLOSING_RUN = /^[\p{Pf}\p{Pe}"']+/u;
|
|
14
|
+
const WORD_CHARACTER = /^[\p{L}\p{N}]/u;
|
|
15
|
+
const STOP_AT_END = /[.?!]$/u;
|
|
16
|
+
// 語頭のアポストロフィの後の大文字(’Tis)も次の文の頭。
|
|
17
|
+
const NEXT_SENTENCE = new RegExp(String.raw`^(?:${NEXT_SENTENCE_HEAD}|\s+’\p{Lu})`, "u");
|
|
18
|
+
const LEADING_SPACE = /^\s*/u;
|
|
19
|
+
|
|
20
|
+
/** 文頭の閉じ引用符の並びの長さ。後ろに字が続けば開き引用符かアポストロフィなので 0。 */
|
|
21
|
+
export const closingRunLength = (sentence: string): number => {
|
|
22
|
+
const run = CLOSING_RUN.exec(sentence)?.[0].length ?? 0;
|
|
23
|
+
return run > 0 && !WORD_CHARACTER.test(sentence.slice(run)) ? run : 0;
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
const follows = (text: string, previous: Span, span: Span): boolean =>
|
|
27
|
+
previous.end === span.start && STOP_AT_END.test(text.slice(previous.start, previous.end));
|
|
28
|
+
|
|
29
|
+
/** previous と span を置き換える文。閉じ引用符が戻らなければ undefined。 */
|
|
30
|
+
const reattached = (text: string, previous: Span, span: Span): Span[] | undefined => {
|
|
31
|
+
const run = follows(text, previous, span) ? closingRunLength(text.slice(span.start, span.end)) : 0;
|
|
32
|
+
if (run === 0) return undefined;
|
|
33
|
+
const quoteEnd = span.start + run;
|
|
34
|
+
const rest = text.slice(quoteEnd, span.end);
|
|
35
|
+
if (rest.trim() === "") return [{ start: previous.start, end: quoteEnd }];
|
|
36
|
+
if (!NEXT_SENTENCE.test(rest)) return [{ start: previous.start, end: span.end }];
|
|
37
|
+
const next = quoteEnd + (LEADING_SPACE.exec(rest)?.[0].length ?? 0);
|
|
38
|
+
return [
|
|
39
|
+
{ start: previous.start, end: quoteEnd },
|
|
40
|
+
{ start: next, end: span.end },
|
|
41
|
+
];
|
|
42
|
+
};
|
|
43
|
+
|
|
44
|
+
/** 文の span の並び(text 上の位置)で、文頭に取り残された閉じ引用符を前の文へ戻したもの。 */
|
|
45
|
+
export const reattachClosingQuotes = (text: string, spans: readonly Span[]): Span[] =>
|
|
46
|
+
spans.reduce<Span[]>((sentences, span) => {
|
|
47
|
+
const previous = sentences.at(-1);
|
|
48
|
+
const replaced = previous === undefined ? undefined : reattached(text, previous, span);
|
|
49
|
+
if (replaced === undefined) sentences.push(span);
|
|
50
|
+
else sentences.splice(-1, 1, ...replaced);
|
|
51
|
+
return sentences;
|
|
52
|
+
}, []);
|
package/src/index.ts
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { loadLexicons } from "./lexicons.ts";
|
|
2
2
|
import { unmarkNumberStops } from "./number-stop.ts";
|
|
3
3
|
import { sentenceSpans } from "./sentence-split.ts";
|
|
4
|
+
import { splitAtQuotedStops } from "./quoted-stop.ts";
|
|
5
|
+
import { reattachClosingQuotes } from "./closing-quote.ts";
|
|
4
6
|
import { structure } from "./structure.ts";
|
|
5
7
|
import { isReady, prepare, tokenize } from "./pos.ts";
|
|
6
8
|
import type { AdapterNeeds, LanguageAdapter, Segmentation, Sentence } from "chaffjs/plugin";
|
|
@@ -25,7 +27,7 @@ const withTokens = (sentence: Sentence): Sentence => {
|
|
|
25
27
|
|
|
26
28
|
/**
|
|
27
29
|
* 英語は sentence-splitter の既定にほぼ任せる。"Dr." "e.g." "U.S." "$3.50" を
|
|
28
|
-
*
|
|
30
|
+
* いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませること、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
|
|
29
31
|
*/
|
|
30
32
|
export const adapter: LanguageAdapter = {
|
|
31
33
|
kind: "language",
|
|
@@ -52,7 +54,8 @@ export const adapter: LanguageAdapter = {
|
|
|
52
54
|
lexicons: loadLexicons(),
|
|
53
55
|
structure,
|
|
54
56
|
segment: (text: string): Segmentation => {
|
|
55
|
-
const
|
|
57
|
+
const quotedStops = sentenceSpans(unmarkNumberStops(text)).flatMap((span) => splitAtQuotedStops(text, span));
|
|
58
|
+
const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => ({ span, text: text.slice(span.start, span.end) }));
|
|
56
59
|
return { sentences: isReady() ? sentences.map(withTokens) : sentences };
|
|
57
60
|
},
|
|
58
61
|
};
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import type { NumberedLine, NumberingContext } from "chaffjs/plugin";
|
|
2
|
+
import { parseRoman } from "./roman.ts";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* An amendment inserts a subsection between two others and numbers it "(A1)" or "(2A)". It is a subsection, written
|
|
6
|
+
* outside the sequence: it has no ordinal, so "(1)" after "(A1)" is still the first.
|
|
7
|
+
*/
|
|
8
|
+
export const INSERTED = "\\d{1,3}[A-Z]{1,2}|[A-Z]{1,2}\\d{1,3}";
|
|
9
|
+
export const IS_INSERTED = new RegExp(`^(?:${INSERTED})$`, "u");
|
|
10
|
+
const MULTI_ROMAN = /^(?:ii|iii|iv|vi|vii|viii|ix)$/u;
|
|
11
|
+
const AMBIGUOUS = /^[ivx]$/u;
|
|
12
|
+
const DIGITS = /^\d+$/u;
|
|
13
|
+
const CAPITALS = /^[A-Z]+$/u;
|
|
14
|
+
const ONE_CAPITAL = /^[A-Z]$/u;
|
|
15
|
+
|
|
16
|
+
export type Style = "letter" | "roman" | "digit" | "capital" | "capital-roman";
|
|
17
|
+
|
|
18
|
+
/** 大文字で書いた同じ並び。"(B)" は英字の、"(II)" はローマ数字の大文字。 */
|
|
19
|
+
const CAPITAL_OF: Readonly<Record<"letter" | "roman", Style>> = { letter: "capital", roman: "capital-roman" };
|
|
20
|
+
|
|
21
|
+
/** 小文字にした番号を形で読む。一文字の "i" は、開いたときに付けた並びの位置で見分ける。ローマ数字なら 1、英字なら 9。 */
|
|
22
|
+
const shapeOf = (lower: string, ordinal: number | undefined): "letter" | "roman" => {
|
|
23
|
+
if (MULTI_ROMAN.test(lower)) return "roman";
|
|
24
|
+
return AMBIGUOUS.test(lower) && ordinal === parseRoman(lower) ? "roman" : "letter";
|
|
25
|
+
};
|
|
26
|
+
|
|
27
|
+
/** 開いている項目の書き方。"(ii)" はローマ数字、"(b)" は英字、"(B)" は大文字の英字。 */
|
|
28
|
+
export const styleOfOpen = (open: NumberedLine): Style | undefined => {
|
|
29
|
+
const inner = /^\((?<n>[A-Za-z0-9]{1,5})\)$/u.exec(open.label)?.groups?.["n"];
|
|
30
|
+
if (inner === undefined) return undefined;
|
|
31
|
+
if (DIGITS.test(inner) || IS_INSERTED.test(inner)) return "digit";
|
|
32
|
+
const shape = shapeOf(inner.toLowerCase(), open.ordinal);
|
|
33
|
+
return CAPITALS.test(inner) ? CAPITAL_OF[shape] : shape;
|
|
34
|
+
};
|
|
35
|
+
|
|
36
|
+
const LETTER_BEFORE_A = "a".charCodeAt(0) - 1;
|
|
37
|
+
|
|
38
|
+
/** "(b)" と "(B)" は 2 番目、"(ii)" と "(II)" も 2 番目。二文字以上の英字("(aa)")は並びが決まらないので付けない。 */
|
|
39
|
+
export const ordinalOf = (raw: string, style: Style): number | undefined => {
|
|
40
|
+
if (style === "digit") return IS_INSERTED.test(raw) ? undefined : Number(raw);
|
|
41
|
+
if (style === "roman" || style === "capital-roman") return parseRoman(raw);
|
|
42
|
+
return raw.length === 1 ? raw.toLowerCase().charCodeAt(0) - LETTER_BEFORE_A : undefined;
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
const styles = (context: NumberingContext): (Style | undefined)[] => context.open.map(styleOfOpen);
|
|
46
|
+
|
|
47
|
+
/** "(h)" の次の "(i)"、"(H)" の次の "(I)" は英字。同じ書き方で開いている英字の次の文字なら、ローマ数字とは読まない。 */
|
|
48
|
+
const follows = (raw: string, style: "letter" | "capital", context: NumberingContext): boolean =>
|
|
49
|
+
context.open.some((open, index) => styles(context)[index] === style && open.number.charCodeAt(0) + 1 === raw.charCodeAt(0));
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* 英字の並びは (a) から始まるので、"(a)" や "(1)" のすぐ下に来た "(i)" は一段深いローマ数字。
|
|
53
|
+
* 米国の規則は (a)(1)(i) の順に下る。見出しのすぐ下の "(i)" は、どちらとも決まらないので英字。
|
|
54
|
+
*/
|
|
55
|
+
const OPENS_ROMAN: ReadonlySet<Style | undefined> = new Set(["letter", "digit"]);
|
|
56
|
+
|
|
57
|
+
/** "(I)" は米国の法典で (i) の一段下の大文字ローマ数字。"(H)" の次なら英字。 */
|
|
58
|
+
const capitalRoman = (raw: string, open: readonly (Style | undefined)[], context: NumberingContext): boolean => {
|
|
59
|
+
const lower = raw.toLowerCase();
|
|
60
|
+
if (MULTI_ROMAN.test(lower)) return open.includes("capital-roman");
|
|
61
|
+
if (!AMBIGUOUS.test(lower) || follows(raw, "capital", context)) return false;
|
|
62
|
+
return open.includes("capital-roman") || (raw === "I" && open.at(-1) === "roman");
|
|
63
|
+
};
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* 米国の規則は (a)(1)(i)(A) と下る。大文字の "(A)" は、開いているローマ数字のすぐ下で始まるか、開いている大文字の続きのときだけ項目。
|
|
67
|
+
* 英国の法令や契約書が一段目に使う "(A)" は、何の下とも決まらないので本文のまま読む。
|
|
68
|
+
*/
|
|
69
|
+
const capitalStyleOf = (raw: string, context: NumberingContext): Style | undefined => {
|
|
70
|
+
const open = styles(context);
|
|
71
|
+
if (capitalRoman(raw, open, context)) return "capital-roman";
|
|
72
|
+
if (!ONE_CAPITAL.test(raw)) return undefined;
|
|
73
|
+
return open.includes("capital") || (raw === "A" && open.at(-1) === "roman") ? "capital" : undefined;
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* "(i)" is a roman numeral right under "(a)" or "(1)", or when a roman list is already open; the letter i otherwise.
|
|
78
|
+
* A capital "(A)" is an item only where a US regulation puts it, under a roman item; undefined elsewhere.
|
|
79
|
+
*/
|
|
80
|
+
export const styleOf = (raw: string, context: NumberingContext): Style | undefined => {
|
|
81
|
+
if (DIGITS.test(raw) || IS_INSERTED.test(raw)) return "digit";
|
|
82
|
+
if (CAPITALS.test(raw)) return capitalStyleOf(raw, context);
|
|
83
|
+
if (MULTI_ROMAN.test(raw)) return "roman";
|
|
84
|
+
if (!AMBIGUOUS.test(raw) || follows(raw, "letter", context)) return "letter";
|
|
85
|
+
const open = styles(context);
|
|
86
|
+
return open.includes("roman") || OPENS_ROMAN.has(open.at(-1)) ? "roman" : "letter";
|
|
87
|
+
};
|
|
88
|
+
|
|
89
|
+
/** The nearest open item written the same way: "(b)" closes "(i)" and sits beside "(a)". */
|
|
90
|
+
const siblingOf = (style: Style, context: NumberingContext): NumberedLine | undefined => {
|
|
91
|
+
const open = styles(context);
|
|
92
|
+
return context.open.findLast((_item, index) => open[index] === style);
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
const IS_CAPITAL: ReadonlySet<Style | undefined> = new Set(["capital", "capital-roman"]);
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* "(A)" の下の "(1)" や "(i)"、"(i)" の下の "(A)" は、上に開いている同じ書き方の項目の続きではなく、一段深い並びの始まり。
|
|
99
|
+
* 米国の規則は (a)(1)(i)(A) の下を、もう一度 (1) や (i) で数える。
|
|
100
|
+
*/
|
|
101
|
+
const startsBelow = (style: Style, ordinal: number | undefined, innermost: Style | undefined): boolean =>
|
|
102
|
+
ordinal === 1 && style !== innermost && (IS_CAPITAL.has(innermost) || (IS_CAPITAL.has(style) && innermost === "roman"));
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* A sibling has the depth of the open item written the same way. A new way of numbering goes one deeper than whatever
|
|
106
|
+
* is open, and so does a first item right under a capital, or a first capital right under a roman item.
|
|
107
|
+
*/
|
|
108
|
+
export const depthFor = (style: Style, context: NumberingContext, ordinal?: number): number => {
|
|
109
|
+
const innermost = context.open.at(-1);
|
|
110
|
+
const deeper = (innermost?.depth ?? 0) + 1;
|
|
111
|
+
if (innermost !== undefined && startsBelow(style, ordinal, styleOfOpen(innermost))) return deeper;
|
|
112
|
+
return siblingOf(style, context)?.depth ?? deeper;
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* 参照の "(A)" や "(I)" は、ローマ数字の "(iii)" の次に書かれたときだけ番地の続き。木も、大文字はローマ数字の下でしか読まない。
|
|
117
|
+
* "section 5(A)" の "(A)" は番地に入れず、これまでどおり "section 5" を引く。"(h)(i)" や番号のすぐ後の "(i)" は、木と同じく英字と読む。
|
|
118
|
+
*/
|
|
119
|
+
export const continuesAddress = (part: string, before: readonly string[]): boolean => {
|
|
120
|
+
if (!CAPITALS.test(part)) return true;
|
|
121
|
+
const [previous, earlier] = [before.at(-1) ?? "", before.at(-2) ?? ""];
|
|
122
|
+
if (MULTI_ROMAN.test(previous)) return true;
|
|
123
|
+
return AMBIGUOUS.test(previous) && earlier !== "" && earlier.charCodeAt(0) + 1 !== previous.charCodeAt(0);
|
|
124
|
+
};
|
|
125
|
+
|
|
126
|
+
/** 番号だけの行 "(5)" は、開いている "(4)" の次のときだけ項目。本文は次の行から始まる。 */
|
|
127
|
+
export const continuesOpen = (style: Style, ordinal: number | undefined, context: NumberingContext): boolean => {
|
|
128
|
+
const previous = siblingOf(style, context)?.ordinal;
|
|
129
|
+
return ordinal !== undefined && previous !== undefined && ordinal === previous + 1;
|
|
130
|
+
};
|
package/src/long-runs.ts
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 空白の無いまま limit 文字を超える並びを、同じ長さの空白で覆う。語ではなく(base64、ハッシュ、区切りの無い記号の列)、
|
|
3
|
+
* 解析器(wink)はその長さの二乗で遅くなる。長さを変えないので、残りの語の位置はそのまま。
|
|
4
|
+
*/
|
|
5
|
+
export const blankLongRuns = (text: string, limit: number): string =>
|
|
6
|
+
text.replace(new RegExp(`\\S{${String(limit + 1)},}`, "gu"), (run) => " ".repeat(run.length));
|
package/src/pos.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import { createRequire } from "node:module";
|
|
2
2
|
import type { Token } from "chaffjs/plugin";
|
|
3
3
|
import { loadLexicons } from "./lexicons.ts";
|
|
4
|
-
import {
|
|
4
|
+
import { blankLongRuns } from "./long-runs.ts";
|
|
5
|
+
import { lowercasedAt, properNounChecked, rereadAt, sentenceInitialCommonWord } from "./proper-noun.ts";
|
|
5
6
|
import { isStativeParticiple, stativeVocabulary } from "./stative-participle.ts";
|
|
6
7
|
|
|
7
8
|
const require = createRequire(import.meta.url);
|
|
@@ -157,18 +158,39 @@ const determinerFeatures = (entry: Tagged): Features => {
|
|
|
157
158
|
return entry.pos === "DT" && ARTICLES.has(entry.value.toLowerCase()) ? { features: { PronType: "Art" } } : {};
|
|
158
159
|
};
|
|
159
160
|
|
|
161
|
+
/** 複数形の名詞は UD の Number=Plur。数の語がそれを数えていれば(five minutes)、one of のような言い回しではなく量。 */
|
|
162
|
+
const PLURAL_TAG = new Set(["NNS", "NNPS"]);
|
|
163
|
+
|
|
164
|
+
const nounOrDeterminerFeatures = (entry: Tagged): Features => (PLURAL_TAG.has(entry.pos) ? { features: { Number: "Plur" } } : determinerFeatures(entry));
|
|
165
|
+
|
|
160
166
|
/** 過去分詞は VerbForm=Part。Based on the review, のような分詞の導入句を、命令形の並び(fix the parser, ship it)と見分ける。 */
|
|
161
167
|
const featuresOf = (tagged: readonly Tagged[], at: number): Features => {
|
|
162
168
|
const entry = tagged[at];
|
|
163
169
|
if (entry === undefined) return {};
|
|
164
|
-
if (entry.pos !== "VBN") return
|
|
170
|
+
if (entry.pos !== "VBN") return nounOrDeterminerFeatures(entry);
|
|
165
171
|
return { features: isPassive(tagged, at) ? { VerbForm: "Part", Voice: "Pass" } : { VerbForm: "Part" } };
|
|
166
172
|
};
|
|
167
173
|
|
|
168
|
-
|
|
174
|
+
/** 解析器が引く語彙(語 → Penn Treebank の品詞の並び)。wink-pos-tagger が自分の依存から読むものと同じ一つを、同じ場所から読む。 */
|
|
175
|
+
type Vocabulary = (word: string) => readonly string[] | undefined;
|
|
176
|
+
|
|
177
|
+
const isTags = (value: unknown): value is readonly string[] => Array.isArray(value) && value.every((tag) => typeof tag === "string");
|
|
178
|
+
|
|
179
|
+
const buildVocabulary = (): Vocabulary => {
|
|
180
|
+
const words: unknown = createRequire(require.resolve("wink-pos-tagger"))("wink-lexicon/src/lexicon.js");
|
|
181
|
+
if (!isRecord(words)) throw new Error("wink-lexicon の語彙が object ではありません");
|
|
182
|
+
return (word) => {
|
|
183
|
+
const tags = Object.hasOwn(words, word) ? words[word] : undefined;
|
|
184
|
+
return isTags(tags) ? tags : undefined;
|
|
185
|
+
};
|
|
186
|
+
};
|
|
187
|
+
|
|
188
|
+
const state: { ready: Tagger | undefined; vocabulary: Vocabulary } = { ready: undefined, vocabulary: () => undefined };
|
|
169
189
|
|
|
170
190
|
export const prepare = (): void => {
|
|
171
|
-
state.ready
|
|
191
|
+
if (state.ready !== undefined) return;
|
|
192
|
+
state.ready = build();
|
|
193
|
+
state.vocabulary = buildVocabulary();
|
|
172
194
|
};
|
|
173
195
|
|
|
174
196
|
export const isReady = (): boolean => state.ready !== undefined;
|
|
@@ -195,8 +217,21 @@ const locate = (text: string, tagged: readonly Tagged[]): Token[] =>
|
|
|
195
217
|
{ tokens: [], cursor: 0 },
|
|
196
218
|
).tokens;
|
|
197
219
|
|
|
220
|
+
/** 英語の語はこれより長くならない。超える並びは語として読まない。 */
|
|
221
|
+
const RUN_LIMIT = 1000;
|
|
222
|
+
|
|
223
|
+
const tagged = (tagger: Tagger, text: string): readonly Tagged[] => toArray(callMethod(tagger, "tagSentence", [text])).filter(isTagged);
|
|
224
|
+
|
|
225
|
+
/** 文頭で大文字になっただけの普通の語を、小文字で書いたときの品詞に戻す。前後の語による判断も効くよう、文ごと解析し直す。 */
|
|
226
|
+
const withSentenceInitialCase = (tagger: Tagger, text: string, entries: readonly Tagged[]): readonly Tagged[] => {
|
|
227
|
+
const at = sentenceInitialCommonWord(entries, state.vocabulary);
|
|
228
|
+
const lowered = lowercasedAt(text, entries[at]?.value);
|
|
229
|
+
return lowered === undefined ? entries : rereadAt(entries, at, tagged(tagger, lowered));
|
|
230
|
+
};
|
|
231
|
+
|
|
198
232
|
export const tokenize = (text: string): Token[] | undefined => {
|
|
199
233
|
const tagger = state.ready;
|
|
200
234
|
if (tagger === undefined) return undefined;
|
|
201
|
-
|
|
235
|
+
const words = blankLongRuns(text, RUN_LIMIT);
|
|
236
|
+
return locate(words, withSentenceInitialCase(tagger, words, tagged(tagger, words)));
|
|
202
237
|
};
|
package/src/proper-noun.ts
CHANGED
|
@@ -19,3 +19,46 @@ export const properNounChecked = (surface: string, pos: string): string => {
|
|
|
19
19
|
if (!LETTER.test(surface)) return nonWord(surface);
|
|
20
20
|
return SMALL.test(surface) && !CAPITAL.test(surface) ? "NOUN" : "PROPN";
|
|
21
21
|
};
|
|
22
|
+
|
|
23
|
+
/** 解析器が返す 1 語。pos は Penn Treebank。 */
|
|
24
|
+
type TaggedWord = { readonly value: string; readonly pos: string; readonly lemma?: string };
|
|
25
|
+
|
|
26
|
+
const PROPER_TAG = new Set(["NNP", "NNPS"]);
|
|
27
|
+
|
|
28
|
+
/** 頭の一字だけが大文字の語(Containers、Such)。文頭の大文字はこの形になる。API や iPhone は名前の書き方なので含めない。 */
|
|
29
|
+
const CAPITALISED = /^\p{Lu}\p{Ll}+$/u;
|
|
30
|
+
|
|
31
|
+
/** 解析器の語彙に、その語が普通の語として載っているか。一度でも固有名詞として載っていれば(may の May)、名前かもしれない。 */
|
|
32
|
+
const isCommonWord = (tags: readonly string[] | undefined): boolean => tags !== undefined && tags.length > 0 && !tags.some((tag) => PROPER_TAG.has(tag));
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* 文頭で大文字になっただけの普通の語の位置。無ければ -1。
|
|
36
|
+
* wink は大文字で始まる名詞と形容詞をすべて固有名詞にするので、Containers are … の Containers も Traditional servers … の Traditional も固有名詞になる。
|
|
37
|
+
* 文頭の大文字は英語の書き方で、名前の印ではない。解析器の語彙が小文字の形を普通の語として知っていれば、名前ではなくその語として読む。
|
|
38
|
+
* 語彙に無い語(Kubernetes、Congress)と、文の途中の大文字の語は名前のまま。
|
|
39
|
+
*/
|
|
40
|
+
export const sentenceInitialCommonWord = (tagged: readonly TaggedWord[], tagsOf: (word: string) => readonly string[] | undefined): number => {
|
|
41
|
+
const at = tagged.findIndex((entry) => LETTER.test(entry.value));
|
|
42
|
+
const first = tagged[at];
|
|
43
|
+
if (first === undefined || !PROPER_TAG.has(first.pos) || !CAPITALISED.test(first.value)) return -1;
|
|
44
|
+
return isCommonWord(tagsOf(first.value.toLowerCase())) ? at : -1;
|
|
45
|
+
};
|
|
46
|
+
|
|
47
|
+
/** 文の中の最初の word を小文字にした文。word が文に無ければ undefined。 */
|
|
48
|
+
export const lowercasedAt = (text: string, word: string | undefined): string | undefined => {
|
|
49
|
+
const start = word === undefined ? -1 : text.indexOf(word);
|
|
50
|
+
if (word === undefined || start === -1) return undefined;
|
|
51
|
+
return `${text.slice(0, start)}${word.toLowerCase()}${text.slice(start + word.length)}`;
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* at の語の品詞と原形を、小文字にして解析し直した結果(again)から取る。表層は元のまま。
|
|
56
|
+
* 解析し直した文の同じ位置に、同じ語の小文字が無ければ、語の切り方が変わったので元のまま。
|
|
57
|
+
*/
|
|
58
|
+
export const rereadAt = (entries: readonly TaggedWord[], at: number, again: readonly TaggedWord[]): readonly TaggedWord[] => {
|
|
59
|
+
const entry = entries[at];
|
|
60
|
+
const reread = again[at];
|
|
61
|
+
if (entry === undefined || reread === undefined || reread.value !== entry.value.toLowerCase()) return entries;
|
|
62
|
+
const word: TaggedWord = { value: entry.value, pos: reread.pos, ...(reread.lemma === undefined ? {} : { lemma: reread.lemma }) };
|
|
63
|
+
return entries.map((original, index) => (index === at ? word : original));
|
|
64
|
+
};
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import type { Span } from "chaffjs/plugin";
|
|
2
|
+
import { isAbbreviation } from "./sentence-split.ts";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* 米国式の引用は、文末のピリオドを閉じ引用符の内側に置く(…called it "a fair trial." Plainly, …)。
|
|
6
|
+
* sentence-splitter は引用符の中の句点で文を切らず、引用符が閉じた後の空白でも切らないので、次の文が前の文につながる。
|
|
7
|
+
*
|
|
8
|
+
* 分割器が返した一つの文を、文末の記号 + 閉じ引用符 + 空白 + 次の文の頭(大文字。開き括弧・引用符の後の大文字も)で切る。
|
|
9
|
+
* 引用符の中の語が略語・頭文字("U.S." "Dr." "J.")のときと、括弧が開いたままのときは切らない。
|
|
10
|
+
*/
|
|
11
|
+
/** 次の文の頭。空白の後の大文字。開き括弧・引用符の後の大文字も文頭である。 */
|
|
12
|
+
export const NEXT_SENTENCE_HEAD = String.raw`\s+[\p{Ps}\p{Pi}"']*\p{Lu}`;
|
|
13
|
+
const QUOTED_STOP = new RegExp(String.raw`[.?!]["'”’]+(?=${NEXT_SENTENCE_HEAD})`, "gu");
|
|
14
|
+
// ピリオドの前の語は、空白・開き引用符・開き括弧の後から数える。
|
|
15
|
+
const WORD_START = /[\s\p{Ps}\p{Pi}"']/u;
|
|
16
|
+
// 一字ずつピリオドを打った語(J. / U.S. / e.g.)は、略語の一覧に無くても略語。
|
|
17
|
+
const INITIALS = /^(?:\p{L}\.)+$/u;
|
|
18
|
+
// 分割器が一組として読む括弧(( [ { ( [ { 【 《 「 『 …)は、Unicode の開き・閉じ括弧の分類にすべて入る。
|
|
19
|
+
const OPENER = /\p{Ps}/u;
|
|
20
|
+
const CLOSER = /\p{Pe}/u;
|
|
21
|
+
|
|
22
|
+
const endsWithAbbreviation = (sentence: string, stop: number): boolean => {
|
|
23
|
+
if (sentence[stop] !== ".") return false;
|
|
24
|
+
const word = `${sentence.slice(0, stop).split(WORD_START).at(-1) ?? ""}.`;
|
|
25
|
+
return isAbbreviation(word) || INITIALS.test(word);
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
// 対の無い閉じ括弧(「1) 最初の項目」の番号)は、後から開く括弧を打ち消さない。
|
|
29
|
+
const depthAfter = (depth: number, char: string): number => {
|
|
30
|
+
if (OPENER.test(char)) return depth + 1;
|
|
31
|
+
return CLOSER.test(char) ? Math.max(0, depth - 1) : depth;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
const bracketOpenAt = (sentence: string, index: number): boolean => Array.from(sentence.slice(0, index)).reduce(depthAfter, 0) > 0;
|
|
35
|
+
|
|
36
|
+
type Cut = { readonly end: number; readonly next: number };
|
|
37
|
+
|
|
38
|
+
const cutsIn = (sentence: string): Cut[] =>
|
|
39
|
+
[...sentence.matchAll(QUOTED_STOP)]
|
|
40
|
+
.filter((match) => !endsWithAbbreviation(sentence, match.index) && !bracketOpenAt(sentence, match.index))
|
|
41
|
+
.map((match) => {
|
|
42
|
+
const end = match.index + match[0].length;
|
|
43
|
+
return { end, next: end + (/^\s+/u.exec(sentence.slice(end))?.[0].length ?? 0) };
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
/** span の文を、閉じ引用符の内側で閉じた文ごとに分けた span。切れ目が無ければ span そのもの一つ。 */
|
|
47
|
+
export const splitAtQuotedStops = (text: string, span: Span): Span[] => {
|
|
48
|
+
const cuts = cutsIn(text.slice(span.start, span.end));
|
|
49
|
+
const starts = [0, ...cuts.map((cut) => cut.next)];
|
|
50
|
+
const ends = [...cuts.map((cut) => cut.end), span.end - span.start];
|
|
51
|
+
return starts.map((start, index) => ({ start: span.start + start, end: span.start + (ends[index] ?? start) }));
|
|
52
|
+
};
|
package/src/sentence-split.ts
CHANGED
|
@@ -45,6 +45,8 @@ const { language } = DefaultAbbrMarkerOptions;
|
|
|
45
45
|
const ABBREVIATIONS = new Set(
|
|
46
46
|
[...language.ABBREVIATIONS, ...language.PREPOSITIVE_ABBREVIATIONS, ...language.EXCLAMATION_WORDS].map((word) => word.toLowerCase()),
|
|
47
47
|
);
|
|
48
|
+
/** 分割器が略語と読む語(ピリオドまで含めて渡す)。「Dr.」「U.S.」「Jan.」。 */
|
|
49
|
+
export const isAbbreviation = (word: string): boolean => ABBREVIATIONS.has(word.toLowerCase());
|
|
48
50
|
// 「J. Smith」のような語は前の語を見て略語か決まる。前の語が切れ目の向こうにあると判定が変わる。
|
|
49
51
|
const CAPITAL_DOT = /\p{Lu}\./gu;
|
|
50
52
|
|
|
@@ -76,7 +78,7 @@ const indexesOf = (text: string, pattern: RegExp): number[] => [...text.matchAll
|
|
|
76
78
|
/** 句点の並びの直後。和文は直前が仮名・漢字の句点、英文は略語でない語のピリオド。 */
|
|
77
79
|
const stopEnds = (text: string): number[] => {
|
|
78
80
|
const japanese = [...text.matchAll(JAPANESE_STOPS)].filter((match) => isCJK(text[match.index - 1]));
|
|
79
|
-
const english = [...text.matchAll(ENGLISH_STOP)].filter((match) => !
|
|
81
|
+
const english = [...text.matchAll(ENGLISH_STOP)].filter((match) => !isAbbreviation(match[0]));
|
|
80
82
|
return [...japanese, ...english].map((match) => match.index + match[0].length).sort((a, b) => a - b);
|
|
81
83
|
};
|
|
82
84
|
|
package/src/structure.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
|
|
2
|
-
import { citedDocumentAfter, citedDocumentBefore } from "./citation.ts";
|
|
2
|
+
import { citedDocumentAfter, citedDocumentBefore, hyphenatedTagAround } from "./citation.ts";
|
|
3
3
|
import { membersAfter } from "./reference-list.ts";
|
|
4
4
|
import { parseRoman } from "./roman.ts";
|
|
5
5
|
import { dates } from "./dates.ts";
|
|
6
6
|
import { definitionScopeDepth, definitions, opensDefinitionScope } from "./definitions.ts";
|
|
7
7
|
import { CHAPTER_DEPTH, PART_DEPTH } from "./depth.ts";
|
|
8
|
+
import { continuesAddress, continuesOpen, depthFor, INSERTED, ordinalOf, styleOf } from "./item-style.ts";
|
|
8
9
|
|
|
9
10
|
// Contracts, specifications and statutes in English. core nests what this reads; it does not know
|
|
10
11
|
// how English numbers its articles.
|
|
@@ -28,14 +29,8 @@ const titleOf = (rest: string): string | undefined => {
|
|
|
28
29
|
|
|
29
30
|
const ARTICLE = /^\s{0,3}(?:ARTICLE|Article)\s+(?<n>\d{1,3}|[IVXLC]{1,7})\b(?<rest>.*)$/u;
|
|
30
31
|
const SECTION = /^\s{0,3}(?:SECTION|Section|§)\s*(?<n>\d{1,3}(?:\.\d{1,3}){0,5})\b(?<rest>.*)$/u;
|
|
31
|
-
/**
|
|
32
|
-
|
|
33
|
-
* outside the sequence: it has no ordinal, so "(1)" after "(A1)" is still the first.
|
|
34
|
-
*/
|
|
35
|
-
const INSERTED = "\\d{1,3}[A-Z]{1,2}|[A-Z]{1,2}\\d{1,3}";
|
|
36
|
-
const LETTERED = new RegExp(`^\\s{0,6}\\((?<n>[a-z]{1,4}|\\d{1,3}|${INSERTED})\\)\\s+(?<rest>\\S.*)$`, "u");
|
|
37
|
-
const IS_INSERTED = new RegExp(`^(?:${INSERTED})$`, "u");
|
|
38
|
-
const MULTI_ROMAN = /^(?:ii|iii|iv|vi|vii|viii|ix)$/u;
|
|
32
|
+
/** "(a) text", "(ii) text", "(B) text", or the label alone on its line with the text below it. */
|
|
33
|
+
const LETTERED = new RegExp(`^\\s{0,6}\\((?<n>[a-z]{1,4}|[A-Z]{1,4}|\\d{1,3}|${INSERTED})\\)(?:\\s+(?<rest>\\S.*))?\\s*$`, "u");
|
|
39
34
|
|
|
40
35
|
const headed = (pattern: RegExp, line: string, numbering: string, label: (n: string) => string): NumberedLine | undefined => {
|
|
41
36
|
const groups = pattern.exec(line)?.groups;
|
|
@@ -56,70 +51,24 @@ const headed = (pattern: RegExp, line: string, numbering: string, label: (n: str
|
|
|
56
51
|
};
|
|
57
52
|
};
|
|
58
53
|
|
|
59
|
-
type Style = "letter" | "roman" | "digit";
|
|
60
|
-
|
|
61
|
-
/**
|
|
62
|
-
* 開いている項目の書き方。"(ii)" はローマ数字、"(b)" は英字。一文字の "(i)" はどちらにも読めるので、
|
|
63
|
-
* 開いたときの読みを覚えておく代わりに、一つ上に英字が開いていたかで決め直す。
|
|
64
|
-
*/
|
|
65
|
-
const styleOfLabel = (open: NumberedLine, index: number, all: readonly NumberedLine[]): Style | undefined => {
|
|
66
|
-
const inner = /^\((?<n>[A-Za-z0-9]{1,5})\)$/u.exec(open.label)?.groups?.["n"];
|
|
67
|
-
if (inner === undefined) return undefined;
|
|
68
|
-
if (/^\d+$/u.test(inner) || IS_INSERTED.test(inner)) return "digit";
|
|
69
|
-
if (MULTI_ROMAN.test(inner)) return "roman";
|
|
70
|
-
const above = all[index - 1];
|
|
71
|
-
return /^[ivx]$/u.test(inner) && above !== undefined && styleOfLabel(above, index - 1, all) === "letter" && above.depth < open.depth ? "roman" : "letter";
|
|
72
|
-
};
|
|
73
|
-
|
|
74
|
-
const styles = (context: NumberingContext): (Style | undefined)[] => context.open.map((open, index, all) => styleOfLabel(open, index, all));
|
|
75
|
-
|
|
76
|
-
/** "(h)" の次の "(i)" は英字。開いている英字の次の文字なら、ローマ数字とは読まない。 */
|
|
77
|
-
const followsLetter = (raw: string, context: NumberingContext): boolean =>
|
|
78
|
-
context.open.some((open, index) => styles(context)[index] === "letter" && open.number.charCodeAt(0) + 1 === raw.charCodeAt(0));
|
|
79
|
-
|
|
80
|
-
/** "(i)" is a roman numeral right under "(a)", or when a roman list is already open; the letter i otherwise. */
|
|
81
|
-
const styleOf = (raw: string, context: NumberingContext): Style => {
|
|
82
|
-
if (/^\d+$/u.test(raw) || IS_INSERTED.test(raw)) return "digit";
|
|
83
|
-
if (MULTI_ROMAN.test(raw)) return "roman";
|
|
84
|
-
if (!/^[ivx]$/u.test(raw) || followsLetter(raw, context)) return "letter";
|
|
85
|
-
const open = styles(context);
|
|
86
|
-
return open.includes("roman") || open.at(-1) === "letter" ? "roman" : "letter";
|
|
87
|
-
};
|
|
88
|
-
|
|
89
|
-
/**
|
|
90
|
-
* A sibling has the depth of the open item written the same way: "(b)" closes "(i)" and sits beside "(a)".
|
|
91
|
-
* A new way of numbering goes one deeper than whatever is open.
|
|
92
|
-
*/
|
|
93
|
-
const depthFor = (style: Style, context: NumberingContext): number => {
|
|
94
|
-
const open = styles(context);
|
|
95
|
-
const sibling = [...context.open].reverse().find((_item, reversed) => open[context.open.length - 1 - reversed] === style);
|
|
96
|
-
return sibling?.depth ?? (context.open.at(-1)?.depth ?? 0) + 1;
|
|
97
|
-
};
|
|
98
|
-
|
|
99
|
-
const LETTER_BEFORE_A = "a".charCodeAt(0) - 1;
|
|
100
|
-
|
|
101
|
-
/** "(b)" は 2 番目、"(ii)" も 2 番目。二文字以上の英字("(aa)")は並びが決まらないので付けない。 */
|
|
102
|
-
const ordinalOf = (raw: string, style: Style): number | undefined => {
|
|
103
|
-
if (style === "digit") return IS_INSERTED.test(raw) ? undefined : Number(raw);
|
|
104
|
-
if (style === "roman") return parseRoman(raw);
|
|
105
|
-
return raw.length === 1 ? raw.charCodeAt(0) - LETTER_BEFORE_A : undefined;
|
|
106
|
-
};
|
|
107
|
-
|
|
108
54
|
const lettered = (line: string, context: NumberingContext): NumberedLine | undefined => {
|
|
109
55
|
const groups = LETTERED.exec(line)?.groups;
|
|
110
56
|
const raw = groups?.["n"];
|
|
111
57
|
if (groups === undefined || raw === undefined) return undefined;
|
|
112
58
|
const style = styleOf(raw, context);
|
|
59
|
+
const ordinal = style === undefined ? undefined : ordinalOf(raw, style);
|
|
60
|
+
const rest = groups["rest"];
|
|
61
|
+
if (style === undefined || (rest === undefined && !continuesOpen(style, ordinal, context))) return undefined;
|
|
113
62
|
// 番地は書かれたままの "ii" を使う。参照「Section 4.2(a)(ii)」も同じ形で書かれるので、そのまま引ける。
|
|
114
63
|
return {
|
|
115
64
|
kind: "item",
|
|
116
|
-
depth: depthFor(style, context),
|
|
117
|
-
ordinal
|
|
65
|
+
depth: depthFor(style, context, ordinal),
|
|
66
|
+
ordinal,
|
|
118
67
|
number: raw,
|
|
119
68
|
absolute: false,
|
|
120
69
|
label: `(${raw})`,
|
|
121
70
|
heading: "",
|
|
122
|
-
rest:
|
|
71
|
+
rest: rest?.trim() ?? "",
|
|
123
72
|
};
|
|
124
73
|
};
|
|
125
74
|
|
|
@@ -144,13 +93,13 @@ const numbered = (line: string, context: NumberingContext): NumberedLine | undef
|
|
|
144
93
|
lettered(line, context);
|
|
145
94
|
|
|
146
95
|
const REFERENCE = /(?<word>\b[Ss]ections?|\b[Aa]rticles?|§) ?(?<n>\d{1,3}(?:\.\d{1,3}){0,5}|[IVXLC]{1,7})\b/gu;
|
|
147
|
-
/** "(a)", "(ii)", "(3)", and an inserted "(A1)" or "(2A)": the same labels the tree reads. */
|
|
148
|
-
const SUBDIVISION = new RegExp(`^\\((?<p>[a-z0-9]{1,4}|${INSERTED})\\)`, "u");
|
|
96
|
+
/** "(a)", "(ii)", "(3)", "(B)", and an inserted "(A1)" or "(2A)": the same labels the tree reads. */
|
|
97
|
+
const SUBDIVISION = new RegExp(`^\\((?<p>[a-z0-9]{1,4}|[A-Z]{1,4}|${INSERTED})\\)`, "u");
|
|
149
98
|
/** The longest label, with its parentheses: "(ZZ999)". */
|
|
150
99
|
const MAX_SUBDIVISION_LENGTH = 7;
|
|
151
100
|
|
|
152
|
-
/** "(a)(
|
|
153
|
-
const MAX_SUBDIVISIONS =
|
|
101
|
+
/** "(a)(1)(i)(A)(1)", a US regulation's deepest paragraph, is as deep as a reference goes; more parentheses are text. */
|
|
102
|
+
const MAX_SUBDIVISIONS = 5;
|
|
154
103
|
|
|
155
104
|
/**
|
|
156
105
|
* "(a)(ii)" のような続きの括弧を、正規表現を複雑にせずに一つずつ読む。
|
|
@@ -161,7 +110,7 @@ const subdivisions = (text: string, from: number): { readonly parts: readonly st
|
|
|
161
110
|
let end = from;
|
|
162
111
|
while (parts.length < MAX_SUBDIVISIONS) {
|
|
163
112
|
const part = SUBDIVISION.exec(text.slice(end, end + MAX_SUBDIVISION_LENGTH))?.groups?.["p"];
|
|
164
|
-
if (part === undefined) break;
|
|
113
|
+
if (part === undefined || !continuesAddress(part, parts)) break;
|
|
165
114
|
parts.push(part);
|
|
166
115
|
end += part.length + 2;
|
|
167
116
|
}
|
|
@@ -191,6 +140,13 @@ const glossedDocument = (gloss: Gloss): string | undefined => {
|
|
|
191
140
|
return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
|
|
192
141
|
};
|
|
193
142
|
|
|
143
|
+
/** The other document a reference names, or else a bracketed tag that the core checks against the document's list. */
|
|
144
|
+
const citation = (text: string, start: number, end: number, document: string | undefined): Readonly<Record<string, string>> => {
|
|
145
|
+
if (document !== undefined) return { document };
|
|
146
|
+
const citedTag = hyphenatedTagAround(text, start, end);
|
|
147
|
+
return citedTag === undefined ? {} : { citedTag };
|
|
148
|
+
};
|
|
149
|
+
|
|
194
150
|
/**
|
|
195
151
|
* "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
|
|
196
152
|
* "Section 9 of the Master Agreement" carries the other document's name, and is not looked up in this tree.
|
|
@@ -207,7 +163,7 @@ const references = (text: string): Mention[] => {
|
|
|
207
163
|
gloss.scanned = Math.max(gloss.scanned, end);
|
|
208
164
|
if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
|
|
209
165
|
const numbering = /^[Aa]/u.test(match.groups?.["word"] ?? "") ? "article" : "section";
|
|
210
|
-
const shared = { numbering, ...(
|
|
166
|
+
const shared = { numbering, ...citation(text, match.index, end, document) };
|
|
211
167
|
const first = { start: match.index, end, attrs: { target: [main, ...parts].join("."), label: text.slice(match.index, end), ...shared } };
|
|
212
168
|
const [plural, roman] = [/s$/u.test(match.groups?.["word"] ?? ""), /^[IVXLC]+$/u.test(match.groups?.["n"] ?? "")];
|
|
213
169
|
return [first, ...membersAfter(text, end, [main, ...parts], shared, plural, roman)];
|