@chaffjs/lang-en 0.15.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/citation.d.ts +5 -2
- package/dist/citation.d.ts.map +1 -1
- package/dist/citation.js +13 -9
- package/dist/citation.js.map +1 -1
- package/dist/code-citation.d.ts +15 -0
- package/dist/code-citation.d.ts.map +1 -0
- package/dist/code-citation.js +34 -0
- package/dist/code-citation.js.map +1 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -4
- package/dist/index.js.map +1 -1
- package/dist/japanese-run.d.ts +2 -0
- package/dist/japanese-run.d.ts.map +1 -0
- package/dist/japanese-run.js +9 -0
- package/dist/japanese-run.js.map +1 -0
- package/dist/label-stop.d.ts +13 -0
- package/dist/label-stop.d.ts.map +1 -0
- package/dist/label-stop.js +13 -0
- package/dist/label-stop.js.map +1 -0
- package/dist/pos.d.ts.map +1 -1
- package/dist/pos.js +37 -17
- package/dist/pos.js.map +1 -1
- package/dist/regexp.d.ts +3 -0
- package/dist/regexp.d.ts.map +1 -0
- package/dist/regexp.js +3 -0
- package/dist/regexp.js.map +1 -0
- package/dist/structure.d.ts.map +1 -1
- package/dist/structure.js +17 -6
- package/dist/structure.js.map +1 -1
- package/lexicons/abbreviated-label.yaml +16 -0
- package/lexicons/ai-tell.yaml +86 -0
- package/lexicons/announcing-opener.yaml +19 -0
- package/lexicons/assistant-residue.yaml +55 -0
- package/lexicons/contrast-frame.yaml +10 -0
- package/lexicons/contrast-lead.yaml +15 -0
- package/lexicons/contrast-turn.yaml +11 -0
- package/lexicons/count-anchor.yaml +7 -0
- package/lexicons/count-counter.yaml +40 -0
- package/lexicons/count-hedge.yaml +67 -0
- package/lexicons/count-number.yaml +16 -0
- package/lexicons/document-kind.yaml +19 -0
- package/lexicons/email-attachment-note.yaml +7 -0
- package/lexicons/email-attribution.yaml +7 -0
- package/lexicons/email-header-field.yaml +28 -0
- package/lexicons/email-written-field.yaml +8 -0
- package/lexicons/figure-elsewhere.yaml +8 -0
- package/lexicons/figure-label.yaml +14 -0
- package/lexicons/measure-unit.yaml +78 -0
- package/lexicons/numbered-label.yaml +26 -0
- package/lexicons/percent-unit.yaml +7 -0
- package/lexicons/placeholder-word.yaml +17 -0
- package/lexicons/range-connector.yaml +12 -0
- package/lexicons/share-exception.yaml +10 -0
- package/lexicons/share-label.yaml +12 -0
- package/lexicons/stock-transition.yaml +17 -0
- package/package.json +1 -1
- package/src/citation.ts +13 -9
- package/src/code-citation.ts +52 -0
- package/src/index.ts +14 -5
- package/src/japanese-run.ts +9 -0
- package/src/label-stop.ts +25 -0
- package/src/pos.ts +36 -20
- package/src/regexp.ts +2 -0
- package/src/structure.ts +22 -6
package/src/citation.ts
CHANGED
|
@@ -52,17 +52,18 @@ export const listMembers = (rest: string, plural: boolean): ListMember[] =>
|
|
|
52
52
|
*/
|
|
53
53
|
const TAG = "(?<tag>[A-Z][A-Z0-9]{1,30})";
|
|
54
54
|
/**
|
|
55
|
-
* "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]"
|
|
56
|
-
* the tag or not, so this tag is a candidate that the core checks
|
|
55
|
+
* "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]", and a numbered citation "[19]" like a blank to
|
|
56
|
+
* fill in. Only the document tells them apart, by listing the tag or not, so this tag is a candidate that the core checks
|
|
57
|
+
* against the document (attrs.citedTag).
|
|
57
58
|
*/
|
|
58
|
-
const
|
|
59
|
+
const LISTED_TAG = "(?<tag>\\d{1,3}|[A-Z][A-Z0-9]{0,30}(?:-[A-Z0-9]{1,30}){1,4})";
|
|
59
60
|
const tagAfter = (tag: string): RegExp => new RegExp(`^\\[${tag}\\]`, "u");
|
|
60
61
|
/** "[HTTP], Section 12.1": the tag written just before the reference. */
|
|
61
62
|
const tagBefore = (tag: string): RegExp => new RegExp(`\\[${tag}\\],?\\s?$`, "u");
|
|
62
63
|
const TAG_AFTER = tagAfter(TAG);
|
|
63
64
|
const TAG_BEFORE = tagBefore(TAG);
|
|
64
|
-
const
|
|
65
|
-
const
|
|
65
|
+
const LISTED_AFTER = tagAfter(LISTED_TAG);
|
|
66
|
+
const LISTED_BEFORE = tagBefore(LISTED_TAG);
|
|
66
67
|
const TAG_REACH = 40;
|
|
67
68
|
|
|
68
69
|
const tagEndingAt = (pattern: RegExp, text: string, start: number): string | undefined =>
|
|
@@ -120,9 +121,12 @@ export const citedDocumentAfter = (text: string, end: number): string | undefine
|
|
|
120
121
|
return words.length === 1 && SELF.has(name) ? undefined : name;
|
|
121
122
|
};
|
|
122
123
|
|
|
123
|
-
/**
|
|
124
|
-
|
|
124
|
+
/**
|
|
125
|
+
* A tag that names another document only if this document lists it, right after a reference ("Section 4.2.3 of
|
|
126
|
+
* [HTTP-CACHING]", "Section 4.2.2.17 of [19]") or just before it ("[HTTP-CACHING], Section 4", "[23], Section 2.17").
|
|
127
|
+
*/
|
|
128
|
+
export const listedTagAround = (text: string, start: number, end: number): string | undefined => {
|
|
125
129
|
const named = afterOf(text, end);
|
|
126
|
-
const following = named === undefined ? undefined :
|
|
127
|
-
return following ?? tagEndingAt(
|
|
130
|
+
const following = named === undefined ? undefined : LISTED_AFTER.exec(named)?.groups?.["tag"];
|
|
131
|
+
return following ?? tagEndingAt(LISTED_BEFORE, text, start);
|
|
128
132
|
};
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import type { Lexicon, LexiconEntry } from "chaffjs/plugin";
|
|
2
|
+
import { escapeRegExp } from "./regexp.ts";
|
|
3
|
+
|
|
4
|
+
// "35 CFR §122", "42 U.S.C. § 1983", "RFC 1122, Section 3.3.4.2": a section of a code or a numbered document named
|
|
5
|
+
// right before the reference, not of this document. Which names are codes is the lexicon's (document-kind).
|
|
6
|
+
|
|
7
|
+
/** Built once from the lexicon; each pattern is undefined when the lexicon names no code of its shape. */
|
|
8
|
+
export type CodeVocabulary = {
|
|
9
|
+
readonly before: RegExp | undefined;
|
|
10
|
+
readonly numberedBefore: RegExp | undefined;
|
|
11
|
+
/** The name of a code that takes a title number, at the start of the text and ending at a word boundary. */
|
|
12
|
+
readonly titled: RegExp | undefined;
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
/** The code's title number ("35 CFR"), then the name, then at most a second "§" ("§§ 1981 and 1983") up to the reference. */
|
|
16
|
+
const TITLE_NUMBER = String.raw`(?:\d{1,3}\s+)?`;
|
|
17
|
+
const SECOND_SIGN = String.raw`\s*(?:§\s*)?$`;
|
|
18
|
+
/** The document's own number after its name, then a comma, an opening parenthesis or a space up to the reference. */
|
|
19
|
+
const OWN_NUMBER = String.raw`\s+\d{1,5}`;
|
|
20
|
+
const NUMBER_TO_REFERENCE = String.raw`(?:,\s*|\s*\(\s*|\s+)(?:§\s*)?$`;
|
|
21
|
+
const NOT_INSIDE_A_WORD = String.raw`(?<![\p{L}\p{N}_.])`;
|
|
22
|
+
const WORD_ENDS = String.raw`(?![\p{L}\p{N}_])`;
|
|
23
|
+
|
|
24
|
+
/** "35 CFR " is a few characters. Only this much before a reference is read, however long the line. */
|
|
25
|
+
const REACH = 40;
|
|
26
|
+
|
|
27
|
+
const namesOf = (entries: readonly LexiconEntry[]): string | undefined => {
|
|
28
|
+
const names = entries.map((entry) => entry.pattern).toSorted((left, right) => right.length - left.length);
|
|
29
|
+
return names.length === 0 ? undefined : `(?:${names.map(escapeRegExp).join("|")})`;
|
|
30
|
+
};
|
|
31
|
+
|
|
32
|
+
/** An entry with position "before" is a name written before its own number ("RFC 1122"); the others follow a title number. */
|
|
33
|
+
export const codeVocabulary = (lexicons: Readonly<Record<string, Lexicon>>): CodeVocabulary => {
|
|
34
|
+
const entries = lexicons["document-kind"] ?? [];
|
|
35
|
+
const titled = namesOf(entries.filter((entry) => entry.position !== "before"));
|
|
36
|
+
const numbered = namesOf(entries.filter((entry) => entry.position === "before"));
|
|
37
|
+
return {
|
|
38
|
+
before: titled === undefined ? undefined : new RegExp(String.raw`${NOT_INSIDE_A_WORD}(?<code>${TITLE_NUMBER}${titled})${SECOND_SIGN}`, "u"),
|
|
39
|
+
numberedBefore:
|
|
40
|
+
numbered === undefined ? undefined : new RegExp(String.raw`${NOT_INSIDE_A_WORD}(?<code>${numbered}${OWN_NUMBER})${NUMBER_TO_REFERENCE}`, "u"),
|
|
41
|
+
titled: titled === undefined ? undefined : new RegExp(String.raw`^${titled}${WORD_ENDS}`, "u"),
|
|
42
|
+
};
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/** The code named right before the reference at `start`, or undefined when none is. */
|
|
46
|
+
export const citedCodeBefore = (text: string, start: number, vocabulary: CodeVocabulary): string | undefined => {
|
|
47
|
+
const before = text.slice(Math.max(0, start - REACH), start);
|
|
48
|
+
return vocabulary.before?.exec(before)?.groups?.["code"] ?? vocabulary.numberedBefore?.exec(before)?.groups?.["code"];
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
/** "40 CFR § 163.25" at the start of a heading: the text after the number starts with a code named after a title number. */
|
|
52
|
+
export const titledCodeAt = (rest: string, vocabulary: CodeVocabulary): boolean => vocabulary.titled?.test(rest) === true;
|
package/src/index.ts
CHANGED
|
@@ -1,14 +1,19 @@
|
|
|
1
1
|
import { loadLexicons } from "./lexicons.ts";
|
|
2
2
|
import { unmarkNumberStops } from "./number-stop.ts";
|
|
3
|
+
import { labelStops, unmarkLabelStops } from "./label-stop.ts";
|
|
3
4
|
import { sentenceSpans } from "./sentence-split.ts";
|
|
4
5
|
import { splitAtQuotedStops } from "./quoted-stop.ts";
|
|
5
6
|
import { reattachClosingQuotes } from "./closing-quote.ts";
|
|
6
7
|
import { structure } from "./structure.ts";
|
|
7
8
|
import { isReady, prepare, tokenize } from "./pos.ts";
|
|
8
|
-
import
|
|
9
|
+
import { isJapaneseRun } from "./japanese-run.ts";
|
|
10
|
+
import type { AdapterNeeds, EmbeddedLanguage, LanguageAdapter, Segmentation, Sentence, Span } from "chaffjs/plugin";
|
|
9
11
|
|
|
10
12
|
// chaff からは型だけを取る。実行時の値依存を作らない。アダプタは単体で動く。
|
|
11
13
|
|
|
14
|
+
const LEXICONS = loadLexicons();
|
|
15
|
+
const LABEL_STOPS = labelStops((LEXICONS["abbreviated-label"] ?? []).map((entry) => entry.pattern));
|
|
16
|
+
|
|
12
17
|
const LATIN_LETTER = /[a-z]/giu;
|
|
13
18
|
const COUNTABLE = /\S/gu;
|
|
14
19
|
|
|
@@ -25,9 +30,13 @@ const withTokens = (sentence: Sentence): Sentence => {
|
|
|
25
30
|
};
|
|
26
31
|
};
|
|
27
32
|
|
|
33
|
+
const JAPANESE: EmbeddedLanguage = { id: "ja", lengthUnit: "char" };
|
|
34
|
+
|
|
35
|
+
const withLanguage = (text: string, span: Span): Sentence => (isJapaneseRun(text) ? { span, text, embeddedLanguage: JAPANESE } : { span, text });
|
|
36
|
+
|
|
28
37
|
/**
|
|
29
38
|
* 英語は sentence-splitter の既定にほぼ任せる。"Dr." "e.g." "U.S." "$3.50" を
|
|
30
|
-
*
|
|
39
|
+
* いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませることと、番号の前の略した名前(FIG. 1、Vol. XLIII)の点で切らないこと、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
|
|
31
40
|
*/
|
|
32
41
|
export const adapter: LanguageAdapter = {
|
|
33
42
|
kind: "language",
|
|
@@ -51,11 +60,11 @@ export const adapter: LanguageAdapter = {
|
|
|
51
60
|
if (total === 0) return 0;
|
|
52
61
|
return [...source.matchAll(LATIN_LETTER)].length / total;
|
|
53
62
|
},
|
|
54
|
-
lexicons:
|
|
63
|
+
lexicons: LEXICONS,
|
|
55
64
|
structure,
|
|
56
65
|
segment: (text: string): Segmentation => {
|
|
57
|
-
const quotedStops = sentenceSpans(unmarkNumberStops(text)).flatMap((span) => splitAtQuotedStops(text, span));
|
|
58
|
-
const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => (
|
|
66
|
+
const quotedStops = sentenceSpans(unmarkNumberStops(unmarkLabelStops(text, LABEL_STOPS))).flatMap((span) => splitAtQuotedStops(text, span));
|
|
67
|
+
const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => withLanguage(text.slice(span.start, span.end), span));
|
|
59
68
|
return { sentences: isReady() ? sentences.map(withTokens) : sentences };
|
|
60
69
|
},
|
|
61
70
|
};
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A Japanese sentence in an English document (a quoted notice, a bilingual title): it has kana, and kana or kanji make
|
|
3
|
+
* up at least half of its letters (digits and symbols are not counted). Kanji alone do not count, as a Chinese name in English text is not Japanese.
|
|
4
|
+
*/
|
|
5
|
+
const KANA = /[\p{Script=Hiragana}\p{Script=Katakana}]/u;
|
|
6
|
+
const JAPANESE = /[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}]/gu;
|
|
7
|
+
const LETTER = /\p{L}/gu;
|
|
8
|
+
|
|
9
|
+
export const isJapaneseRun = (text: string): boolean => KANA.test(text) && [...text.matchAll(JAPANESE)].length * 2 >= [...text.matchAll(LETTER)].length;
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { escapeRegExp } from "./regexp.ts";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* sentence-splitter ends a sentence at the full stop of a label before its number: "FIG." and "1 illustrates …" were two
|
|
5
|
+
* sentences, and so were "Vol." and "XLIII (1979)". Where a number follows, that full stop is replaced with a letter
|
|
6
|
+
* before splitting. The length does not change, so the spans fit the original text.
|
|
7
|
+
* A lone "I" after a label is the pronoun ("He said No. I left."), and a word ("the last Fig. The tree") is not a number.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Built once from the lexicon; undefined when no label ends in a full stop. */
|
|
11
|
+
export type LabelStops = RegExp | undefined;
|
|
12
|
+
|
|
13
|
+
const NUMBER_AFTER = String.raw`(?=\s+(?:\d|(?:[IVXLCDM]{2,}|[VXLCDM])(?![\p{L}\p{N}_])))`;
|
|
14
|
+
const PLAIN_LETTER = "n";
|
|
15
|
+
|
|
16
|
+
/** The labels as written and in capitals, without their full stop ("Fig." → Fig, FIG). */
|
|
17
|
+
export const labelStops = (labels: readonly string[]): LabelStops => {
|
|
18
|
+
const stems = labels.filter((label) => label.endsWith(".")).flatMap((label) => [label.slice(0, -1), label.slice(0, -1).toUpperCase()]);
|
|
19
|
+
if (stems.length === 0) return undefined;
|
|
20
|
+
return new RegExp(String.raw`(?<![\p{L}\p{N}_.])(?:${[...new Set(stems)].map(escapeRegExp).join("|")})\.${NUMBER_AFTER}`, "gu");
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
/** The text with the full stop of every label before a number replaced by a letter. */
|
|
24
|
+
export const unmarkLabelStops = (text: string, stops: LabelStops): string =>
|
|
25
|
+
stops === undefined ? text : text.replace(stops, (label: string) => `${label.slice(0, -1)}${PLAIN_LETTER}`);
|
package/src/pos.ts
CHANGED
|
@@ -100,9 +100,22 @@ const BE = new Set(["be", "am", "is", "are", "was", "were", "been", "being"]);
|
|
|
100
100
|
|
|
101
101
|
const isBe = (entry: Tagged): boolean => BE.has(entry.lemma ?? entry.value.toLowerCase()) || BE.has(entry.value.toLowerCase());
|
|
102
102
|
|
|
103
|
+
/**
|
|
104
|
+
* end(0 以上)より前で test に合う最後の位置。無ければ -1。過去分詞の多い長い文で、分詞ごとに文の頭から写すと語数の二乗になるので、後ろから探す。
|
|
105
|
+
*/
|
|
106
|
+
const lastIndexBefore = (tagged: readonly Tagged[], end: number, test: (entry: Tagged) => boolean): number => {
|
|
107
|
+
let at = Math.min(end, tagged.length);
|
|
108
|
+
while (at > 0) {
|
|
109
|
+
at -= 1;
|
|
110
|
+
const entry = tagged[at];
|
|
111
|
+
if (entry !== undefined && test(entry)) return at;
|
|
112
|
+
}
|
|
113
|
+
return -1;
|
|
114
|
+
};
|
|
115
|
+
|
|
103
116
|
/** 過去分詞の前の be の位置。無ければ -1。 */
|
|
104
117
|
const beBefore = (tagged: readonly Tagged[], at: number): number => {
|
|
105
|
-
const head = tagged
|
|
118
|
+
const head = lastIndexBefore(tagged, at, (entry) => !SKIPPABLE.has(entry.pos));
|
|
106
119
|
const entry = tagged[head];
|
|
107
120
|
return entry !== undefined && isBe(entry) ? head : -1;
|
|
108
121
|
};
|
|
@@ -125,11 +138,11 @@ const NOMINAL_TAG = new Set(["NN", "NNS", "NNP", "NNPS", "PRP", "CD", "DT"]);
|
|
|
125
138
|
const RELATIVE_TAG = new Set(["WDT", "WP"]);
|
|
126
139
|
|
|
127
140
|
const inRelativeClause = (tagged: readonly Tagged[], be: number): boolean => {
|
|
128
|
-
const lead = tagged
|
|
141
|
+
const lead = lastIndexBefore(tagged, be, (entry) => !isAuxiliary(entry));
|
|
129
142
|
const relative = tagged[lead];
|
|
130
143
|
if (relative === undefined || !RELATIVE_TAG.has(relative.pos)) return false;
|
|
131
144
|
// 文頭の That was decided. / Which was chosen? は、前に指す名詞が無いので述語。
|
|
132
|
-
const antecedent = tagged
|
|
145
|
+
const antecedent = tagged[lastIndexBefore(tagged, lead, (entry) => entry.pos !== ",")];
|
|
133
146
|
return antecedent !== undefined && NOMINAL_TAG.has(antecedent.pos);
|
|
134
147
|
};
|
|
135
148
|
|
|
@@ -230,23 +243,26 @@ const withEmphasis = (token: Token): Token =>
|
|
|
230
243
|
* wink は位置を返さないので、表層を順に照合して復元する。
|
|
231
244
|
* 見つからないものは飛ばし、カーソルは進めない。位置の当てずっぽうを下流に流さない。
|
|
232
245
|
*/
|
|
233
|
-
const locate = (text: string, tagged: readonly Tagged[]): Token[] =>
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
}
|
|
246
|
-
|
|
247
|
-
}
|
|
248
|
-
|
|
249
|
-
|
|
246
|
+
const locate = (text: string, tagged: readonly Tagged[]): Token[] => {
|
|
247
|
+
// 語を足すたびに並びを作り直すと、長い文で語数の二乗になる。一つの並びに足していく。
|
|
248
|
+
const tokens: Token[] = [];
|
|
249
|
+
let cursor = 0;
|
|
250
|
+
tagged.forEach((entry, at) => {
|
|
251
|
+
const start = text.indexOf(entry.value, cursor);
|
|
252
|
+
if (start === -1) return;
|
|
253
|
+
const end = start + entry.value.length;
|
|
254
|
+
const token = {
|
|
255
|
+
span: { start, end },
|
|
256
|
+
surface: entry.value,
|
|
257
|
+
pos: properNounChecked(entry.value, upos(entry.pos)),
|
|
258
|
+
...(entry.lemma === undefined ? {} : { lemma: entry.lemma }),
|
|
259
|
+
...featuresOf(tagged, at),
|
|
260
|
+
};
|
|
261
|
+
tokens.push(withEmphasis(token));
|
|
262
|
+
cursor = end;
|
|
263
|
+
});
|
|
264
|
+
return tokens;
|
|
265
|
+
};
|
|
250
266
|
|
|
251
267
|
/** 英語の語はこれより長くならない。超える並びは語として読まない。 */
|
|
252
268
|
const RUN_LIMIT = 1000;
|
package/src/regexp.ts
ADDED
package/src/structure.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
|
|
2
|
-
import { citedDocumentAfter, citedDocumentBefore,
|
|
2
|
+
import { citedDocumentAfter, citedDocumentBefore, listedTagAround } from "./citation.ts";
|
|
3
|
+
import { citedCodeBefore, codeVocabulary, titledCodeAt } from "./code-citation.ts";
|
|
4
|
+
import { loadLexicons } from "./lexicons.ts";
|
|
3
5
|
import { membersAfter } from "./reference-list.ts";
|
|
4
6
|
import { parseRoman } from "./roman.ts";
|
|
5
7
|
import { dates } from "./dates.ts";
|
|
@@ -140,16 +142,19 @@ const glossedDocument = (gloss: Gloss): string | undefined => {
|
|
|
140
142
|
return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
|
|
141
143
|
};
|
|
142
144
|
|
|
145
|
+
const LEXICONS = loadLexicons();
|
|
146
|
+
const CODES = codeVocabulary(LEXICONS);
|
|
147
|
+
|
|
143
148
|
/** The other document a reference names, or else a bracketed tag that the core checks against the document's list. */
|
|
144
149
|
const citation = (text: string, start: number, end: number, document: string | undefined): Readonly<Record<string, string>> => {
|
|
145
150
|
if (document !== undefined) return { document };
|
|
146
|
-
const citedTag =
|
|
151
|
+
const citedTag = listedTagAround(text, start, end);
|
|
147
152
|
return citedTag === undefined ? {} : { citedTag };
|
|
148
153
|
};
|
|
149
154
|
|
|
150
155
|
/**
|
|
151
156
|
* "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
|
|
152
|
-
* "Section 9 of the Master Agreement"
|
|
157
|
+
* "Section 9 of the Master Agreement" and "35 CFR §122" carry the other document's name, and are not looked up in this tree.
|
|
153
158
|
*/
|
|
154
159
|
const references = (text: string): Mention[] => {
|
|
155
160
|
const gloss: Gloss = { depth: 0, scanned: 0, anchors: [] };
|
|
@@ -158,7 +163,7 @@ const references = (text: string): Mention[] => {
|
|
|
158
163
|
if (main === undefined) return [];
|
|
159
164
|
const { parts, end } = subdivisions(text, match.index + match[0].length);
|
|
160
165
|
advance(gloss, text, match.index);
|
|
161
|
-
const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index);
|
|
166
|
+
const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index) ?? citedCodeBefore(text, match.index, CODES);
|
|
162
167
|
const document = cited ?? glossedDocument(gloss);
|
|
163
168
|
gloss.scanned = Math.max(gloss.scanned, end);
|
|
164
169
|
if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
|
|
@@ -278,8 +283,19 @@ const quantities = (text: string): Mention[] =>
|
|
|
278
283
|
return unit === undefined || Number.isNaN(value) || isWordChar(text[match.index - 1]) ? [] : [{ start: match.index, end, attrs: { value, unit } }];
|
|
279
284
|
});
|
|
280
285
|
|
|
281
|
-
|
|
282
|
-
|
|
286
|
+
const MEASURE_UNITS = (LEXICONS["measure-unit"] ?? []).map((entry) => entry.pattern);
|
|
287
|
+
|
|
288
|
+
/** A letter, digit or hyphen right after the symbol makes it the start of a word: "2.1 mmap", "5.2.2.4 min-fresh". */
|
|
289
|
+
const CONTINUES_WORD = /^[\p{Script=Latin}\p{Nd}_-]/u;
|
|
290
|
+
|
|
291
|
+
const startsWithMeasureUnit = (rest: string): boolean => MEASURE_UNITS.some((unit) => rest.startsWith(unit) && !CONTINUES_WORD.test(rest.slice(unit.length)));
|
|
292
|
+
|
|
293
|
+
/**
|
|
294
|
+
* "2.5 days", "1.5 times" and "1.5 mM in each" are amounts, not section 2.5 titled "days". "40 CFR § 163.25" is title 40 of
|
|
295
|
+
* another code, not section 40 of this document.
|
|
296
|
+
*/
|
|
297
|
+
const countedAfter = (_number: string, rest: string): boolean =>
|
|
298
|
+
unitAfter(` ${rest}`, 0) !== undefined || startsWithMeasureUnit(rest) || titledCodeAt(rest, CODES);
|
|
283
299
|
|
|
284
300
|
export const structure: StructurePatterns = {
|
|
285
301
|
numbered,
|