@chaffjs/lang-en 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/apostrophe.d.ts +2 -0
- package/dist/apostrophe.d.ts.map +1 -0
- package/dist/apostrophe.js +8 -0
- package/dist/apostrophe.js.map +1 -0
- package/dist/citation.d.ts +2 -0
- package/dist/citation.d.ts.map +1 -1
- package/dist/citation.js +37 -12
- package/dist/citation.js.map +1 -1
- package/dist/closing-quote.d.ts +6 -0
- package/dist/closing-quote.d.ts.map +1 -0
- package/dist/closing-quote.js +50 -0
- package/dist/closing-quote.js.map +1 -0
- package/dist/dates.js +1 -1
- package/dist/dates.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -2
- package/dist/index.js.map +1 -1
- package/dist/item-style.d.ts +29 -0
- package/dist/item-style.d.ts.map +1 -0
- package/dist/item-style.js +124 -0
- package/dist/item-style.js.map +1 -0
- package/dist/lexicons.d.ts.map +1 -1
- package/dist/lexicons.js +5 -6
- package/dist/lexicons.js.map +1 -1
- package/dist/long-runs.d.ts +6 -0
- package/dist/long-runs.d.ts.map +1 -0
- package/dist/long-runs.js +6 -0
- package/dist/long-runs.js.map +1 -0
- package/dist/pos.d.ts.map +1 -1
- package/dist/pos.js +40 -6
- package/dist/pos.js.map +1 -1
- package/dist/proper-noun.d.ts +21 -0
- package/dist/proper-noun.d.ts.map +1 -1
- package/dist/proper-noun.js +37 -0
- package/dist/proper-noun.js.map +1 -1
- package/dist/quoted-stop.d.ts +13 -0
- package/dist/quoted-stop.d.ts.map +1 -0
- package/dist/quoted-stop.js +45 -0
- package/dist/quoted-stop.js.map +1 -0
- package/dist/sentence-split.d.ts +2 -0
- package/dist/sentence-split.d.ts.map +1 -1
- package/dist/sentence-split.js +4 -2
- package/dist/sentence-split.js.map +1 -1
- package/dist/structure.d.ts.map +1 -1
- package/dist/structure.js +26 -67
- package/dist/structure.js.map +1 -1
- package/lexicons/common-acronym.yaml +4 -0
- package/lexicons/date-time-unit.yaml +6 -0
- package/lexicons/emphasis-word.yaml +1 -0
- package/lexicons/excessive-hedging.yaml +6 -0
- package/lexicons/fixed-phrase.yaml +14 -0
- package/lexicons/hedge-frame.yaml +9 -0
- package/lexicons/hedge-scope.yaml +7 -0
- package/lexicons/honorific.yaml +15 -0
- package/lexicons/http-method.yaml +14 -0
- package/lexicons/participle-word.yaml +8 -0
- package/lexicons/quantity-noun.yaml +6 -0
- package/lexicons/relative-word.yaml +10 -0
- package/lexicons/subject-pronoun.yaml +13 -0
- package/lexicons/superlative-amount.yaml +7 -0
- package/lexicons/superlative-bound.yaml +8 -0
- package/package.json +1 -1
- package/src/apostrophe.ts +8 -0
- package/src/citation.ts +41 -13
- package/src/closing-quote.ts +52 -0
- package/src/dates.ts +1 -1
- package/src/index.ts +5 -2
- package/src/item-style.ts +130 -0
- package/src/lexicons.ts +9 -8
- package/src/long-runs.ts +6 -0
- package/src/pos.ts +47 -6
- package/src/proper-noun.ts +43 -0
- package/src/quoted-stop.ts +52 -0
- package/src/sentence-split.ts +4 -2
- package/src/structure.ts +25 -69
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import type { Span } from "chaffjs/plugin";
|
|
2
|
+
import { isAbbreviation } from "./sentence-split.ts";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* 米国式の引用は、文末のピリオドを閉じ引用符の内側に置く(…called it "a fair trial." Plainly, …)。
|
|
6
|
+
* sentence-splitter は引用符の中の句点で文を切らず、引用符が閉じた後の空白でも切らないので、次の文が前の文につながる。
|
|
7
|
+
*
|
|
8
|
+
* 分割器が返した一つの文を、文末の記号 + 閉じ引用符 + 空白 + 次の文の頭(大文字。開き括弧・引用符の後の大文字も)で切る。
|
|
9
|
+
* 引用符の中の語が略語・頭文字("U.S." "Dr." "J.")のときと、括弧が開いたままのときは切らない。
|
|
10
|
+
*/
|
|
11
|
+
/** 次の文の頭。空白の後の大文字。開き括弧・引用符の後の大文字も文頭である。 */
|
|
12
|
+
export const NEXT_SENTENCE_HEAD = String.raw`\s+[\p{Ps}\p{Pi}"']*\p{Lu}`;
|
|
13
|
+
const QUOTED_STOP = new RegExp(String.raw`[.?!]["'”’]+(?=${NEXT_SENTENCE_HEAD})`, "gu");
|
|
14
|
+
// ピリオドの前の語は、空白・開き引用符・開き括弧の後から数える。
|
|
15
|
+
const WORD_START = /[\s\p{Ps}\p{Pi}"']/u;
|
|
16
|
+
// 一字ずつピリオドを打った語(J. / U.S. / e.g.)は、略語の一覧に無くても略語。
|
|
17
|
+
const INITIALS = /^(?:\p{L}\.)+$/u;
|
|
18
|
+
// 分割器が一組として読む括弧(( [ { ( [ { 【 《 「 『 …)は、Unicode の開き・閉じ括弧の分類にすべて入る。
|
|
19
|
+
const OPENER = /\p{Ps}/u;
|
|
20
|
+
const CLOSER = /\p{Pe}/u;
|
|
21
|
+
|
|
22
|
+
const endsWithAbbreviation = (sentence: string, stop: number): boolean => {
|
|
23
|
+
if (sentence[stop] !== ".") return false;
|
|
24
|
+
const word = `${sentence.slice(0, stop).split(WORD_START).at(-1) ?? ""}.`;
|
|
25
|
+
return isAbbreviation(word) || INITIALS.test(word);
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
// 対の無い閉じ括弧(「1) 最初の項目」の番号)は、後から開く括弧を打ち消さない。
|
|
29
|
+
const depthAfter = (depth: number, char: string): number => {
|
|
30
|
+
if (OPENER.test(char)) return depth + 1;
|
|
31
|
+
return CLOSER.test(char) ? Math.max(0, depth - 1) : depth;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
const bracketOpenAt = (sentence: string, index: number): boolean => Array.from(sentence.slice(0, index)).reduce(depthAfter, 0) > 0;
|
|
35
|
+
|
|
36
|
+
type Cut = { readonly end: number; readonly next: number };
|
|
37
|
+
|
|
38
|
+
const cutsIn = (sentence: string): Cut[] =>
|
|
39
|
+
[...sentence.matchAll(QUOTED_STOP)]
|
|
40
|
+
.filter((match) => !endsWithAbbreviation(sentence, match.index) && !bracketOpenAt(sentence, match.index))
|
|
41
|
+
.map((match) => {
|
|
42
|
+
const end = match.index + match[0].length;
|
|
43
|
+
return { end, next: end + (/^\s+/u.exec(sentence.slice(end))?.[0].length ?? 0) };
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
/** span の文を、閉じ引用符の内側で閉じた文ごとに分けた span。切れ目が無ければ span そのもの一つ。 */
|
|
47
|
+
export const splitAtQuotedStops = (text: string, span: Span): Span[] => {
|
|
48
|
+
const cuts = cutsIn(text.slice(span.start, span.end));
|
|
49
|
+
const starts = [0, ...cuts.map((cut) => cut.next)];
|
|
50
|
+
const ends = [...cuts.map((cut) => cut.end), span.end - span.start];
|
|
51
|
+
return starts.map((start, index) => ({ start: span.start + start, end: span.start + (ends[index] ?? start) }));
|
|
52
|
+
};
|
package/src/sentence-split.ts
CHANGED
|
@@ -45,6 +45,8 @@ const { language } = DefaultAbbrMarkerOptions;
|
|
|
45
45
|
const ABBREVIATIONS = new Set(
|
|
46
46
|
[...language.ABBREVIATIONS, ...language.PREPOSITIVE_ABBREVIATIONS, ...language.EXCLAMATION_WORDS].map((word) => word.toLowerCase()),
|
|
47
47
|
);
|
|
48
|
+
/** 分割器が略語と読む語(ピリオドまで含めて渡す)。「Dr.」「U.S.」「Jan.」。 */
|
|
49
|
+
export const isAbbreviation = (word: string): boolean => ABBREVIATIONS.has(word.toLowerCase());
|
|
48
50
|
// 「J. Smith」のような語は前の語を見て略語か決まる。前の語が切れ目の向こうにあると判定が変わる。
|
|
49
51
|
const CAPITAL_DOT = /\p{Lu}\./gu;
|
|
50
52
|
|
|
@@ -76,8 +78,8 @@ const indexesOf = (text: string, pattern: RegExp): number[] => [...text.matchAll
|
|
|
76
78
|
/** 句点の並びの直後。和文は直前が仮名・漢字の句点、英文は略語でない語のピリオド。 */
|
|
77
79
|
const stopEnds = (text: string): number[] => {
|
|
78
80
|
const japanese = [...text.matchAll(JAPANESE_STOPS)].filter((match) => isCJK(text[match.index - 1]));
|
|
79
|
-
const english = [...text.matchAll(ENGLISH_STOP)].filter((match) => !
|
|
80
|
-
return [...japanese, ...english].map((match) => match.index + match[0].length).
|
|
81
|
+
const english = [...text.matchAll(ENGLISH_STOP)].filter((match) => !isAbbreviation(match[0]));
|
|
82
|
+
return [...japanese, ...english].map((match) => match.index + match[0].length).toSorted((a, b) => a - b);
|
|
81
83
|
};
|
|
82
84
|
|
|
83
85
|
const skipSpaces = (text: string, from: number): number => {
|
package/src/structure.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
|
|
2
|
-
import { citedDocumentAfter, citedDocumentBefore } from "./citation.ts";
|
|
2
|
+
import { citedDocumentAfter, citedDocumentBefore, hyphenatedTagAround } from "./citation.ts";
|
|
3
3
|
import { membersAfter } from "./reference-list.ts";
|
|
4
4
|
import { parseRoman } from "./roman.ts";
|
|
5
5
|
import { dates } from "./dates.ts";
|
|
6
6
|
import { definitionScopeDepth, definitions, opensDefinitionScope } from "./definitions.ts";
|
|
7
7
|
import { CHAPTER_DEPTH, PART_DEPTH } from "./depth.ts";
|
|
8
|
+
import { continuesAddress, continuesOpen, depthFor, INSERTED, ordinalOf, styleOf } from "./item-style.ts";
|
|
8
9
|
|
|
9
10
|
// Contracts, specifications and statutes in English. core nests what this reads; it does not know
|
|
10
11
|
// how English numbers its articles.
|
|
@@ -28,14 +29,8 @@ const titleOf = (rest: string): string | undefined => {
|
|
|
28
29
|
|
|
29
30
|
const ARTICLE = /^\s{0,3}(?:ARTICLE|Article)\s+(?<n>\d{1,3}|[IVXLC]{1,7})\b(?<rest>.*)$/u;
|
|
30
31
|
const SECTION = /^\s{0,3}(?:SECTION|Section|§)\s*(?<n>\d{1,3}(?:\.\d{1,3}){0,5})\b(?<rest>.*)$/u;
|
|
31
|
-
/**
|
|
32
|
-
|
|
33
|
-
* outside the sequence: it has no ordinal, so "(1)" after "(A1)" is still the first.
|
|
34
|
-
*/
|
|
35
|
-
const INSERTED = "\\d{1,3}[A-Z]{1,2}|[A-Z]{1,2}\\d{1,3}";
|
|
36
|
-
const LETTERED = new RegExp(`^\\s{0,6}\\((?<n>[a-z]{1,4}|\\d{1,3}|${INSERTED})\\)\\s+(?<rest>\\S.*)$`, "u");
|
|
37
|
-
const IS_INSERTED = new RegExp(`^(?:${INSERTED})$`, "u");
|
|
38
|
-
const MULTI_ROMAN = /^(?:ii|iii|iv|vi|vii|viii|ix)$/u;
|
|
32
|
+
/** "(a) text", "(ii) text", "(B) text", or the label alone on its line with the text below it. */
|
|
33
|
+
const LETTERED = new RegExp(`^\\s{0,6}\\((?<n>[a-z]{1,4}|[A-Z]{1,4}|\\d{1,3}|${INSERTED})\\)(?:\\s+(?<rest>\\S.*))?\\s*$`, "u");
|
|
39
34
|
|
|
40
35
|
const headed = (pattern: RegExp, line: string, numbering: string, label: (n: string) => string): NumberedLine | undefined => {
|
|
41
36
|
const groups = pattern.exec(line)?.groups;
|
|
@@ -56,70 +51,24 @@ const headed = (pattern: RegExp, line: string, numbering: string, label: (n: str
|
|
|
56
51
|
};
|
|
57
52
|
};
|
|
58
53
|
|
|
59
|
-
type Style = "letter" | "roman" | "digit";
|
|
60
|
-
|
|
61
|
-
/**
|
|
62
|
-
* 開いている項目の書き方。"(ii)" はローマ数字、"(b)" は英字。一文字の "(i)" はどちらにも読めるので、
|
|
63
|
-
* 開いたときの読みを覚えておく代わりに、一つ上に英字が開いていたかで決め直す。
|
|
64
|
-
*/
|
|
65
|
-
const styleOfLabel = (open: NumberedLine, index: number, all: readonly NumberedLine[]): Style | undefined => {
|
|
66
|
-
const inner = /^\((?<n>[A-Za-z0-9]{1,5})\)$/u.exec(open.label)?.groups?.["n"];
|
|
67
|
-
if (inner === undefined) return undefined;
|
|
68
|
-
if (/^\d+$/u.test(inner) || IS_INSERTED.test(inner)) return "digit";
|
|
69
|
-
if (MULTI_ROMAN.test(inner)) return "roman";
|
|
70
|
-
const above = all[index - 1];
|
|
71
|
-
return /^[ivx]$/u.test(inner) && above !== undefined && styleOfLabel(above, index - 1, all) === "letter" && above.depth < open.depth ? "roman" : "letter";
|
|
72
|
-
};
|
|
73
|
-
|
|
74
|
-
const styles = (context: NumberingContext): (Style | undefined)[] => context.open.map((open, index, all) => styleOfLabel(open, index, all));
|
|
75
|
-
|
|
76
|
-
/** "(h)" の次の "(i)" は英字。開いている英字の次の文字なら、ローマ数字とは読まない。 */
|
|
77
|
-
const followsLetter = (raw: string, context: NumberingContext): boolean =>
|
|
78
|
-
context.open.some((open, index) => styles(context)[index] === "letter" && open.number.charCodeAt(0) + 1 === raw.charCodeAt(0));
|
|
79
|
-
|
|
80
|
-
/** "(i)" is a roman numeral right under "(a)", or when a roman list is already open; the letter i otherwise. */
|
|
81
|
-
const styleOf = (raw: string, context: NumberingContext): Style => {
|
|
82
|
-
if (/^\d+$/u.test(raw) || IS_INSERTED.test(raw)) return "digit";
|
|
83
|
-
if (MULTI_ROMAN.test(raw)) return "roman";
|
|
84
|
-
if (!/^[ivx]$/u.test(raw) || followsLetter(raw, context)) return "letter";
|
|
85
|
-
const open = styles(context);
|
|
86
|
-
return open.includes("roman") || open.at(-1) === "letter" ? "roman" : "letter";
|
|
87
|
-
};
|
|
88
|
-
|
|
89
|
-
/**
|
|
90
|
-
* A sibling has the depth of the open item written the same way: "(b)" closes "(i)" and sits beside "(a)".
|
|
91
|
-
* A new way of numbering goes one deeper than whatever is open.
|
|
92
|
-
*/
|
|
93
|
-
const depthFor = (style: Style, context: NumberingContext): number => {
|
|
94
|
-
const open = styles(context);
|
|
95
|
-
const sibling = [...context.open].reverse().find((_item, reversed) => open[context.open.length - 1 - reversed] === style);
|
|
96
|
-
return sibling?.depth ?? (context.open.at(-1)?.depth ?? 0) + 1;
|
|
97
|
-
};
|
|
98
|
-
|
|
99
|
-
const LETTER_BEFORE_A = "a".charCodeAt(0) - 1;
|
|
100
|
-
|
|
101
|
-
/** "(b)" は 2 番目、"(ii)" も 2 番目。二文字以上の英字("(aa)")は並びが決まらないので付けない。 */
|
|
102
|
-
const ordinalOf = (raw: string, style: Style): number | undefined => {
|
|
103
|
-
if (style === "digit") return IS_INSERTED.test(raw) ? undefined : Number(raw);
|
|
104
|
-
if (style === "roman") return parseRoman(raw);
|
|
105
|
-
return raw.length === 1 ? raw.charCodeAt(0) - LETTER_BEFORE_A : undefined;
|
|
106
|
-
};
|
|
107
|
-
|
|
108
54
|
const lettered = (line: string, context: NumberingContext): NumberedLine | undefined => {
|
|
109
55
|
const groups = LETTERED.exec(line)?.groups;
|
|
110
56
|
const raw = groups?.["n"];
|
|
111
57
|
if (groups === undefined || raw === undefined) return undefined;
|
|
112
58
|
const style = styleOf(raw, context);
|
|
59
|
+
const ordinal = style === undefined ? undefined : ordinalOf(raw, style);
|
|
60
|
+
const rest = groups["rest"];
|
|
61
|
+
if (style === undefined || (rest === undefined && !continuesOpen(style, ordinal, context))) return undefined;
|
|
113
62
|
// 番地は書かれたままの "ii" を使う。参照「Section 4.2(a)(ii)」も同じ形で書かれるので、そのまま引ける。
|
|
114
63
|
return {
|
|
115
64
|
kind: "item",
|
|
116
|
-
depth: depthFor(style, context),
|
|
117
|
-
ordinal
|
|
65
|
+
depth: depthFor(style, context, ordinal),
|
|
66
|
+
ordinal,
|
|
118
67
|
number: raw,
|
|
119
68
|
absolute: false,
|
|
120
69
|
label: `(${raw})`,
|
|
121
70
|
heading: "",
|
|
122
|
-
rest:
|
|
71
|
+
rest: rest?.trim() ?? "",
|
|
123
72
|
};
|
|
124
73
|
};
|
|
125
74
|
|
|
@@ -144,13 +93,13 @@ const numbered = (line: string, context: NumberingContext): NumberedLine | undef
|
|
|
144
93
|
lettered(line, context);
|
|
145
94
|
|
|
146
95
|
const REFERENCE = /(?<word>\b[Ss]ections?|\b[Aa]rticles?|§) ?(?<n>\d{1,3}(?:\.\d{1,3}){0,5}|[IVXLC]{1,7})\b/gu;
|
|
147
|
-
/** "(a)", "(ii)", "(3)", and an inserted "(A1)" or "(2A)": the same labels the tree reads. */
|
|
148
|
-
const SUBDIVISION = new RegExp(`^\\((?<p>[a-z0-9]{1,4}|${INSERTED})\\)`, "u");
|
|
96
|
+
/** "(a)", "(ii)", "(3)", "(B)", and an inserted "(A1)" or "(2A)": the same labels the tree reads. */
|
|
97
|
+
const SUBDIVISION = new RegExp(`^\\((?<p>[a-z0-9]{1,4}|[A-Z]{1,4}|${INSERTED})\\)`, "u");
|
|
149
98
|
/** The longest label, with its parentheses: "(ZZ999)". */
|
|
150
99
|
const MAX_SUBDIVISION_LENGTH = 7;
|
|
151
100
|
|
|
152
|
-
/** "(a)(
|
|
153
|
-
const MAX_SUBDIVISIONS =
|
|
101
|
+
/** "(a)(1)(i)(A)(1)", a US regulation's deepest paragraph, is as deep as a reference goes; more parentheses are text. */
|
|
102
|
+
const MAX_SUBDIVISIONS = 5;
|
|
154
103
|
|
|
155
104
|
/**
|
|
156
105
|
* "(a)(ii)" のような続きの括弧を、正規表現を複雑にせずに一つずつ読む。
|
|
@@ -161,7 +110,7 @@ const subdivisions = (text: string, from: number): { readonly parts: readonly st
|
|
|
161
110
|
let end = from;
|
|
162
111
|
while (parts.length < MAX_SUBDIVISIONS) {
|
|
163
112
|
const part = SUBDIVISION.exec(text.slice(end, end + MAX_SUBDIVISION_LENGTH))?.groups?.["p"];
|
|
164
|
-
if (part === undefined) break;
|
|
113
|
+
if (part === undefined || !continuesAddress(part, parts)) break;
|
|
165
114
|
parts.push(part);
|
|
166
115
|
end += part.length + 2;
|
|
167
116
|
}
|
|
@@ -191,6 +140,13 @@ const glossedDocument = (gloss: Gloss): string | undefined => {
|
|
|
191
140
|
return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
|
|
192
141
|
};
|
|
193
142
|
|
|
143
|
+
/** The other document a reference names, or else a bracketed tag that the core checks against the document's list. */
|
|
144
|
+
const citation = (text: string, start: number, end: number, document: string | undefined): Readonly<Record<string, string>> => {
|
|
145
|
+
if (document !== undefined) return { document };
|
|
146
|
+
const citedTag = hyphenatedTagAround(text, start, end);
|
|
147
|
+
return citedTag === undefined ? {} : { citedTag };
|
|
148
|
+
};
|
|
149
|
+
|
|
194
150
|
/**
|
|
195
151
|
* "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
|
|
196
152
|
* "Section 9 of the Master Agreement" carries the other document's name, and is not looked up in this tree.
|
|
@@ -207,9 +163,9 @@ const references = (text: string): Mention[] => {
|
|
|
207
163
|
gloss.scanned = Math.max(gloss.scanned, end);
|
|
208
164
|
if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
|
|
209
165
|
const numbering = /^[Aa]/u.test(match.groups?.["word"] ?? "") ? "article" : "section";
|
|
210
|
-
const shared = { numbering, ...(
|
|
166
|
+
const shared = { numbering, ...citation(text, match.index, end, document) };
|
|
211
167
|
const first = { start: match.index, end, attrs: { target: [main, ...parts].join("."), label: text.slice(match.index, end), ...shared } };
|
|
212
|
-
const [plural, roman] = [
|
|
168
|
+
const [plural, roman] = [(match.groups?.["word"] ?? "").endsWith("s"), /^[IVXLC]+$/u.test(match.groups?.["n"] ?? "")];
|
|
213
169
|
return [first, ...membersAfter(text, end, [main, ...parts], shared, plural, roman)];
|
|
214
170
|
});
|
|
215
171
|
};
|
|
@@ -263,7 +219,7 @@ const obligations = (text: string): Mention[] => {
|
|
|
263
219
|
kept.push({ start, end, attrs: { marker, type } });
|
|
264
220
|
});
|
|
265
221
|
});
|
|
266
|
-
return kept.
|
|
222
|
+
return kept.toSorted((left, right) => left.start - right.start);
|
|
267
223
|
};
|
|
268
224
|
|
|
269
225
|
/** Find the number first, then look at what is right before (a currency) and right after (a unit). */
|