@chaffjs/lang-en 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/dist/citation.d.ts +5 -2
  2. package/dist/citation.d.ts.map +1 -1
  3. package/dist/citation.js +13 -9
  4. package/dist/citation.js.map +1 -1
  5. package/dist/code-citation.d.ts +15 -0
  6. package/dist/code-citation.d.ts.map +1 -0
  7. package/dist/code-citation.js +34 -0
  8. package/dist/code-citation.js.map +1 -0
  9. package/dist/definitions.d.ts.map +1 -1
  10. package/dist/definitions.js +4 -2
  11. package/dist/definitions.js.map +1 -1
  12. package/dist/index.d.ts +1 -1
  13. package/dist/index.d.ts.map +1 -1
  14. package/dist/index.js +10 -4
  15. package/dist/index.js.map +1 -1
  16. package/dist/japanese-run.d.ts +2 -0
  17. package/dist/japanese-run.d.ts.map +1 -0
  18. package/dist/japanese-run.js +9 -0
  19. package/dist/japanese-run.js.map +1 -0
  20. package/dist/label-stop.d.ts +13 -0
  21. package/dist/label-stop.d.ts.map +1 -0
  22. package/dist/label-stop.js +13 -0
  23. package/dist/label-stop.js.map +1 -0
  24. package/dist/lexicons.d.ts.map +1 -1
  25. package/dist/lexicons.js +4 -0
  26. package/dist/lexicons.js.map +1 -1
  27. package/dist/pos.d.ts.map +1 -1
  28. package/dist/pos.js +37 -17
  29. package/dist/pos.js.map +1 -1
  30. package/dist/regexp.d.ts +3 -0
  31. package/dist/regexp.d.ts.map +1 -0
  32. package/dist/regexp.js +3 -0
  33. package/dist/regexp.js.map +1 -0
  34. package/dist/structure.d.ts.map +1 -1
  35. package/dist/structure.js +17 -6
  36. package/dist/structure.js.map +1 -1
  37. package/lexicons/abbreviated-label.yaml +16 -0
  38. package/lexicons/ai-tell.yaml +98 -0
  39. package/lexicons/announcing-opener.yaml +19 -0
  40. package/lexicons/assistant-residue.yaml +55 -0
  41. package/lexicons/contrast-frame.yaml +10 -0
  42. package/lexicons/contrast-lead.yaml +15 -0
  43. package/lexicons/contrast-turn.yaml +11 -0
  44. package/lexicons/count-anchor.yaml +7 -0
  45. package/lexicons/count-counter.yaml +90 -0
  46. package/lexicons/count-hedge.yaml +67 -0
  47. package/lexicons/count-number.yaml +16 -0
  48. package/lexicons/cushion-phrase.yaml +6 -0
  49. package/lexicons/date-change-word.yaml +27 -0
  50. package/lexicons/document-kind.yaml +19 -0
  51. package/lexicons/double-negative.yaml +25 -0
  52. package/lexicons/email-attachment-note.yaml +7 -0
  53. package/lexicons/email-attribution.yaml +7 -0
  54. package/lexicons/email-header-field.yaml +28 -0
  55. package/lexicons/email-written-field.yaml +8 -0
  56. package/lexicons/enumeration-frame.yaml +37 -0
  57. package/lexicons/enumeration-joiner.yaml +9 -0
  58. package/lexicons/enumeration-negation.yaml +13 -0
  59. package/lexicons/figure-elsewhere.yaml +8 -0
  60. package/lexicons/figure-label.yaml +14 -0
  61. package/lexicons/measure-unit.yaml +78 -0
  62. package/lexicons/nominalization-phrase.yaml +27 -0
  63. package/lexicons/numbered-label.yaml +26 -0
  64. package/lexicons/percent-unit.yaml +7 -0
  65. package/lexicons/place-region.yaml +247 -0
  66. package/lexicons/placeholder-word.yaml +17 -0
  67. package/lexicons/range-connector.yaml +12 -0
  68. package/lexicons/range-frame.yaml +9 -0
  69. package/lexicons/requirement-either.yaml +5 -0
  70. package/lexicons/requirement-loophole.yaml +17 -0
  71. package/lexicons/requirement-marker.yaml +9 -0
  72. package/lexicons/requirement-modal-must.yaml +7 -0
  73. package/lexicons/requirement-open-end.yaml +9 -0
  74. package/lexicons/share-exception.yaml +10 -0
  75. package/lexicons/share-label.yaml +12 -0
  76. package/lexicons/spelling-ize.yaml +273 -0
  77. package/lexicons/spelling-variant.yaml +260 -0
  78. package/lexicons/stock-transition.yaml +17 -0
  79. package/lexicons/vague-clause-pointer.yaml +10 -0
  80. package/lexicons/vague-figure-pointer.yaml +14 -0
  81. package/package.json +1 -1
  82. package/src/citation.ts +13 -9
  83. package/src/code-citation.ts +52 -0
  84. package/src/definitions.ts +4 -2
  85. package/src/index.ts +14 -5
  86. package/src/japanese-run.ts +9 -0
  87. package/src/label-stop.ts +25 -0
  88. package/src/lexicons.ts +4 -0
  89. package/src/pos.ts +36 -20
  90. package/src/regexp.ts +2 -0
  91. package/src/structure.ts +22 -6
package/src/index.ts CHANGED
@@ -1,14 +1,19 @@
1
1
  import { loadLexicons } from "./lexicons.ts";
2
2
  import { unmarkNumberStops } from "./number-stop.ts";
3
+ import { labelStops, unmarkLabelStops } from "./label-stop.ts";
3
4
  import { sentenceSpans } from "./sentence-split.ts";
4
5
  import { splitAtQuotedStops } from "./quoted-stop.ts";
5
6
  import { reattachClosingQuotes } from "./closing-quote.ts";
6
7
  import { structure } from "./structure.ts";
7
8
  import { isReady, prepare, tokenize } from "./pos.ts";
8
- import type { AdapterNeeds, LanguageAdapter, Segmentation, Sentence } from "chaffjs/plugin";
9
+ import { isJapaneseRun } from "./japanese-run.ts";
10
+ import type { AdapterNeeds, EmbeddedLanguage, LanguageAdapter, Segmentation, Sentence, Span } from "chaffjs/plugin";
9
11
 
10
12
  // chaff からは型だけを取る。実行時の値依存を作らない。アダプタは単体で動く。
11
13
 
14
+ const LEXICONS = loadLexicons();
15
+ const LABEL_STOPS = labelStops((LEXICONS["abbreviated-label"] ?? []).map((entry) => entry.pattern));
16
+
12
17
  const LATIN_LETTER = /[a-z]/giu;
13
18
  const COUNTABLE = /\S/gu;
14
19
 
@@ -25,9 +30,13 @@ const withTokens = (sentence: Sentence): Sentence => {
25
30
  };
26
31
  };
27
32
 
33
+ const JAPANESE: EmbeddedLanguage = { id: "ja", lengthUnit: "char" };
34
+
35
+ const withLanguage = (text: string, span: Span): Sentence => (isJapaneseRun(text) ? { span, text, embeddedLanguage: JAPANESE } : { span, text });
36
+
28
37
  /**
29
38
  * 英語は sentence-splitter の既定にほぼ任せる。"Dr." "e.g." "U.S." "$3.50" を
30
- * いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませること、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
39
+ * いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませることと、番号の前の略した名前(FIG. 1、Vol. XLIII)の点で切らないこと、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
31
40
  */
32
41
  export const adapter: LanguageAdapter = {
33
42
  kind: "language",
@@ -51,11 +60,11 @@ export const adapter: LanguageAdapter = {
51
60
  if (total === 0) return 0;
52
61
  return [...source.matchAll(LATIN_LETTER)].length / total;
53
62
  },
54
- lexicons: loadLexicons(),
63
+ lexicons: LEXICONS,
55
64
  structure,
56
65
  segment: (text: string): Segmentation => {
57
- const quotedStops = sentenceSpans(unmarkNumberStops(text)).flatMap((span) => splitAtQuotedStops(text, span));
58
- const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => ({ span, text: text.slice(span.start, span.end) }));
66
+ const quotedStops = sentenceSpans(unmarkNumberStops(unmarkLabelStops(text, LABEL_STOPS))).flatMap((span) => splitAtQuotedStops(text, span));
67
+ const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => withLanguage(text.slice(span.start, span.end), span));
59
68
  return { sentences: isReady() ? sentences.map(withTokens) : sentences };
60
69
  },
61
70
  };
@@ -0,0 +1,9 @@
1
+ /**
2
+ * A Japanese sentence in an English document (a quoted notice, a bilingual title): it has kana, and kana or kanji make
3
+ * up at least half of its letters (digits and symbols are not counted). Kanji alone do not count, as a Chinese name in English text is not Japanese.
4
+ */
5
+ const KANA = /[\p{Script=Hiragana}\p{Script=Katakana}]/u;
6
+ const JAPANESE = /[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}]/gu;
7
+ const LETTER = /\p{L}/gu;
8
+
9
+ export const isJapaneseRun = (text: string): boolean => KANA.test(text) && [...text.matchAll(JAPANESE)].length * 2 >= [...text.matchAll(LETTER)].length;
@@ -0,0 +1,25 @@
1
+ import { escapeRegExp } from "./regexp.ts";
2
+
3
+ /**
4
+ * sentence-splitter ends a sentence at the full stop of a label before its number: "FIG." and "1 illustrates …" were two
5
+ * sentences, and so were "Vol." and "XLIII (1979)". Where a number follows, that full stop is replaced with a letter
6
+ * before splitting. The length does not change, so the spans fit the original text.
7
+ * A lone "I" after a label is the pronoun ("He said No. I left."), and a word ("the last Fig. The tree") is not a number.
8
+ */
9
+
10
+ /** Built once from the lexicon; undefined when no label ends in a full stop. */
11
+ export type LabelStops = RegExp | undefined;
12
+
13
+ const NUMBER_AFTER = String.raw`(?=\s+(?:\d|(?:[IVXLCDM]{2,}|[VXLCDM])(?![\p{L}\p{N}_])))`;
14
+ const PLAIN_LETTER = "n";
15
+
16
+ /** The labels as written and in capitals, without their full stop ("Fig." → Fig, FIG). */
17
+ export const labelStops = (labels: readonly string[]): LabelStops => {
18
+ const stems = labels.filter((label) => label.endsWith(".")).flatMap((label) => [label.slice(0, -1), label.slice(0, -1).toUpperCase()]);
19
+ if (stems.length === 0) return undefined;
20
+ return new RegExp(String.raw`(?<![\p{L}\p{N}_.])(?:${[...new Set(stems)].map(escapeRegExp).join("|")})\.${NUMBER_AFTER}`, "gu");
21
+ };
22
+
23
+ /** The text with the full stop of every label before a number replaced by a letter. */
24
+ export const unmarkLabelStops = (text: string, stops: LabelStops): string =>
25
+ stops === undefined ? text : text.replace(stops, (label: string) => `${label.slice(0, -1)}${PLAIN_LETTER}`);
package/src/lexicons.ts CHANGED
@@ -14,11 +14,15 @@ const toEntry = (raw: unknown): LexiconEntry | undefined => {
14
14
  if (!isRecord(raw) || typeof raw["pattern"] !== "string") return undefined;
15
15
  const weight = raw["weight"];
16
16
  const instead = raw["instead_of"];
17
+ const rewrite = raw["rewrite"];
18
+ const group = raw["group"];
17
19
  return {
18
20
  pattern: raw["pattern"],
19
21
  weight: typeof weight === "number" ? weight : undefined,
20
22
  instead_of: typeof instead === "string" ? instead : undefined,
21
23
  position: POSITIONS.find((position) => position === raw["position"]),
24
+ rewrite: typeof rewrite === "string" ? rewrite : undefined,
25
+ ...(typeof group === "string" ? { group } : {}),
22
26
  };
23
27
  };
24
28
 
package/src/pos.ts CHANGED
@@ -100,9 +100,22 @@ const BE = new Set(["be", "am", "is", "are", "was", "were", "been", "being"]);
100
100
 
101
101
  const isBe = (entry: Tagged): boolean => BE.has(entry.lemma ?? entry.value.toLowerCase()) || BE.has(entry.value.toLowerCase());
102
102
 
103
+ /**
104
+ * end(0 以上)より前で test に合う最後の位置。無ければ -1。過去分詞の多い長い文で、分詞ごとに文の頭から写すと語数の二乗になるので、後ろから探す。
105
+ */
106
+ const lastIndexBefore = (tagged: readonly Tagged[], end: number, test: (entry: Tagged) => boolean): number => {
107
+ let at = Math.min(end, tagged.length);
108
+ while (at > 0) {
109
+ at -= 1;
110
+ const entry = tagged[at];
111
+ if (entry !== undefined && test(entry)) return at;
112
+ }
113
+ return -1;
114
+ };
115
+
103
116
  /** 過去分詞の前の be の位置。無ければ -1。 */
104
117
  const beBefore = (tagged: readonly Tagged[], at: number): number => {
105
- const head = tagged.slice(0, at).findLastIndex((entry) => !SKIPPABLE.has(entry.pos));
118
+ const head = lastIndexBefore(tagged, at, (entry) => !SKIPPABLE.has(entry.pos));
106
119
  const entry = tagged[head];
107
120
  return entry !== undefined && isBe(entry) ? head : -1;
108
121
  };
@@ -125,11 +138,11 @@ const NOMINAL_TAG = new Set(["NN", "NNS", "NNP", "NNPS", "PRP", "CD", "DT"]);
125
138
  const RELATIVE_TAG = new Set(["WDT", "WP"]);
126
139
 
127
140
  const inRelativeClause = (tagged: readonly Tagged[], be: number): boolean => {
128
- const lead = tagged.slice(0, be).findLastIndex((entry) => !isAuxiliary(entry));
141
+ const lead = lastIndexBefore(tagged, be, (entry) => !isAuxiliary(entry));
129
142
  const relative = tagged[lead];
130
143
  if (relative === undefined || !RELATIVE_TAG.has(relative.pos)) return false;
131
144
  // 文頭の That was decided. / Which was chosen? は、前に指す名詞が無いので述語。
132
- const antecedent = tagged.slice(0, lead).findLast((entry) => entry.pos !== ",");
145
+ const antecedent = tagged[lastIndexBefore(tagged, lead, (entry) => entry.pos !== ",")];
133
146
  return antecedent !== undefined && NOMINAL_TAG.has(antecedent.pos);
134
147
  };
135
148
 
@@ -230,23 +243,26 @@ const withEmphasis = (token: Token): Token =>
230
243
  * wink は位置を返さないので、表層を順に照合して復元する。
231
244
  * 見つからないものは飛ばし、カーソルは進めない。位置の当てずっぽうを下流に流さない。
232
245
  */
233
- const locate = (text: string, tagged: readonly Tagged[]): Token[] =>
234
- tagged.reduce<{ tokens: Token[]; cursor: number }>(
235
- (acc, entry, at) => {
236
- const start = text.indexOf(entry.value, acc.cursor);
237
- if (start === -1) return acc;
238
- const end = start + entry.value.length;
239
- const token = {
240
- span: { start, end },
241
- surface: entry.value,
242
- pos: properNounChecked(entry.value, upos(entry.pos)),
243
- ...(entry.lemma === undefined ? {} : { lemma: entry.lemma }),
244
- ...featuresOf(tagged, at),
245
- };
246
- return { tokens: [...acc.tokens, withEmphasis(token)], cursor: end };
247
- },
248
- { tokens: [], cursor: 0 },
249
- ).tokens;
246
+ const locate = (text: string, tagged: readonly Tagged[]): Token[] => {
247
+ // 語を足すたびに並びを作り直すと、長い文で語数の二乗になる。一つの並びに足していく。
248
+ const tokens: Token[] = [];
249
+ let cursor = 0;
250
+ tagged.forEach((entry, at) => {
251
+ const start = text.indexOf(entry.value, cursor);
252
+ if (start === -1) return;
253
+ const end = start + entry.value.length;
254
+ const token = {
255
+ span: { start, end },
256
+ surface: entry.value,
257
+ pos: properNounChecked(entry.value, upos(entry.pos)),
258
+ ...(entry.lemma === undefined ? {} : { lemma: entry.lemma }),
259
+ ...featuresOf(tagged, at),
260
+ };
261
+ tokens.push(withEmphasis(token));
262
+ cursor = end;
263
+ });
264
+ return tokens;
265
+ };
250
266
 
251
267
  /** 英語の語はこれより長くならない。超える並びは語として読まない。 */
252
268
  const RUN_LIMIT = 1000;
package/src/regexp.ts ADDED
@@ -0,0 +1,2 @@
1
+ /** A word from a lexicon, matched as written: a sign in it is that character, not a pattern. */
2
+ export const escapeRegExp = (text: string): string => text.replace(/[.*+?^${}()|[\]\\]/gu, String.raw`\$&`);
package/src/structure.ts CHANGED
@@ -1,5 +1,7 @@
1
1
  import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
2
- import { citedDocumentAfter, citedDocumentBefore, hyphenatedTagAround } from "./citation.ts";
2
+ import { citedDocumentAfter, citedDocumentBefore, listedTagAround } from "./citation.ts";
3
+ import { citedCodeBefore, codeVocabulary, titledCodeAt } from "./code-citation.ts";
4
+ import { loadLexicons } from "./lexicons.ts";
3
5
  import { membersAfter } from "./reference-list.ts";
4
6
  import { parseRoman } from "./roman.ts";
5
7
  import { dates } from "./dates.ts";
@@ -140,16 +142,19 @@ const glossedDocument = (gloss: Gloss): string | undefined => {
140
142
  return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
141
143
  };
142
144
 
145
+ const LEXICONS = loadLexicons();
146
+ const CODES = codeVocabulary(LEXICONS);
147
+
143
148
  /** The other document a reference names, or else a bracketed tag that the core checks against the document's list. */
144
149
  const citation = (text: string, start: number, end: number, document: string | undefined): Readonly<Record<string, string>> => {
145
150
  if (document !== undefined) return { document };
146
- const citedTag = hyphenatedTagAround(text, start, end);
151
+ const citedTag = listedTagAround(text, start, end);
147
152
  return citedTag === undefined ? {} : { citedTag };
148
153
  };
149
154
 
150
155
  /**
151
156
  * "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
152
- * "Section 9 of the Master Agreement" carries the other document's name, and is not looked up in this tree.
157
+ * "Section 9 of the Master Agreement" and "35 CFR §122" carry the other document's name, and are not looked up in this tree.
153
158
  */
154
159
  const references = (text: string): Mention[] => {
155
160
  const gloss: Gloss = { depth: 0, scanned: 0, anchors: [] };
@@ -158,7 +163,7 @@ const references = (text: string): Mention[] => {
158
163
  if (main === undefined) return [];
159
164
  const { parts, end } = subdivisions(text, match.index + match[0].length);
160
165
  advance(gloss, text, match.index);
161
- const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index);
166
+ const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index) ?? citedCodeBefore(text, match.index, CODES);
162
167
  const document = cited ?? glossedDocument(gloss);
163
168
  gloss.scanned = Math.max(gloss.scanned, end);
164
169
  if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
@@ -278,8 +283,19 @@ const quantities = (text: string): Mention[] =>
278
283
  return unit === undefined || Number.isNaN(value) || isWordChar(text[match.index - 1]) ? [] : [{ start: match.index, end, attrs: { value, unit } }];
279
284
  });
280
285
 
281
- /** "2.5 days" and "1.5 times" are amounts, not section 2.5 titled "days". */
282
- const countedAfter = (_number: string, rest: string): boolean => unitAfter(` ${rest}`, 0) !== undefined;
286
+ const MEASURE_UNITS = (LEXICONS["measure-unit"] ?? []).map((entry) => entry.pattern);
287
+
288
+ /** A letter, digit or hyphen right after the symbol makes it the start of a word: "2.1 mmap", "5.2.2.4 min-fresh". */
289
+ const CONTINUES_WORD = /^[\p{Script=Latin}\p{Nd}_-]/u;
290
+
291
+ const startsWithMeasureUnit = (rest: string): boolean => MEASURE_UNITS.some((unit) => rest.startsWith(unit) && !CONTINUES_WORD.test(rest.slice(unit.length)));
292
+
293
+ /**
294
+ * "2.5 days", "1.5 times" and "1.5 mM in each" are amounts, not section 2.5 titled "days". "40 CFR § 163.25" is title 40 of
295
+ * another code, not section 40 of this document.
296
+ */
297
+ const countedAfter = (_number: string, rest: string): boolean =>
298
+ unitAfter(` ${rest}`, 0) !== undefined || startsWithMeasureUnit(rest) || titledCodeAt(rest, CODES);
283
299
 
284
300
  export const structure: StructurePatterns = {
285
301
  numbered,