@chaffjs/lang-en 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/dist/citation.d.ts +5 -2
  2. package/dist/citation.d.ts.map +1 -1
  3. package/dist/citation.js +13 -9
  4. package/dist/citation.js.map +1 -1
  5. package/dist/code-citation.d.ts +15 -0
  6. package/dist/code-citation.d.ts.map +1 -0
  7. package/dist/code-citation.js +34 -0
  8. package/dist/code-citation.js.map +1 -0
  9. package/dist/dates.d.ts.map +1 -1
  10. package/dist/dates.js +22 -4
  11. package/dist/dates.js.map +1 -1
  12. package/dist/emphasis.d.ts +7 -0
  13. package/dist/emphasis.d.ts.map +1 -0
  14. package/dist/emphasis.js +15 -0
  15. package/dist/emphasis.js.map +1 -0
  16. package/dist/index.d.ts +1 -1
  17. package/dist/index.d.ts.map +1 -1
  18. package/dist/index.js +10 -4
  19. package/dist/index.js.map +1 -1
  20. package/dist/japanese-run.d.ts +2 -0
  21. package/dist/japanese-run.d.ts.map +1 -0
  22. package/dist/japanese-run.js +9 -0
  23. package/dist/japanese-run.js.map +1 -0
  24. package/dist/label-stop.d.ts +13 -0
  25. package/dist/label-stop.d.ts.map +1 -0
  26. package/dist/label-stop.js +13 -0
  27. package/dist/label-stop.js.map +1 -0
  28. package/dist/pos.d.ts.map +1 -1
  29. package/dist/pos.js +61 -18
  30. package/dist/pos.js.map +1 -1
  31. package/dist/regexp.d.ts +3 -0
  32. package/dist/regexp.d.ts.map +1 -0
  33. package/dist/regexp.js +3 -0
  34. package/dist/regexp.js.map +1 -0
  35. package/dist/structure.d.ts.map +1 -1
  36. package/dist/structure.js +17 -6
  37. package/dist/structure.js.map +1 -1
  38. package/lexicons/abbreviated-label.yaml +16 -0
  39. package/lexicons/ai-tell.yaml +86 -0
  40. package/lexicons/announcing-opener.yaml +19 -0
  41. package/lexicons/assistant-residue.yaml +55 -0
  42. package/lexicons/contrast-frame.yaml +10 -0
  43. package/lexicons/contrast-lead.yaml +15 -0
  44. package/lexicons/contrast-turn.yaml +11 -0
  45. package/lexicons/count-adjective.yaml +8 -0
  46. package/lexicons/count-anchor.yaml +7 -0
  47. package/lexicons/count-counter.yaml +40 -0
  48. package/lexicons/count-hedge.yaml +67 -0
  49. package/lexicons/count-number.yaml +16 -0
  50. package/lexicons/dependent-possessive.yaml +10 -0
  51. package/lexicons/document-kind.yaml +19 -0
  52. package/lexicons/email-attachment-note.yaml +7 -0
  53. package/lexicons/email-attribution.yaml +7 -0
  54. package/lexicons/email-header-field.yaml +28 -0
  55. package/lexicons/email-written-field.yaml +8 -0
  56. package/lexicons/emphasis-word.yaml +4 -2
  57. package/lexicons/example-marker.yaml +12 -0
  58. package/lexicons/figure-elsewhere.yaml +8 -0
  59. package/lexicons/figure-label.yaml +14 -0
  60. package/lexicons/finite-auxiliary.yaml +9 -0
  61. package/lexicons/invariant-noun.yaml +26 -0
  62. package/lexicons/measure-unit.yaml +78 -0
  63. package/lexicons/misnumbered-phrase.yaml +7 -0
  64. package/lexicons/name-title.yaml +20 -0
  65. package/lexicons/numbered-label.yaml +26 -0
  66. package/lexicons/pair-opener.yaml +13 -0
  67. package/lexicons/percent-unit.yaml +7 -0
  68. package/lexicons/placeholder-word.yaml +17 -0
  69. package/lexicons/plural-determiner.yaml +10 -0
  70. package/lexicons/range-connector.yaml +12 -0
  71. package/lexicons/share-exception.yaml +10 -0
  72. package/lexicons/share-label.yaml +12 -0
  73. package/lexicons/singular-determiner.yaml +13 -0
  74. package/lexicons/stock-transition.yaml +17 -0
  75. package/package.json +1 -1
  76. package/src/citation.ts +13 -9
  77. package/src/code-citation.ts +52 -0
  78. package/src/dates.ts +22 -4
  79. package/src/emphasis.ts +16 -0
  80. package/src/index.ts +14 -5
  81. package/src/japanese-run.ts +9 -0
  82. package/src/label-stop.ts +25 -0
  83. package/src/pos.ts +63 -21
  84. package/src/regexp.ts +2 -0
  85. package/src/structure.ts +22 -6
package/src/pos.ts CHANGED
@@ -5,6 +5,7 @@ import { loadLexicons } from "./lexicons.ts";
5
5
  import { blankLongRuns } from "./long-runs.ts";
6
6
  import { lowercasedAt, properNounChecked, rereadAt, sentenceInitialCommonWord } from "./proper-noun.ts";
7
7
  import { isStativeParticiple, stativeVocabulary } from "./stative-participle.ts";
8
+ import { isEmphasisedAdverb } from "./emphasis.ts";
8
9
 
9
10
  const require = createRequire(import.meta.url);
10
11
 
@@ -99,9 +100,22 @@ const BE = new Set(["be", "am", "is", "are", "was", "were", "been", "being"]);
99
100
 
100
101
  const isBe = (entry: Tagged): boolean => BE.has(entry.lemma ?? entry.value.toLowerCase()) || BE.has(entry.value.toLowerCase());
101
102
 
103
+ /**
104
+ * end(0 以上)より前で test に合う最後の位置。無ければ -1。過去分詞の多い長い文で、分詞ごとに文の頭から写すと語数の二乗になるので、後ろから探す。
105
+ */
106
+ const lastIndexBefore = (tagged: readonly Tagged[], end: number, test: (entry: Tagged) => boolean): number => {
107
+ let at = Math.min(end, tagged.length);
108
+ while (at > 0) {
109
+ at -= 1;
110
+ const entry = tagged[at];
111
+ if (entry !== undefined && test(entry)) return at;
112
+ }
113
+ return -1;
114
+ };
115
+
102
116
  /** 過去分詞の前の be の位置。無ければ -1。 */
103
117
  const beBefore = (tagged: readonly Tagged[], at: number): number => {
104
- const head = tagged.slice(0, at).findLastIndex((entry) => !SKIPPABLE.has(entry.pos));
118
+ const head = lastIndexBefore(tagged, at, (entry) => !SKIPPABLE.has(entry.pos));
105
119
  const entry = tagged[head];
106
120
  return entry !== undefined && isBe(entry) ? head : -1;
107
121
  };
@@ -124,11 +138,11 @@ const NOMINAL_TAG = new Set(["NN", "NNS", "NNP", "NNPS", "PRP", "CD", "DT"]);
124
138
  const RELATIVE_TAG = new Set(["WDT", "WP"]);
125
139
 
126
140
  const inRelativeClause = (tagged: readonly Tagged[], be: number): boolean => {
127
- const lead = tagged.slice(0, be).findLastIndex((entry) => !isAuxiliary(entry));
141
+ const lead = lastIndexBefore(tagged, be, (entry) => !isAuxiliary(entry));
128
142
  const relative = tagged[lead];
129
143
  if (relative === undefined || !RELATIVE_TAG.has(relative.pos)) return false;
130
144
  // 文頭の That was decided. / Which was chosen? は、前に指す名詞が無いので述語。
131
- const antecedent = tagged.slice(0, lead).findLast((entry) => entry.pos !== ",");
145
+ const antecedent = tagged[lastIndexBefore(tagged, lead, (entry) => entry.pos !== ",")];
132
146
  return antecedent !== undefined && NOMINAL_TAG.has(antecedent.pos);
133
147
  };
134
148
 
@@ -164,6 +178,27 @@ const PLURAL_TAG = new Set(["NNS", "NNPS"]);
164
178
 
165
179
  const nounOrDeterminerFeatures = (entry: Tagged): Features => (PLURAL_TAG.has(entry.pos) ? { features: { Number: "Plur" } } : determinerFeatures(entry));
166
180
 
181
+ const COMMON_NOUN_TAG = new Set(["NN", "NNS"]);
182
+ const NOUN_TAG = new Set(["NN", "NNS", "NNP", "NNPS"]);
183
+ const ADJECTIVE_TAG = new Set(["JJ", "JJR", "JJS"]);
184
+
185
+ /**
186
+ * 解析器が一つに決めた読みの、ほかの読み。名詞と付けた語が動詞にもなる(works / report)なら AlsoVerb=Yes、
187
+ * 形容詞と付けた語が名詞にもなる(individual / key)なら AlsoNoun=Yes。an individual works は名詞と動詞とも読める。
188
+ * 語彙に無い名詞・形容詞(tribunal / stimuli)は、品詞も単数・複数も解析器が形から当てたものなので Guess=Yes。
189
+ */
190
+ const otherReading = (entry: Tagged): Readonly<Record<string, string>> => {
191
+ const tags = state.vocabulary(entry.value.toLowerCase());
192
+ if (tags === undefined) return NOUN_TAG.has(entry.pos) || ADJECTIVE_TAG.has(entry.pos) ? { Guess: "Yes" } : {};
193
+ if (COMMON_NOUN_TAG.has(entry.pos) && tags.some((tag) => tag.startsWith("VB"))) return { AlsoVerb: "Yes" };
194
+ return ADJECTIVE_TAG.has(entry.pos) && tags.some((tag) => COMMON_NOUN_TAG.has(tag)) ? { AlsoNoun: "Yes" } : {};
195
+ };
196
+
197
+ const withOtherReading = (entry: Tagged, found: Features): Features => {
198
+ const other = otherReading(entry);
199
+ return Object.keys(other).length === 0 ? found : { features: { ...found.features, ...other } };
200
+ };
201
+
167
202
  /**
168
203
  * 過去分詞は VerbForm=Part。Based on the review, のような分詞の導入句を、命令形の並び(fix the parser, ship it)と見分ける。
169
204
  * -ing 形は VerbForm=Ger。解析器は動名詞と現在分詞を分けないので、過去分詞を見る判断(Part)には混ぜない。
@@ -172,7 +207,7 @@ const featuresOf = (tagged: readonly Tagged[], at: number): Features => {
172
207
  const entry = tagged[at];
173
208
  if (entry === undefined) return {};
174
209
  if (entry.pos === "VBG") return { features: { VerbForm: "Ger" } };
175
- if (entry.pos !== "VBN") return nounOrDeterminerFeatures(entry);
210
+ if (entry.pos !== "VBN") return withOtherReading(entry, nounOrDeterminerFeatures(entry));
176
211
  return { features: isPassive(tagged, at) ? { VerbForm: "Part", Voice: "Pass" } : { VerbForm: "Part" } };
177
212
  };
178
213
 
@@ -200,27 +235,34 @@ export const prepare = (): void => {
200
235
 
201
236
  export const isReady = (): boolean => state.ready !== undefined;
202
237
 
238
+ /** 大文字で強調した副詞(NEVER)は、解析器が名前と付けても副詞。Emph=Yes は、略語ではないと detector に伝える。 */
239
+ const withEmphasis = (token: Token): Token =>
240
+ isEmphasisedAdverb(token.surface, state.vocabulary) ? { ...token, pos: "ADV", lemma: token.surface.toLowerCase(), features: { Emph: "Yes" } } : token;
241
+
203
242
  /**
204
243
  * wink は位置を返さないので、表層を順に照合して復元する。
205
244
  * 見つからないものは飛ばし、カーソルは進めない。位置の当てずっぽうを下流に流さない。
206
245
  */
207
- const locate = (text: string, tagged: readonly Tagged[]): Token[] =>
208
- tagged.reduce<{ tokens: Token[]; cursor: number }>(
209
- (acc, entry, at) => {
210
- const start = text.indexOf(entry.value, acc.cursor);
211
- if (start === -1) return acc;
212
- const end = start + entry.value.length;
213
- const token = {
214
- span: { start, end },
215
- surface: entry.value,
216
- pos: properNounChecked(entry.value, upos(entry.pos)),
217
- ...(entry.lemma === undefined ? {} : { lemma: entry.lemma }),
218
- ...featuresOf(tagged, at),
219
- };
220
- return { tokens: [...acc.tokens, token], cursor: end };
221
- },
222
- { tokens: [], cursor: 0 },
223
- ).tokens;
246
+ const locate = (text: string, tagged: readonly Tagged[]): Token[] => {
247
+ // 語を足すたびに並びを作り直すと、長い文で語数の二乗になる。一つの並びに足していく。
248
+ const tokens: Token[] = [];
249
+ let cursor = 0;
250
+ tagged.forEach((entry, at) => {
251
+ const start = text.indexOf(entry.value, cursor);
252
+ if (start === -1) return;
253
+ const end = start + entry.value.length;
254
+ const token = {
255
+ span: { start, end },
256
+ surface: entry.value,
257
+ pos: properNounChecked(entry.value, upos(entry.pos)),
258
+ ...(entry.lemma === undefined ? {} : { lemma: entry.lemma }),
259
+ ...featuresOf(tagged, at),
260
+ };
261
+ tokens.push(withEmphasis(token));
262
+ cursor = end;
263
+ });
264
+ return tokens;
265
+ };
224
266
 
225
267
  /** 英語の語はこれより長くならない。超える並びは語として読まない。 */
226
268
  const RUN_LIMIT = 1000;
package/src/regexp.ts ADDED
@@ -0,0 +1,2 @@
1
+ /** A word from a lexicon, matched as written: a sign in it is that character, not a pattern. */
2
+ export const escapeRegExp = (text: string): string => text.replace(/[.*+?^${}()|[\]\\]/gu, String.raw`\$&`);
package/src/structure.ts CHANGED
@@ -1,5 +1,7 @@
1
1
  import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
2
- import { citedDocumentAfter, citedDocumentBefore, hyphenatedTagAround } from "./citation.ts";
2
+ import { citedDocumentAfter, citedDocumentBefore, listedTagAround } from "./citation.ts";
3
+ import { citedCodeBefore, codeVocabulary, titledCodeAt } from "./code-citation.ts";
4
+ import { loadLexicons } from "./lexicons.ts";
3
5
  import { membersAfter } from "./reference-list.ts";
4
6
  import { parseRoman } from "./roman.ts";
5
7
  import { dates } from "./dates.ts";
@@ -140,16 +142,19 @@ const glossedDocument = (gloss: Gloss): string | undefined => {
140
142
  return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
141
143
  };
142
144
 
145
+ const LEXICONS = loadLexicons();
146
+ const CODES = codeVocabulary(LEXICONS);
147
+
143
148
  /** The other document a reference names, or else a bracketed tag that the core checks against the document's list. */
144
149
  const citation = (text: string, start: number, end: number, document: string | undefined): Readonly<Record<string, string>> => {
145
150
  if (document !== undefined) return { document };
146
- const citedTag = hyphenatedTagAround(text, start, end);
151
+ const citedTag = listedTagAround(text, start, end);
147
152
  return citedTag === undefined ? {} : { citedTag };
148
153
  };
149
154
 
150
155
  /**
151
156
  * "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
152
- * "Section 9 of the Master Agreement" carries the other document's name, and is not looked up in this tree.
157
+ * "Section 9 of the Master Agreement" and "35 CFR §122" carry the other document's name, and are not looked up in this tree.
153
158
  */
154
159
  const references = (text: string): Mention[] => {
155
160
  const gloss: Gloss = { depth: 0, scanned: 0, anchors: [] };
@@ -158,7 +163,7 @@ const references = (text: string): Mention[] => {
158
163
  if (main === undefined) return [];
159
164
  const { parts, end } = subdivisions(text, match.index + match[0].length);
160
165
  advance(gloss, text, match.index);
161
- const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index);
166
+ const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index) ?? citedCodeBefore(text, match.index, CODES);
162
167
  const document = cited ?? glossedDocument(gloss);
163
168
  gloss.scanned = Math.max(gloss.scanned, end);
164
169
  if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
@@ -278,8 +283,19 @@ const quantities = (text: string): Mention[] =>
278
283
  return unit === undefined || Number.isNaN(value) || isWordChar(text[match.index - 1]) ? [] : [{ start: match.index, end, attrs: { value, unit } }];
279
284
  });
280
285
 
281
- /** "2.5 days" and "1.5 times" are amounts, not section 2.5 titled "days". */
282
- const countedAfter = (_number: string, rest: string): boolean => unitAfter(` ${rest}`, 0) !== undefined;
286
+ const MEASURE_UNITS = (LEXICONS["measure-unit"] ?? []).map((entry) => entry.pattern);
287
+
288
+ /** A letter, digit or hyphen right after the symbol makes it the start of a word: "2.1 mmap", "5.2.2.4 min-fresh". */
289
+ const CONTINUES_WORD = /^[\p{Script=Latin}\p{Nd}_-]/u;
290
+
291
+ const startsWithMeasureUnit = (rest: string): boolean => MEASURE_UNITS.some((unit) => rest.startsWith(unit) && !CONTINUES_WORD.test(rest.slice(unit.length)));
292
+
293
+ /**
294
+ * "2.5 days", "1.5 times" and "1.5 mM in each" are amounts, not section 2.5 titled "days". "40 CFR § 163.25" is title 40 of
295
+ * another code, not section 40 of this document.
296
+ */
297
+ const countedAfter = (_number: string, rest: string): boolean =>
298
+ unitAfter(` ${rest}`, 0) !== undefined || startsWithMeasureUnit(rest) || titledCodeAt(rest, CODES);
283
299
 
284
300
  export const structure: StructurePatterns = {
285
301
  numbered,