@chaffjs/lang-en 0.14.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/citation.d.ts +5 -2
- package/dist/citation.d.ts.map +1 -1
- package/dist/citation.js +13 -9
- package/dist/citation.js.map +1 -1
- package/dist/code-citation.d.ts +15 -0
- package/dist/code-citation.d.ts.map +1 -0
- package/dist/code-citation.js +34 -0
- package/dist/code-citation.js.map +1 -0
- package/dist/dates.d.ts.map +1 -1
- package/dist/dates.js +22 -4
- package/dist/dates.js.map +1 -1
- package/dist/emphasis.d.ts +7 -0
- package/dist/emphasis.d.ts.map +1 -0
- package/dist/emphasis.js +15 -0
- package/dist/emphasis.js.map +1 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -4
- package/dist/index.js.map +1 -1
- package/dist/japanese-run.d.ts +2 -0
- package/dist/japanese-run.d.ts.map +1 -0
- package/dist/japanese-run.js +9 -0
- package/dist/japanese-run.js.map +1 -0
- package/dist/label-stop.d.ts +13 -0
- package/dist/label-stop.d.ts.map +1 -0
- package/dist/label-stop.js +13 -0
- package/dist/label-stop.js.map +1 -0
- package/dist/pos.d.ts.map +1 -1
- package/dist/pos.js +61 -18
- package/dist/pos.js.map +1 -1
- package/dist/regexp.d.ts +3 -0
- package/dist/regexp.d.ts.map +1 -0
- package/dist/regexp.js +3 -0
- package/dist/regexp.js.map +1 -0
- package/dist/structure.d.ts.map +1 -1
- package/dist/structure.js +17 -6
- package/dist/structure.js.map +1 -1
- package/lexicons/abbreviated-label.yaml +16 -0
- package/lexicons/ai-tell.yaml +86 -0
- package/lexicons/announcing-opener.yaml +19 -0
- package/lexicons/assistant-residue.yaml +55 -0
- package/lexicons/contrast-frame.yaml +10 -0
- package/lexicons/contrast-lead.yaml +15 -0
- package/lexicons/contrast-turn.yaml +11 -0
- package/lexicons/count-adjective.yaml +8 -0
- package/lexicons/count-anchor.yaml +7 -0
- package/lexicons/count-counter.yaml +40 -0
- package/lexicons/count-hedge.yaml +67 -0
- package/lexicons/count-number.yaml +16 -0
- package/lexicons/dependent-possessive.yaml +10 -0
- package/lexicons/document-kind.yaml +19 -0
- package/lexicons/email-attachment-note.yaml +7 -0
- package/lexicons/email-attribution.yaml +7 -0
- package/lexicons/email-header-field.yaml +28 -0
- package/lexicons/email-written-field.yaml +8 -0
- package/lexicons/emphasis-word.yaml +4 -2
- package/lexicons/example-marker.yaml +12 -0
- package/lexicons/figure-elsewhere.yaml +8 -0
- package/lexicons/figure-label.yaml +14 -0
- package/lexicons/finite-auxiliary.yaml +9 -0
- package/lexicons/invariant-noun.yaml +26 -0
- package/lexicons/measure-unit.yaml +78 -0
- package/lexicons/misnumbered-phrase.yaml +7 -0
- package/lexicons/name-title.yaml +20 -0
- package/lexicons/numbered-label.yaml +26 -0
- package/lexicons/pair-opener.yaml +13 -0
- package/lexicons/percent-unit.yaml +7 -0
- package/lexicons/placeholder-word.yaml +17 -0
- package/lexicons/plural-determiner.yaml +10 -0
- package/lexicons/range-connector.yaml +12 -0
- package/lexicons/share-exception.yaml +10 -0
- package/lexicons/share-label.yaml +12 -0
- package/lexicons/singular-determiner.yaml +13 -0
- package/lexicons/stock-transition.yaml +17 -0
- package/package.json +1 -1
- package/src/citation.ts +13 -9
- package/src/code-citation.ts +52 -0
- package/src/dates.ts +22 -4
- package/src/emphasis.ts +16 -0
- package/src/index.ts +14 -5
- package/src/japanese-run.ts +9 -0
- package/src/label-stop.ts +25 -0
- package/src/pos.ts +63 -21
- package/src/regexp.ts +2 -0
- package/src/structure.ts +22 -6
package/src/pos.ts
CHANGED
|
@@ -5,6 +5,7 @@ import { loadLexicons } from "./lexicons.ts";
|
|
|
5
5
|
import { blankLongRuns } from "./long-runs.ts";
|
|
6
6
|
import { lowercasedAt, properNounChecked, rereadAt, sentenceInitialCommonWord } from "./proper-noun.ts";
|
|
7
7
|
import { isStativeParticiple, stativeVocabulary } from "./stative-participle.ts";
|
|
8
|
+
import { isEmphasisedAdverb } from "./emphasis.ts";
|
|
8
9
|
|
|
9
10
|
const require = createRequire(import.meta.url);
|
|
10
11
|
|
|
@@ -99,9 +100,22 @@ const BE = new Set(["be", "am", "is", "are", "was", "were", "been", "being"]);
|
|
|
99
100
|
|
|
100
101
|
const isBe = (entry: Tagged): boolean => BE.has(entry.lemma ?? entry.value.toLowerCase()) || BE.has(entry.value.toLowerCase());
|
|
101
102
|
|
|
103
|
+
/**
|
|
104
|
+
* end(0 以上)より前で test に合う最後の位置。無ければ -1。過去分詞の多い長い文で、分詞ごとに文の頭から写すと語数の二乗になるので、後ろから探す。
|
|
105
|
+
*/
|
|
106
|
+
const lastIndexBefore = (tagged: readonly Tagged[], end: number, test: (entry: Tagged) => boolean): number => {
|
|
107
|
+
let at = Math.min(end, tagged.length);
|
|
108
|
+
while (at > 0) {
|
|
109
|
+
at -= 1;
|
|
110
|
+
const entry = tagged[at];
|
|
111
|
+
if (entry !== undefined && test(entry)) return at;
|
|
112
|
+
}
|
|
113
|
+
return -1;
|
|
114
|
+
};
|
|
115
|
+
|
|
102
116
|
/** 過去分詞の前の be の位置。無ければ -1。 */
|
|
103
117
|
const beBefore = (tagged: readonly Tagged[], at: number): number => {
|
|
104
|
-
const head = tagged
|
|
118
|
+
const head = lastIndexBefore(tagged, at, (entry) => !SKIPPABLE.has(entry.pos));
|
|
105
119
|
const entry = tagged[head];
|
|
106
120
|
return entry !== undefined && isBe(entry) ? head : -1;
|
|
107
121
|
};
|
|
@@ -124,11 +138,11 @@ const NOMINAL_TAG = new Set(["NN", "NNS", "NNP", "NNPS", "PRP", "CD", "DT"]);
|
|
|
124
138
|
const RELATIVE_TAG = new Set(["WDT", "WP"]);
|
|
125
139
|
|
|
126
140
|
const inRelativeClause = (tagged: readonly Tagged[], be: number): boolean => {
|
|
127
|
-
const lead = tagged
|
|
141
|
+
const lead = lastIndexBefore(tagged, be, (entry) => !isAuxiliary(entry));
|
|
128
142
|
const relative = tagged[lead];
|
|
129
143
|
if (relative === undefined || !RELATIVE_TAG.has(relative.pos)) return false;
|
|
130
144
|
// 文頭の That was decided. / Which was chosen? は、前に指す名詞が無いので述語。
|
|
131
|
-
const antecedent = tagged
|
|
145
|
+
const antecedent = tagged[lastIndexBefore(tagged, lead, (entry) => entry.pos !== ",")];
|
|
132
146
|
return antecedent !== undefined && NOMINAL_TAG.has(antecedent.pos);
|
|
133
147
|
};
|
|
134
148
|
|
|
@@ -164,6 +178,27 @@ const PLURAL_TAG = new Set(["NNS", "NNPS"]);
|
|
|
164
178
|
|
|
165
179
|
const nounOrDeterminerFeatures = (entry: Tagged): Features => (PLURAL_TAG.has(entry.pos) ? { features: { Number: "Plur" } } : determinerFeatures(entry));
|
|
166
180
|
|
|
181
|
+
const COMMON_NOUN_TAG = new Set(["NN", "NNS"]);
|
|
182
|
+
const NOUN_TAG = new Set(["NN", "NNS", "NNP", "NNPS"]);
|
|
183
|
+
const ADJECTIVE_TAG = new Set(["JJ", "JJR", "JJS"]);
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* 解析器が一つに決めた読みの、ほかの読み。名詞と付けた語が動詞にもなる(works / report)なら AlsoVerb=Yes、
|
|
187
|
+
* 形容詞と付けた語が名詞にもなる(individual / key)なら AlsoNoun=Yes。an individual works は名詞と動詞とも読める。
|
|
188
|
+
* 語彙に無い名詞・形容詞(tribunal / stimuli)は、品詞も単数・複数も解析器が形から当てたものなので Guess=Yes。
|
|
189
|
+
*/
|
|
190
|
+
const otherReading = (entry: Tagged): Readonly<Record<string, string>> => {
|
|
191
|
+
const tags = state.vocabulary(entry.value.toLowerCase());
|
|
192
|
+
if (tags === undefined) return NOUN_TAG.has(entry.pos) || ADJECTIVE_TAG.has(entry.pos) ? { Guess: "Yes" } : {};
|
|
193
|
+
if (COMMON_NOUN_TAG.has(entry.pos) && tags.some((tag) => tag.startsWith("VB"))) return { AlsoVerb: "Yes" };
|
|
194
|
+
return ADJECTIVE_TAG.has(entry.pos) && tags.some((tag) => COMMON_NOUN_TAG.has(tag)) ? { AlsoNoun: "Yes" } : {};
|
|
195
|
+
};
|
|
196
|
+
|
|
197
|
+
const withOtherReading = (entry: Tagged, found: Features): Features => {
|
|
198
|
+
const other = otherReading(entry);
|
|
199
|
+
return Object.keys(other).length === 0 ? found : { features: { ...found.features, ...other } };
|
|
200
|
+
};
|
|
201
|
+
|
|
167
202
|
/**
|
|
168
203
|
* 過去分詞は VerbForm=Part。Based on the review, のような分詞の導入句を、命令形の並び(fix the parser, ship it)と見分ける。
|
|
169
204
|
* -ing 形は VerbForm=Ger。解析器は動名詞と現在分詞を分けないので、過去分詞を見る判断(Part)には混ぜない。
|
|
@@ -172,7 +207,7 @@ const featuresOf = (tagged: readonly Tagged[], at: number): Features => {
|
|
|
172
207
|
const entry = tagged[at];
|
|
173
208
|
if (entry === undefined) return {};
|
|
174
209
|
if (entry.pos === "VBG") return { features: { VerbForm: "Ger" } };
|
|
175
|
-
if (entry.pos !== "VBN") return nounOrDeterminerFeatures(entry);
|
|
210
|
+
if (entry.pos !== "VBN") return withOtherReading(entry, nounOrDeterminerFeatures(entry));
|
|
176
211
|
return { features: isPassive(tagged, at) ? { VerbForm: "Part", Voice: "Pass" } : { VerbForm: "Part" } };
|
|
177
212
|
};
|
|
178
213
|
|
|
@@ -200,27 +235,34 @@ export const prepare = (): void => {
|
|
|
200
235
|
|
|
201
236
|
export const isReady = (): boolean => state.ready !== undefined;
|
|
202
237
|
|
|
238
|
+
/** 大文字で強調した副詞(NEVER)は、解析器が名前と付けても副詞。Emph=Yes は、略語ではないと detector に伝える。 */
|
|
239
|
+
const withEmphasis = (token: Token): Token =>
|
|
240
|
+
isEmphasisedAdverb(token.surface, state.vocabulary) ? { ...token, pos: "ADV", lemma: token.surface.toLowerCase(), features: { Emph: "Yes" } } : token;
|
|
241
|
+
|
|
203
242
|
/**
|
|
204
243
|
* wink は位置を返さないので、表層を順に照合して復元する。
|
|
205
244
|
* 見つからないものは飛ばし、カーソルは進めない。位置の当てずっぽうを下流に流さない。
|
|
206
245
|
*/
|
|
207
|
-
const locate = (text: string, tagged: readonly Tagged[]): Token[] =>
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
}
|
|
220
|
-
|
|
221
|
-
}
|
|
222
|
-
|
|
223
|
-
|
|
246
|
+
const locate = (text: string, tagged: readonly Tagged[]): Token[] => {
|
|
247
|
+
// 語を足すたびに並びを作り直すと、長い文で語数の二乗になる。一つの並びに足していく。
|
|
248
|
+
const tokens: Token[] = [];
|
|
249
|
+
let cursor = 0;
|
|
250
|
+
tagged.forEach((entry, at) => {
|
|
251
|
+
const start = text.indexOf(entry.value, cursor);
|
|
252
|
+
if (start === -1) return;
|
|
253
|
+
const end = start + entry.value.length;
|
|
254
|
+
const token = {
|
|
255
|
+
span: { start, end },
|
|
256
|
+
surface: entry.value,
|
|
257
|
+
pos: properNounChecked(entry.value, upos(entry.pos)),
|
|
258
|
+
...(entry.lemma === undefined ? {} : { lemma: entry.lemma }),
|
|
259
|
+
...featuresOf(tagged, at),
|
|
260
|
+
};
|
|
261
|
+
tokens.push(withEmphasis(token));
|
|
262
|
+
cursor = end;
|
|
263
|
+
});
|
|
264
|
+
return tokens;
|
|
265
|
+
};
|
|
224
266
|
|
|
225
267
|
/** 英語の語はこれより長くならない。超える並びは語として読まない。 */
|
|
226
268
|
const RUN_LIMIT = 1000;
|
package/src/regexp.ts
ADDED
package/src/structure.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import type { Mention, NumberedLine, NumberingContext, StructurePatterns } from "chaffjs/plugin";
|
|
2
|
-
import { citedDocumentAfter, citedDocumentBefore,
|
|
2
|
+
import { citedDocumentAfter, citedDocumentBefore, listedTagAround } from "./citation.ts";
|
|
3
|
+
import { citedCodeBefore, codeVocabulary, titledCodeAt } from "./code-citation.ts";
|
|
4
|
+
import { loadLexicons } from "./lexicons.ts";
|
|
3
5
|
import { membersAfter } from "./reference-list.ts";
|
|
4
6
|
import { parseRoman } from "./roman.ts";
|
|
5
7
|
import { dates } from "./dates.ts";
|
|
@@ -140,16 +142,19 @@ const glossedDocument = (gloss: Gloss): string | undefined => {
|
|
|
140
142
|
return anchor !== undefined && gloss.depth > anchor.depth ? anchor.document : undefined;
|
|
141
143
|
};
|
|
142
144
|
|
|
145
|
+
const LEXICONS = loadLexicons();
|
|
146
|
+
const CODES = codeVocabulary(LEXICONS);
|
|
147
|
+
|
|
143
148
|
/** The other document a reference names, or else a bracketed tag that the core checks against the document's list. */
|
|
144
149
|
const citation = (text: string, start: number, end: number, document: string | undefined): Readonly<Record<string, string>> => {
|
|
145
150
|
if (document !== undefined) return { document };
|
|
146
|
-
const citedTag =
|
|
151
|
+
const citedTag = listedTagAround(text, start, end);
|
|
147
152
|
return citedTag === undefined ? {} : { citedTag };
|
|
148
153
|
};
|
|
149
154
|
|
|
150
155
|
/**
|
|
151
156
|
* "Section 4.2(a)" → 4.2.a, "Article III" → 3. The same addresses the tree gives.
|
|
152
|
-
* "Section 9 of the Master Agreement"
|
|
157
|
+
* "Section 9 of the Master Agreement" and "35 CFR §122" carry the other document's name, and are not looked up in this tree.
|
|
153
158
|
*/
|
|
154
159
|
const references = (text: string): Mention[] => {
|
|
155
160
|
const gloss: Gloss = { depth: 0, scanned: 0, anchors: [] };
|
|
@@ -158,7 +163,7 @@ const references = (text: string): Mention[] => {
|
|
|
158
163
|
if (main === undefined) return [];
|
|
159
164
|
const { parts, end } = subdivisions(text, match.index + match[0].length);
|
|
160
165
|
advance(gloss, text, match.index);
|
|
161
|
-
const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index);
|
|
166
|
+
const cited = citedDocumentAfter(text, end) ?? citedDocumentBefore(text, match.index) ?? citedCodeBefore(text, match.index, CODES);
|
|
162
167
|
const document = cited ?? glossedDocument(gloss);
|
|
163
168
|
gloss.scanned = Math.max(gloss.scanned, end);
|
|
164
169
|
if (cited !== undefined) gloss.anchors.push({ document: cited, depth: gloss.depth });
|
|
@@ -278,8 +283,19 @@ const quantities = (text: string): Mention[] =>
|
|
|
278
283
|
return unit === undefined || Number.isNaN(value) || isWordChar(text[match.index - 1]) ? [] : [{ start: match.index, end, attrs: { value, unit } }];
|
|
279
284
|
});
|
|
280
285
|
|
|
281
|
-
|
|
282
|
-
|
|
286
|
+
const MEASURE_UNITS = (LEXICONS["measure-unit"] ?? []).map((entry) => entry.pattern);
|
|
287
|
+
|
|
288
|
+
/** A letter, digit or hyphen right after the symbol makes it the start of a word: "2.1 mmap", "5.2.2.4 min-fresh". */
|
|
289
|
+
const CONTINUES_WORD = /^[\p{Script=Latin}\p{Nd}_-]/u;
|
|
290
|
+
|
|
291
|
+
const startsWithMeasureUnit = (rest: string): boolean => MEASURE_UNITS.some((unit) => rest.startsWith(unit) && !CONTINUES_WORD.test(rest.slice(unit.length)));
|
|
292
|
+
|
|
293
|
+
/**
|
|
294
|
+
* "2.5 days", "1.5 times" and "1.5 mM in each" are amounts, not section 2.5 titled "days". "40 CFR § 163.25" is title 40 of
|
|
295
|
+
* another code, not section 40 of this document.
|
|
296
|
+
*/
|
|
297
|
+
const countedAfter = (_number: string, rest: string): boolean =>
|
|
298
|
+
unitAfter(` ${rest}`, 0) !== undefined || startsWithMeasureUnit(rest) || titledCodeAt(rest, CODES);
|
|
283
299
|
|
|
284
300
|
export const structure: StructurePatterns = {
|
|
285
301
|
numbered,
|