@chaffjs/lang-en 0.14.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/citation.d.ts +5 -2
- package/dist/citation.d.ts.map +1 -1
- package/dist/citation.js +13 -9
- package/dist/citation.js.map +1 -1
- package/dist/code-citation.d.ts +15 -0
- package/dist/code-citation.d.ts.map +1 -0
- package/dist/code-citation.js +34 -0
- package/dist/code-citation.js.map +1 -0
- package/dist/dates.d.ts.map +1 -1
- package/dist/dates.js +22 -4
- package/dist/dates.js.map +1 -1
- package/dist/emphasis.d.ts +7 -0
- package/dist/emphasis.d.ts.map +1 -0
- package/dist/emphasis.js +15 -0
- package/dist/emphasis.js.map +1 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -4
- package/dist/index.js.map +1 -1
- package/dist/japanese-run.d.ts +2 -0
- package/dist/japanese-run.d.ts.map +1 -0
- package/dist/japanese-run.js +9 -0
- package/dist/japanese-run.js.map +1 -0
- package/dist/label-stop.d.ts +13 -0
- package/dist/label-stop.d.ts.map +1 -0
- package/dist/label-stop.js +13 -0
- package/dist/label-stop.js.map +1 -0
- package/dist/pos.d.ts.map +1 -1
- package/dist/pos.js +61 -18
- package/dist/pos.js.map +1 -1
- package/dist/regexp.d.ts +3 -0
- package/dist/regexp.d.ts.map +1 -0
- package/dist/regexp.js +3 -0
- package/dist/regexp.js.map +1 -0
- package/dist/structure.d.ts.map +1 -1
- package/dist/structure.js +17 -6
- package/dist/structure.js.map +1 -1
- package/lexicons/abbreviated-label.yaml +16 -0
- package/lexicons/ai-tell.yaml +86 -0
- package/lexicons/announcing-opener.yaml +19 -0
- package/lexicons/assistant-residue.yaml +55 -0
- package/lexicons/contrast-frame.yaml +10 -0
- package/lexicons/contrast-lead.yaml +15 -0
- package/lexicons/contrast-turn.yaml +11 -0
- package/lexicons/count-adjective.yaml +8 -0
- package/lexicons/count-anchor.yaml +7 -0
- package/lexicons/count-counter.yaml +40 -0
- package/lexicons/count-hedge.yaml +67 -0
- package/lexicons/count-number.yaml +16 -0
- package/lexicons/dependent-possessive.yaml +10 -0
- package/lexicons/document-kind.yaml +19 -0
- package/lexicons/email-attachment-note.yaml +7 -0
- package/lexicons/email-attribution.yaml +7 -0
- package/lexicons/email-header-field.yaml +28 -0
- package/lexicons/email-written-field.yaml +8 -0
- package/lexicons/emphasis-word.yaml +4 -2
- package/lexicons/example-marker.yaml +12 -0
- package/lexicons/figure-elsewhere.yaml +8 -0
- package/lexicons/figure-label.yaml +14 -0
- package/lexicons/finite-auxiliary.yaml +9 -0
- package/lexicons/invariant-noun.yaml +26 -0
- package/lexicons/measure-unit.yaml +78 -0
- package/lexicons/misnumbered-phrase.yaml +7 -0
- package/lexicons/name-title.yaml +20 -0
- package/lexicons/numbered-label.yaml +26 -0
- package/lexicons/pair-opener.yaml +13 -0
- package/lexicons/percent-unit.yaml +7 -0
- package/lexicons/placeholder-word.yaml +17 -0
- package/lexicons/plural-determiner.yaml +10 -0
- package/lexicons/range-connector.yaml +12 -0
- package/lexicons/share-exception.yaml +10 -0
- package/lexicons/share-label.yaml +12 -0
- package/lexicons/singular-determiner.yaml +13 -0
- package/lexicons/stock-transition.yaml +17 -0
- package/package.json +1 -1
- package/src/citation.ts +13 -9
- package/src/code-citation.ts +52 -0
- package/src/dates.ts +22 -4
- package/src/emphasis.ts +16 -0
- package/src/index.ts +14 -5
- package/src/japanese-run.ts +9 -0
- package/src/label-stop.ts +25 -0
- package/src/pos.ts +63 -21
- package/src/regexp.ts +2 -0
- package/src/structure.ts +22 -6
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Nouns whose singular and plural are spelled alike, or that name one thing while ending in -s. The tagger may read
|
|
2
|
+
# either number into them, so agreement-slip does not count them: "a series", "a means", "these staff", "this data".
|
|
3
|
+
id: invariant-noun
|
|
4
|
+
language: en
|
|
5
|
+
entries:
|
|
6
|
+
- pattern: news
|
|
7
|
+
- pattern: series
|
|
8
|
+
- pattern: species
|
|
9
|
+
- pattern: means
|
|
10
|
+
- pattern: data
|
|
11
|
+
- pattern: media
|
|
12
|
+
- pattern: criteria
|
|
13
|
+
- pattern: phenomena
|
|
14
|
+
- pattern: staff
|
|
15
|
+
- pattern: personnel
|
|
16
|
+
- pattern: police
|
|
17
|
+
- pattern: headquarters
|
|
18
|
+
- pattern: crossroads
|
|
19
|
+
- pattern: whereabouts
|
|
20
|
+
- pattern: savings
|
|
21
|
+
- pattern: sheep
|
|
22
|
+
- pattern: fish
|
|
23
|
+
- pattern: deer
|
|
24
|
+
- pattern: aircraft
|
|
25
|
+
- pattern: spacecraft
|
|
26
|
+
- pattern: offspring
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# Unit symbols written after a measured number (1.5 mM, 37 °C, 2.5 mg/kg). A decimal at the start of a line followed
|
|
2
|
+
# by one of these is an amount, not a dotted section number. Matched as written, and only when no letter, digit or
|
|
3
|
+
# hyphen follows ("2.1 mmap" and "5.2.2 min-fresh" stay section titles). Symbols that begin with a capital letter
|
|
4
|
+
# (Da, Pa, GB, W, V) are left out: a section title begins with a capital too, and "3.1 Da Vinci" is a title.
|
|
5
|
+
id: measure-unit
|
|
6
|
+
language: en
|
|
7
|
+
entries:
|
|
8
|
+
- pattern: "mM"
|
|
9
|
+
- pattern: "µM"
|
|
10
|
+
- pattern: "μM"
|
|
11
|
+
- pattern: "nM"
|
|
12
|
+
- pattern: "pM"
|
|
13
|
+
- pattern: "mol"
|
|
14
|
+
- pattern: "mmol"
|
|
15
|
+
- pattern: "µmol"
|
|
16
|
+
- pattern: "μmol"
|
|
17
|
+
- pattern: "nmol"
|
|
18
|
+
- pattern: "pmol"
|
|
19
|
+
- pattern: "kg"
|
|
20
|
+
- pattern: "mg"
|
|
21
|
+
- pattern: "µg"
|
|
22
|
+
- pattern: "μg"
|
|
23
|
+
- pattern: "ng"
|
|
24
|
+
- pattern: "pg"
|
|
25
|
+
- pattern: "kDa"
|
|
26
|
+
- pattern: "mL"
|
|
27
|
+
- pattern: "ml"
|
|
28
|
+
- pattern: "µL"
|
|
29
|
+
- pattern: "μL"
|
|
30
|
+
- pattern: "µl"
|
|
31
|
+
- pattern: "μl"
|
|
32
|
+
- pattern: "dL"
|
|
33
|
+
- pattern: "nL"
|
|
34
|
+
- pattern: "km"
|
|
35
|
+
- pattern: "cm"
|
|
36
|
+
- pattern: "mm"
|
|
37
|
+
- pattern: "µm"
|
|
38
|
+
- pattern: "μm"
|
|
39
|
+
- pattern: "nm"
|
|
40
|
+
- pattern: "ms"
|
|
41
|
+
- pattern: "µs"
|
|
42
|
+
- pattern: "μs"
|
|
43
|
+
- pattern: "ns"
|
|
44
|
+
- pattern: "min"
|
|
45
|
+
- pattern: "sec"
|
|
46
|
+
- pattern: "hr"
|
|
47
|
+
- pattern: "hrs"
|
|
48
|
+
- pattern: "kHz"
|
|
49
|
+
- pattern: "mA"
|
|
50
|
+
- pattern: "mV"
|
|
51
|
+
- pattern: "kV"
|
|
52
|
+
- pattern: "kW"
|
|
53
|
+
- pattern: "kWh"
|
|
54
|
+
- pattern: "mAh"
|
|
55
|
+
- pattern: "kPa"
|
|
56
|
+
- pattern: "mmHg"
|
|
57
|
+
- pattern: "atm"
|
|
58
|
+
- pattern: "psi"
|
|
59
|
+
- pattern: "kB"
|
|
60
|
+
- pattern: "bp"
|
|
61
|
+
- pattern: "kb"
|
|
62
|
+
- pattern: "kbp"
|
|
63
|
+
- pattern: "ppm"
|
|
64
|
+
- pattern: "ppb"
|
|
65
|
+
- pattern: "rpm"
|
|
66
|
+
- pattern: "dB"
|
|
67
|
+
- pattern: "mCi"
|
|
68
|
+
- pattern: "µCi"
|
|
69
|
+
- pattern: "μCi"
|
|
70
|
+
- pattern: "mGy"
|
|
71
|
+
- pattern: "mSv"
|
|
72
|
+
- pattern: "°C"
|
|
73
|
+
- pattern: "°F"
|
|
74
|
+
- pattern: "℃"
|
|
75
|
+
- pattern: "℉"
|
|
76
|
+
- pattern: "°"
|
|
77
|
+
- pattern: "%"
|
|
78
|
+
- pattern: "‰"
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Phrases with a plural where English takes the singular: "remains ones of the most intriguing" is "one of the most".
|
|
2
|
+
# Not counted after a determiner, an adjective or a pronoun, where "ones" is the pronoun ("the ones of the most use").
|
|
3
|
+
id: misnumbered-phrase
|
|
4
|
+
language: en
|
|
5
|
+
entries:
|
|
6
|
+
- pattern: ones of the most
|
|
7
|
+
- pattern: ones of the least
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# Official titles set before a surname printed in capitals: a hearing transcript's speakers (Senator HAWLEY., Secretary
|
|
2
|
+
# BLINKEN.) and the Japanese government's romanised names, surname first (Prime Minister ABE Shinzo). A word in capitals
|
|
3
|
+
# right after one of these is a name, not an acronym. "Minister" also covers Prime Minister and Foreign Minister, "President"
|
|
4
|
+
# Vice President. Titles often followed by an acronym (Chair, General, Justice) are left out. Matched as written.
|
|
5
|
+
id: name-title
|
|
6
|
+
language: en
|
|
7
|
+
entries:
|
|
8
|
+
- pattern: "President"
|
|
9
|
+
- pattern: "Minister"
|
|
10
|
+
- pattern: "Secretary"
|
|
11
|
+
- pattern: "Secretary-General"
|
|
12
|
+
- pattern: "Chief Justice"
|
|
13
|
+
- pattern: "Governor"
|
|
14
|
+
- pattern: "Senator"
|
|
15
|
+
- pattern: "Ambassador"
|
|
16
|
+
- pattern: "Mayor"
|
|
17
|
+
- pattern: "Chancellor"
|
|
18
|
+
- pattern: "Commissioner"
|
|
19
|
+
- pattern: "Chairman"
|
|
20
|
+
- pattern: "Chairwoman"
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Words written before a number to name an item of a series (Step 3, Example 4). At the head of a heading they are a
|
|
2
|
+
# label, which the text below never repeats, so heading-echo does not count them in the overlap. position is the side
|
|
3
|
+
# of the number the word is written on. Matched as written.
|
|
4
|
+
id: numbered-label
|
|
5
|
+
language: en
|
|
6
|
+
entries:
|
|
7
|
+
- pattern: Step
|
|
8
|
+
position: before
|
|
9
|
+
- pattern: Example
|
|
10
|
+
position: before
|
|
11
|
+
- pattern: Case
|
|
12
|
+
position: before
|
|
13
|
+
- pattern: Exercise
|
|
14
|
+
position: before
|
|
15
|
+
- pattern: Question
|
|
16
|
+
position: before
|
|
17
|
+
- pattern: Problem
|
|
18
|
+
position: before
|
|
19
|
+
- pattern: Lesson
|
|
20
|
+
position: before
|
|
21
|
+
- pattern: Stage
|
|
22
|
+
position: before
|
|
23
|
+
- pattern: Figure
|
|
24
|
+
position: before
|
|
25
|
+
- pattern: Table
|
|
26
|
+
position: before
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Words that take exactly two things, each with the conjunction that joins its pair: "between Provider and Customer",
|
|
2
|
+
# "both state and federal", "either email or post", "whether written or oral". When one stands in the item right
|
|
3
|
+
# before its own conjunction, oxford-comma-consistency reads that conjunction as part of the item, not as the list's
|
|
4
|
+
# last joint: "the Key Terms between Provider and Customer, and any policies". "between Acme, Beta and Gamma" is still
|
|
5
|
+
# a list: "between" is not in the last item. "the impact of either option and implementation" is still a list: "either"
|
|
6
|
+
# pairs with "or", not "and".
|
|
7
|
+
id: pair-opener
|
|
8
|
+
language: en
|
|
9
|
+
entries:
|
|
10
|
+
- pattern: between and
|
|
11
|
+
- pattern: both and
|
|
12
|
+
- pattern: either or
|
|
13
|
+
- pattern: whether or
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# What a template writes inside the brackets of a blank: [Your Name], [Insert Date], [Company Name].
|
|
2
|
+
# A bracket counts when it holds one of these, alone or followed by more words.
|
|
3
|
+
id: placeholder-word
|
|
4
|
+
language: en
|
|
5
|
+
entries:
|
|
6
|
+
- pattern: your
|
|
7
|
+
- pattern: insert
|
|
8
|
+
- pattern: recipient
|
|
9
|
+
- pattern: name
|
|
10
|
+
- pattern: company name
|
|
11
|
+
- pattern: client name
|
|
12
|
+
- pattern: project name
|
|
13
|
+
- pattern: product name
|
|
14
|
+
- pattern: date
|
|
15
|
+
- pattern: describe
|
|
16
|
+
- pattern: specific
|
|
17
|
+
- pattern: add
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Determiners that count several things. A singular noun after one of these ("these new version") is a slip. They also
|
|
2
|
+
# stand alone ("These help us"), so a noun is counted only when an adjective comes between.
|
|
3
|
+
#
|
|
4
|
+
# "one of the" is left out. It takes a plural ("one of the most important features"), but in the corpus it only ever
|
|
5
|
+
# stood before a singular in "some one of the name of Cecily", which is right: "some one" is "someone".
|
|
6
|
+
id: plural-determiner
|
|
7
|
+
language: en
|
|
8
|
+
entries:
|
|
9
|
+
- pattern: these
|
|
10
|
+
- pattern: those
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# Marks and words that, alone between two dates, make a period ("5 April – 9 April", "May 1 through May 5").
|
|
2
|
+
# "to" is not one: "moved from March 10 to March 3" changes a date rather than spanning a period.
|
|
3
|
+
id: range-connector
|
|
4
|
+
language: en
|
|
5
|
+
entries:
|
|
6
|
+
- pattern: "-"
|
|
7
|
+
- pattern: –
|
|
8
|
+
- pattern: —
|
|
9
|
+
- pattern: through
|
|
10
|
+
- pattern: thru
|
|
11
|
+
- pattern: until
|
|
12
|
+
- pattern: till
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Words that mark a count whose percentages need not add up to 100%. When the header or the text above has one, the
|
|
2
|
+
# percentages are not added.
|
|
3
|
+
id: share-exception
|
|
4
|
+
language: en
|
|
5
|
+
entries:
|
|
6
|
+
- pattern: multiple answers
|
|
7
|
+
- pattern: multiple responses
|
|
8
|
+
- pattern: more than one
|
|
9
|
+
- pattern: select all
|
|
10
|
+
- pattern: all that apply
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# Words that name the parts of one whole (they add up to 100%). When a table's header, or the sentence or heading right
|
|
2
|
+
# above a list or a table, has one of these, its percentages are added and compared with 100%.
|
|
3
|
+
# "rate" and "percentage" are left out: they also name a rate per row. So is a bare "share" ("share price return").
|
|
4
|
+
id: share-label
|
|
5
|
+
language: en
|
|
6
|
+
entries:
|
|
7
|
+
- pattern: breakdown
|
|
8
|
+
- pattern: market share
|
|
9
|
+
- pattern: share of
|
|
10
|
+
- pattern: composition
|
|
11
|
+
- pattern: split
|
|
12
|
+
- pattern: allocation
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Determiners that count one thing. A plural noun after one of these ("a significant changes", "each new users") is a
|
|
2
|
+
# slip. "that" is left out: the tagger cannot tell the demonstrative from the conjunction ("so that users can").
|
|
3
|
+
# Only the articles never stand alone; after the others a noun is counted only when an adjective comes between, because
|
|
4
|
+
# "This results in a loss" is a pronoun and a verb, not a determiner and a noun.
|
|
5
|
+
id: singular-determiner
|
|
6
|
+
language: en
|
|
7
|
+
entries:
|
|
8
|
+
- pattern: a
|
|
9
|
+
- pattern: an
|
|
10
|
+
- pattern: another
|
|
11
|
+
- pattern: each
|
|
12
|
+
- pattern: every
|
|
13
|
+
- pattern: this
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# Sentence openers that glue without saying how. Words that carry the argument (However, Therefore, Thus) are not here.
|
|
2
|
+
# "In addition," keeps its comma: "In addition to X" opens a phrase, not a transition.
|
|
3
|
+
id: stock-transition
|
|
4
|
+
language: en
|
|
5
|
+
entries:
|
|
6
|
+
- pattern: Additionally
|
|
7
|
+
- pattern: Moreover
|
|
8
|
+
- pattern: Furthermore
|
|
9
|
+
- pattern: In addition,
|
|
10
|
+
- pattern: Notably
|
|
11
|
+
- pattern: Importantly
|
|
12
|
+
- pattern: Crucially
|
|
13
|
+
- pattern: Ultimately
|
|
14
|
+
- pattern: Overall,
|
|
15
|
+
- pattern: In conclusion
|
|
16
|
+
- pattern: In summary
|
|
17
|
+
- pattern: That said
|
package/package.json
CHANGED
package/src/citation.ts
CHANGED
|
@@ -52,17 +52,18 @@ export const listMembers = (rest: string, plural: boolean): ListMember[] =>
|
|
|
52
52
|
*/
|
|
53
53
|
const TAG = "(?<tag>[A-Z][A-Z0-9]{1,30})";
|
|
54
54
|
/**
|
|
55
|
-
* "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]"
|
|
56
|
-
* the tag or not, so this tag is a candidate that the core checks
|
|
55
|
+
* "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]", and a numbered citation "[19]" like a blank to
|
|
56
|
+
* fill in. Only the document tells them apart, by listing the tag or not, so this tag is a candidate that the core checks
|
|
57
|
+
* against the document (attrs.citedTag).
|
|
57
58
|
*/
|
|
58
|
-
const
|
|
59
|
+
const LISTED_TAG = "(?<tag>\\d{1,3}|[A-Z][A-Z0-9]{0,30}(?:-[A-Z0-9]{1,30}){1,4})";
|
|
59
60
|
const tagAfter = (tag: string): RegExp => new RegExp(`^\\[${tag}\\]`, "u");
|
|
60
61
|
/** "[HTTP], Section 12.1": the tag written just before the reference. */
|
|
61
62
|
const tagBefore = (tag: string): RegExp => new RegExp(`\\[${tag}\\],?\\s?$`, "u");
|
|
62
63
|
const TAG_AFTER = tagAfter(TAG);
|
|
63
64
|
const TAG_BEFORE = tagBefore(TAG);
|
|
64
|
-
const
|
|
65
|
-
const
|
|
65
|
+
const LISTED_AFTER = tagAfter(LISTED_TAG);
|
|
66
|
+
const LISTED_BEFORE = tagBefore(LISTED_TAG);
|
|
66
67
|
const TAG_REACH = 40;
|
|
67
68
|
|
|
68
69
|
const tagEndingAt = (pattern: RegExp, text: string, start: number): string | undefined =>
|
|
@@ -120,9 +121,12 @@ export const citedDocumentAfter = (text: string, end: number): string | undefine
|
|
|
120
121
|
return words.length === 1 && SELF.has(name) ? undefined : name;
|
|
121
122
|
};
|
|
122
123
|
|
|
123
|
-
/**
|
|
124
|
-
|
|
124
|
+
/**
|
|
125
|
+
* A tag that names another document only if this document lists it, right after a reference ("Section 4.2.3 of
|
|
126
|
+
* [HTTP-CACHING]", "Section 4.2.2.17 of [19]") or just before it ("[HTTP-CACHING], Section 4", "[23], Section 2.17").
|
|
127
|
+
*/
|
|
128
|
+
export const listedTagAround = (text: string, start: number, end: number): string | undefined => {
|
|
125
129
|
const named = afterOf(text, end);
|
|
126
|
-
const following = named === undefined ? undefined :
|
|
127
|
-
return following ?? tagEndingAt(
|
|
130
|
+
const following = named === undefined ? undefined : LISTED_AFTER.exec(named)?.groups?.["tag"];
|
|
131
|
+
return following ?? tagEndingAt(LISTED_BEFORE, text, start);
|
|
128
132
|
};
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import type { Lexicon, LexiconEntry } from "chaffjs/plugin";
|
|
2
|
+
import { escapeRegExp } from "./regexp.ts";
|
|
3
|
+
|
|
4
|
+
// "35 CFR §122", "42 U.S.C. § 1983", "RFC 1122, Section 3.3.4.2": a section of a code or a numbered document named
|
|
5
|
+
// right before the reference, not of this document. Which names are codes is the lexicon's (document-kind).
|
|
6
|
+
|
|
7
|
+
/** Built once from the lexicon; each pattern is undefined when the lexicon names no code of its shape. */
|
|
8
|
+
export type CodeVocabulary = {
|
|
9
|
+
readonly before: RegExp | undefined;
|
|
10
|
+
readonly numberedBefore: RegExp | undefined;
|
|
11
|
+
/** The name of a code that takes a title number, at the start of the text and ending at a word boundary. */
|
|
12
|
+
readonly titled: RegExp | undefined;
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
/** The code's title number ("35 CFR"), then the name, then at most a second "§" ("§§ 1981 and 1983") up to the reference. */
|
|
16
|
+
const TITLE_NUMBER = String.raw`(?:\d{1,3}\s+)?`;
|
|
17
|
+
const SECOND_SIGN = String.raw`\s*(?:§\s*)?$`;
|
|
18
|
+
/** The document's own number after its name, then a comma, an opening parenthesis or a space up to the reference. */
|
|
19
|
+
const OWN_NUMBER = String.raw`\s+\d{1,5}`;
|
|
20
|
+
const NUMBER_TO_REFERENCE = String.raw`(?:,\s*|\s*\(\s*|\s+)(?:§\s*)?$`;
|
|
21
|
+
const NOT_INSIDE_A_WORD = String.raw`(?<![\p{L}\p{N}_.])`;
|
|
22
|
+
const WORD_ENDS = String.raw`(?![\p{L}\p{N}_])`;
|
|
23
|
+
|
|
24
|
+
/** "35 CFR " is a few characters. Only this much before a reference is read, however long the line. */
|
|
25
|
+
const REACH = 40;
|
|
26
|
+
|
|
27
|
+
const namesOf = (entries: readonly LexiconEntry[]): string | undefined => {
|
|
28
|
+
const names = entries.map((entry) => entry.pattern).toSorted((left, right) => right.length - left.length);
|
|
29
|
+
return names.length === 0 ? undefined : `(?:${names.map(escapeRegExp).join("|")})`;
|
|
30
|
+
};
|
|
31
|
+
|
|
32
|
+
/** An entry with position "before" is a name written before its own number ("RFC 1122"); the others follow a title number. */
|
|
33
|
+
export const codeVocabulary = (lexicons: Readonly<Record<string, Lexicon>>): CodeVocabulary => {
|
|
34
|
+
const entries = lexicons["document-kind"] ?? [];
|
|
35
|
+
const titled = namesOf(entries.filter((entry) => entry.position !== "before"));
|
|
36
|
+
const numbered = namesOf(entries.filter((entry) => entry.position === "before"));
|
|
37
|
+
return {
|
|
38
|
+
before: titled === undefined ? undefined : new RegExp(String.raw`${NOT_INSIDE_A_WORD}(?<code>${TITLE_NUMBER}${titled})${SECOND_SIGN}`, "u"),
|
|
39
|
+
numberedBefore:
|
|
40
|
+
numbered === undefined ? undefined : new RegExp(String.raw`${NOT_INSIDE_A_WORD}(?<code>${numbered}${OWN_NUMBER})${NUMBER_TO_REFERENCE}`, "u"),
|
|
41
|
+
titled: titled === undefined ? undefined : new RegExp(String.raw`^${titled}${WORD_ENDS}`, "u"),
|
|
42
|
+
};
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/** The code named right before the reference at `start`, or undefined when none is. */
|
|
46
|
+
export const citedCodeBefore = (text: string, start: number, vocabulary: CodeVocabulary): string | undefined => {
|
|
47
|
+
const before = text.slice(Math.max(0, start - REACH), start);
|
|
48
|
+
return vocabulary.before?.exec(before)?.groups?.["code"] ?? vocabulary.numberedBefore?.exec(before)?.groups?.["code"];
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
/** "40 CFR § 163.25" at the start of a heading: the text after the number starts with a code named after a title number. */
|
|
52
|
+
export const titledCodeAt = (rest: string, vocabulary: CodeVocabulary): boolean => vocabulary.titled?.test(rest) === true;
|
package/src/dates.ts
CHANGED
|
@@ -3,7 +3,12 @@ import type { Mention } from "chaffjs/plugin";
|
|
|
3
3
|
// Dates in English text, and the weekday written beside one.
|
|
4
4
|
|
|
5
5
|
const MONTHS = ["january", "february", "march", "april", "may", "june", "july", "august", "september", "october", "november", "december"];
|
|
6
|
-
|
|
6
|
+
/** "Sep", "Sept": a month cut short, which may carry a period ("Sept. 12, 2025"). */
|
|
7
|
+
const ABBREVIATED: ReadonlyMap<string, number> = new Map([...MONTHS.map((name, index) => [name.slice(0, 3), index + 1] as const), ["sept", 9]]);
|
|
8
|
+
/** Title case or capitals ("SEP 01, 2022" on a web page): a lower-case "mar" or "may" is never a month. */
|
|
9
|
+
const MONTH_WORD = /\b(?<month>[A-Z][a-z]{2,8}|[A-Z]{3,9})\b/gu;
|
|
10
|
+
/** "12-SEP-2025", the day-month-year form of labels and logs. Only a three-letter month is written this way. */
|
|
11
|
+
const HYPHENATED_DATE = /(?<![\w-])(?<d>\d{1,2})-(?<month>[A-Z][a-z]{2}|[A-Z]{3})-(?<y>\d{4})(?![\w-])/gu;
|
|
7
12
|
const ISO_DATE = /\b(?<y>\d{4})-(?<m>\d{2})-(?<d>\d{2})\b/gu;
|
|
8
13
|
const DAY_BEFORE = /(?<d>\d{1,2})(?:st|nd|rd|th)? $/u;
|
|
9
14
|
const DAY_YEAR_AFTER = /^ (?<d>\d{1,2})(?:st|nd|rd|th)?,? (?<y>\d{4})\b/u;
|
|
@@ -41,12 +46,21 @@ const monthYear = (text: string, at: number, end: number, month: number): Mentio
|
|
|
41
46
|
* The month is found first and its neighbours read with anchored patterns, never one long alternation.
|
|
42
47
|
*/
|
|
43
48
|
const namedDate = (text: string, match: RegExpExecArray): Mention | undefined => {
|
|
44
|
-
const
|
|
49
|
+
const word = (match.groups?.["month"] ?? "").toLowerCase();
|
|
50
|
+
const full = MONTHS.indexOf(word) + 1;
|
|
51
|
+
const month = full === 0 ? (ABBREVIATED.get(word) ?? 0) : full;
|
|
45
52
|
if (month === 0) return undefined;
|
|
46
|
-
const end = match.index + match[0].length;
|
|
53
|
+
const end = match.index + match[0].length + (full === 0 && text.charAt(match.index + match[0].length) === "." ? 1 : 0);
|
|
47
54
|
return monthDayYear(text, match.index, end, month) ?? monthYear(text, match.index, end, month);
|
|
48
55
|
};
|
|
49
56
|
|
|
57
|
+
const hyphenatedDate = (match: RegExpExecArray): Mention[] => {
|
|
58
|
+
const month = ABBREVIATED.get((match.groups?.["month"] ?? "").toLowerCase());
|
|
59
|
+
if (month === undefined) return [];
|
|
60
|
+
const value = `${match.groups?.["y"] ?? ""}-${pad(String(month))}-${pad(match.groups?.["d"] ?? "")}`;
|
|
61
|
+
return [{ start: match.index, end: match.index + match[0].length, attrs: { value } }];
|
|
62
|
+
};
|
|
63
|
+
|
|
50
64
|
const isoDate = (match: RegExpExecArray): Mention => ({
|
|
51
65
|
start: match.index,
|
|
52
66
|
end: match.index + match[0].length,
|
|
@@ -88,6 +102,10 @@ const withWeekday = (text: string, date: Mention): Mention => {
|
|
|
88
102
|
};
|
|
89
103
|
|
|
90
104
|
export const dates = (text: string): Mention[] =>
|
|
91
|
-
[
|
|
105
|
+
[
|
|
106
|
+
...[...text.matchAll(MONTH_WORD)].flatMap((match) => namedDate(text, match) ?? []),
|
|
107
|
+
...[...text.matchAll(HYPHENATED_DATE)].flatMap(hyphenatedDate),
|
|
108
|
+
...[...text.matchAll(ISO_DATE)].map(isoDate),
|
|
109
|
+
]
|
|
92
110
|
.toSorted((left, right) => left.start - right.start)
|
|
93
111
|
.map((date) => withWeekday(text, date));
|
package/src/emphasis.ts
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
// A word written in capitals for emphasis (the office will NEVER call you), told from an acronym by the dictionary.
|
|
2
|
+
|
|
3
|
+
const CAPITALS = /^\p{Lu}{2,}$/u;
|
|
4
|
+
|
|
5
|
+
const ADVERB_TAGS: ReadonlySet<string> = new Set(["RB", "RBR", "RBS"]);
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* The dictionary knows the lower-case word, and only as an adverb (never, always, also). An acronym names a thing, so a word
|
|
9
|
+
* with no noun or adjective reading cannot be one. A word that is also a noun or an adjective (fast, eagle, cheese) is kept:
|
|
10
|
+
* FAST and EAGLE are a telescope and a simulation, and the context tagger cannot tell them from emphasis.
|
|
11
|
+
*/
|
|
12
|
+
export const isEmphasisedAdverb = (surface: string, tagsOf: (word: string) => readonly string[] | undefined): boolean => {
|
|
13
|
+
if (!CAPITALS.test(surface)) return false;
|
|
14
|
+
const tags = tagsOf(surface.toLowerCase()) ?? [];
|
|
15
|
+
return tags.length > 0 && tags.every((tag) => ADVERB_TAGS.has(tag));
|
|
16
|
+
};
|
package/src/index.ts
CHANGED
|
@@ -1,14 +1,19 @@
|
|
|
1
1
|
import { loadLexicons } from "./lexicons.ts";
|
|
2
2
|
import { unmarkNumberStops } from "./number-stop.ts";
|
|
3
|
+
import { labelStops, unmarkLabelStops } from "./label-stop.ts";
|
|
3
4
|
import { sentenceSpans } from "./sentence-split.ts";
|
|
4
5
|
import { splitAtQuotedStops } from "./quoted-stop.ts";
|
|
5
6
|
import { reattachClosingQuotes } from "./closing-quote.ts";
|
|
6
7
|
import { structure } from "./structure.ts";
|
|
7
8
|
import { isReady, prepare, tokenize } from "./pos.ts";
|
|
8
|
-
import
|
|
9
|
+
import { isJapaneseRun } from "./japanese-run.ts";
|
|
10
|
+
import type { AdapterNeeds, EmbeddedLanguage, LanguageAdapter, Segmentation, Sentence, Span } from "chaffjs/plugin";
|
|
9
11
|
|
|
10
12
|
// chaff からは型だけを取る。実行時の値依存を作らない。アダプタは単体で動く。
|
|
11
13
|
|
|
14
|
+
const LEXICONS = loadLexicons();
|
|
15
|
+
const LABEL_STOPS = labelStops((LEXICONS["abbreviated-label"] ?? []).map((entry) => entry.pattern));
|
|
16
|
+
|
|
12
17
|
const LATIN_LETTER = /[a-z]/giu;
|
|
13
18
|
const COUNTABLE = /\S/gu;
|
|
14
19
|
|
|
@@ -25,9 +30,13 @@ const withTokens = (sentence: Sentence): Sentence => {
|
|
|
25
30
|
};
|
|
26
31
|
};
|
|
27
32
|
|
|
33
|
+
const JAPANESE: EmbeddedLanguage = { id: "ja", lengthUnit: "char" };
|
|
34
|
+
|
|
35
|
+
const withLanguage = (text: string, span: Span): Sentence => (isJapaneseRun(text) ? { span, text, embeddedLanguage: JAPANESE } : { span, text });
|
|
36
|
+
|
|
28
37
|
/**
|
|
29
38
|
* 英語は sentence-splitter の既定にほぼ任せる。"Dr." "e.g." "U.S." "$3.50" を
|
|
30
|
-
*
|
|
39
|
+
* いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませることと、番号の前の略した名前(FIG. 1、Vol. XLIII)の点で切らないこと、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
|
|
31
40
|
*/
|
|
32
41
|
export const adapter: LanguageAdapter = {
|
|
33
42
|
kind: "language",
|
|
@@ -51,11 +60,11 @@ export const adapter: LanguageAdapter = {
|
|
|
51
60
|
if (total === 0) return 0;
|
|
52
61
|
return [...source.matchAll(LATIN_LETTER)].length / total;
|
|
53
62
|
},
|
|
54
|
-
lexicons:
|
|
63
|
+
lexicons: LEXICONS,
|
|
55
64
|
structure,
|
|
56
65
|
segment: (text: string): Segmentation => {
|
|
57
|
-
const quotedStops = sentenceSpans(unmarkNumberStops(text)).flatMap((span) => splitAtQuotedStops(text, span));
|
|
58
|
-
const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => (
|
|
66
|
+
const quotedStops = sentenceSpans(unmarkNumberStops(unmarkLabelStops(text, LABEL_STOPS))).flatMap((span) => splitAtQuotedStops(text, span));
|
|
67
|
+
const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => withLanguage(text.slice(span.start, span.end), span));
|
|
59
68
|
return { sentences: isReady() ? sentences.map(withTokens) : sentences };
|
|
60
69
|
},
|
|
61
70
|
};
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A Japanese sentence in an English document (a quoted notice, a bilingual title): it has kana, and kana or kanji make
|
|
3
|
+
* up at least half of its letters (digits and symbols are not counted). Kanji alone do not count, as a Chinese name in English text is not Japanese.
|
|
4
|
+
*/
|
|
5
|
+
const KANA = /[\p{Script=Hiragana}\p{Script=Katakana}]/u;
|
|
6
|
+
const JAPANESE = /[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}]/gu;
|
|
7
|
+
const LETTER = /\p{L}/gu;
|
|
8
|
+
|
|
9
|
+
export const isJapaneseRun = (text: string): boolean => KANA.test(text) && [...text.matchAll(JAPANESE)].length * 2 >= [...text.matchAll(LETTER)].length;
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { escapeRegExp } from "./regexp.ts";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* sentence-splitter ends a sentence at the full stop of a label before its number: "FIG." and "1 illustrates …" were two
|
|
5
|
+
* sentences, and so were "Vol." and "XLIII (1979)". Where a number follows, that full stop is replaced with a letter
|
|
6
|
+
* before splitting. The length does not change, so the spans fit the original text.
|
|
7
|
+
* A lone "I" after a label is the pronoun ("He said No. I left."), and a word ("the last Fig. The tree") is not a number.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Built once from the lexicon; undefined when no label ends in a full stop. */
|
|
11
|
+
export type LabelStops = RegExp | undefined;
|
|
12
|
+
|
|
13
|
+
const NUMBER_AFTER = String.raw`(?=\s+(?:\d|(?:[IVXLCDM]{2,}|[VXLCDM])(?![\p{L}\p{N}_])))`;
|
|
14
|
+
const PLAIN_LETTER = "n";
|
|
15
|
+
|
|
16
|
+
/** The labels as written and in capitals, without their full stop ("Fig." → Fig, FIG). */
|
|
17
|
+
export const labelStops = (labels: readonly string[]): LabelStops => {
|
|
18
|
+
const stems = labels.filter((label) => label.endsWith(".")).flatMap((label) => [label.slice(0, -1), label.slice(0, -1).toUpperCase()]);
|
|
19
|
+
if (stems.length === 0) return undefined;
|
|
20
|
+
return new RegExp(String.raw`(?<![\p{L}\p{N}_.])(?:${[...new Set(stems)].map(escapeRegExp).join("|")})\.${NUMBER_AFTER}`, "gu");
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
/** The text with the full stop of every label before a number replaced by a letter. */
|
|
24
|
+
export const unmarkLabelStops = (text: string, stops: LabelStops): string =>
|
|
25
|
+
stops === undefined ? text : text.replace(stops, (label: string) => `${label.slice(0, -1)}${PLAIN_LETTER}`);
|