@chaffjs/lang-en 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/dist/citation.d.ts +5 -2
  2. package/dist/citation.d.ts.map +1 -1
  3. package/dist/citation.js +13 -9
  4. package/dist/citation.js.map +1 -1
  5. package/dist/code-citation.d.ts +15 -0
  6. package/dist/code-citation.d.ts.map +1 -0
  7. package/dist/code-citation.js +34 -0
  8. package/dist/code-citation.js.map +1 -0
  9. package/dist/dates.d.ts.map +1 -1
  10. package/dist/dates.js +22 -4
  11. package/dist/dates.js.map +1 -1
  12. package/dist/emphasis.d.ts +7 -0
  13. package/dist/emphasis.d.ts.map +1 -0
  14. package/dist/emphasis.js +15 -0
  15. package/dist/emphasis.js.map +1 -0
  16. package/dist/index.d.ts +1 -1
  17. package/dist/index.d.ts.map +1 -1
  18. package/dist/index.js +10 -4
  19. package/dist/index.js.map +1 -1
  20. package/dist/japanese-run.d.ts +2 -0
  21. package/dist/japanese-run.d.ts.map +1 -0
  22. package/dist/japanese-run.js +9 -0
  23. package/dist/japanese-run.js.map +1 -0
  24. package/dist/label-stop.d.ts +13 -0
  25. package/dist/label-stop.d.ts.map +1 -0
  26. package/dist/label-stop.js +13 -0
  27. package/dist/label-stop.js.map +1 -0
  28. package/dist/pos.d.ts.map +1 -1
  29. package/dist/pos.js +61 -18
  30. package/dist/pos.js.map +1 -1
  31. package/dist/regexp.d.ts +3 -0
  32. package/dist/regexp.d.ts.map +1 -0
  33. package/dist/regexp.js +3 -0
  34. package/dist/regexp.js.map +1 -0
  35. package/dist/structure.d.ts.map +1 -1
  36. package/dist/structure.js +17 -6
  37. package/dist/structure.js.map +1 -1
  38. package/lexicons/abbreviated-label.yaml +16 -0
  39. package/lexicons/ai-tell.yaml +86 -0
  40. package/lexicons/announcing-opener.yaml +19 -0
  41. package/lexicons/assistant-residue.yaml +55 -0
  42. package/lexicons/contrast-frame.yaml +10 -0
  43. package/lexicons/contrast-lead.yaml +15 -0
  44. package/lexicons/contrast-turn.yaml +11 -0
  45. package/lexicons/count-adjective.yaml +8 -0
  46. package/lexicons/count-anchor.yaml +7 -0
  47. package/lexicons/count-counter.yaml +40 -0
  48. package/lexicons/count-hedge.yaml +67 -0
  49. package/lexicons/count-number.yaml +16 -0
  50. package/lexicons/dependent-possessive.yaml +10 -0
  51. package/lexicons/document-kind.yaml +19 -0
  52. package/lexicons/email-attachment-note.yaml +7 -0
  53. package/lexicons/email-attribution.yaml +7 -0
  54. package/lexicons/email-header-field.yaml +28 -0
  55. package/lexicons/email-written-field.yaml +8 -0
  56. package/lexicons/emphasis-word.yaml +4 -2
  57. package/lexicons/example-marker.yaml +12 -0
  58. package/lexicons/figure-elsewhere.yaml +8 -0
  59. package/lexicons/figure-label.yaml +14 -0
  60. package/lexicons/finite-auxiliary.yaml +9 -0
  61. package/lexicons/invariant-noun.yaml +26 -0
  62. package/lexicons/measure-unit.yaml +78 -0
  63. package/lexicons/misnumbered-phrase.yaml +7 -0
  64. package/lexicons/name-title.yaml +20 -0
  65. package/lexicons/numbered-label.yaml +26 -0
  66. package/lexicons/pair-opener.yaml +13 -0
  67. package/lexicons/percent-unit.yaml +7 -0
  68. package/lexicons/placeholder-word.yaml +17 -0
  69. package/lexicons/plural-determiner.yaml +10 -0
  70. package/lexicons/range-connector.yaml +12 -0
  71. package/lexicons/share-exception.yaml +10 -0
  72. package/lexicons/share-label.yaml +12 -0
  73. package/lexicons/singular-determiner.yaml +13 -0
  74. package/lexicons/stock-transition.yaml +17 -0
  75. package/package.json +1 -1
  76. package/src/citation.ts +13 -9
  77. package/src/code-citation.ts +52 -0
  78. package/src/dates.ts +22 -4
  79. package/src/emphasis.ts +16 -0
  80. package/src/index.ts +14 -5
  81. package/src/japanese-run.ts +9 -0
  82. package/src/label-stop.ts +25 -0
  83. package/src/pos.ts +63 -21
  84. package/src/regexp.ts +2 -0
  85. package/src/structure.ts +22 -6
@@ -0,0 +1,9 @@
1
+ # Forms of "be" and "have" that are never a noun, so a possessive directly before one ("their is") is a slip.
2
+ id: finite-auxiliary
3
+ language: en
4
+ entries:
5
+ - pattern: is
6
+ - pattern: are
7
+ - pattern: was
8
+ - pattern: were
9
+ - pattern: has
@@ -0,0 +1,26 @@
1
+ # Nouns whose singular and plural are spelled alike, or that name one thing while ending in -s. The tagger may read
2
+ # either number into them, so agreement-slip does not count them: "a series", "a means", "these staff", "this data".
3
+ id: invariant-noun
4
+ language: en
5
+ entries:
6
+ - pattern: news
7
+ - pattern: series
8
+ - pattern: species
9
+ - pattern: means
10
+ - pattern: data
11
+ - pattern: media
12
+ - pattern: criteria
13
+ - pattern: phenomena
14
+ - pattern: staff
15
+ - pattern: personnel
16
+ - pattern: police
17
+ - pattern: headquarters
18
+ - pattern: crossroads
19
+ - pattern: whereabouts
20
+ - pattern: savings
21
+ - pattern: sheep
22
+ - pattern: fish
23
+ - pattern: deer
24
+ - pattern: aircraft
25
+ - pattern: spacecraft
26
+ - pattern: offspring
@@ -0,0 +1,78 @@
1
+ # Unit symbols written after a measured number (1.5 mM, 37 °C, 2.5 mg/kg). A decimal at the start of a line followed
2
+ # by one of these is an amount, not a dotted section number. Matched as written, and only when no letter, digit or
3
+ # hyphen follows ("2.1 mmap" and "5.2.2 min-fresh" stay section titles). Symbols that begin with a capital letter
4
+ # (Da, Pa, GB, W, V) are left out: a section title begins with a capital too, and "3.1 Da Vinci" is a title.
5
+ id: measure-unit
6
+ language: en
7
+ entries:
8
+ - pattern: "mM"
9
+ - pattern: "µM"
10
+ - pattern: "μM"
11
+ - pattern: "nM"
12
+ - pattern: "pM"
13
+ - pattern: "mol"
14
+ - pattern: "mmol"
15
+ - pattern: "µmol"
16
+ - pattern: "μmol"
17
+ - pattern: "nmol"
18
+ - pattern: "pmol"
19
+ - pattern: "kg"
20
+ - pattern: "mg"
21
+ - pattern: "µg"
22
+ - pattern: "μg"
23
+ - pattern: "ng"
24
+ - pattern: "pg"
25
+ - pattern: "kDa"
26
+ - pattern: "mL"
27
+ - pattern: "ml"
28
+ - pattern: "µL"
29
+ - pattern: "μL"
30
+ - pattern: "µl"
31
+ - pattern: "μl"
32
+ - pattern: "dL"
33
+ - pattern: "nL"
34
+ - pattern: "km"
35
+ - pattern: "cm"
36
+ - pattern: "mm"
37
+ - pattern: "µm"
38
+ - pattern: "μm"
39
+ - pattern: "nm"
40
+ - pattern: "ms"
41
+ - pattern: "µs"
42
+ - pattern: "μs"
43
+ - pattern: "ns"
44
+ - pattern: "min"
45
+ - pattern: "sec"
46
+ - pattern: "hr"
47
+ - pattern: "hrs"
48
+ - pattern: "kHz"
49
+ - pattern: "mA"
50
+ - pattern: "mV"
51
+ - pattern: "kV"
52
+ - pattern: "kW"
53
+ - pattern: "kWh"
54
+ - pattern: "mAh"
55
+ - pattern: "kPa"
56
+ - pattern: "mmHg"
57
+ - pattern: "atm"
58
+ - pattern: "psi"
59
+ - pattern: "kB"
60
+ - pattern: "bp"
61
+ - pattern: "kb"
62
+ - pattern: "kbp"
63
+ - pattern: "ppm"
64
+ - pattern: "ppb"
65
+ - pattern: "rpm"
66
+ - pattern: "dB"
67
+ - pattern: "mCi"
68
+ - pattern: "µCi"
69
+ - pattern: "μCi"
70
+ - pattern: "mGy"
71
+ - pattern: "mSv"
72
+ - pattern: "°C"
73
+ - pattern: "°F"
74
+ - pattern: "℃"
75
+ - pattern: "℉"
76
+ - pattern: "°"
77
+ - pattern: "%"
78
+ - pattern: "‰"
@@ -0,0 +1,7 @@
1
+ # Phrases with a plural where English takes the singular: "remains ones of the most intriguing" is "one of the most".
2
+ # Not counted after a determiner, an adjective or a pronoun, where "ones" is the pronoun ("the ones of the most use").
3
+ id: misnumbered-phrase
4
+ language: en
5
+ entries:
6
+ - pattern: ones of the most
7
+ - pattern: ones of the least
@@ -0,0 +1,20 @@
1
+ # Official titles set before a surname printed in capitals: a hearing transcript's speakers (Senator HAWLEY., Secretary
2
+ # BLINKEN.) and the Japanese government's romanised names, surname first (Prime Minister ABE Shinzo). A word in capitals
3
+ # right after one of these is a name, not an acronym. "Minister" also covers Prime Minister and Foreign Minister, "President"
4
+ # Vice President. Titles often followed by an acronym (Chair, General, Justice) are left out. Matched as written.
5
+ id: name-title
6
+ language: en
7
+ entries:
8
+ - pattern: "President"
9
+ - pattern: "Minister"
10
+ - pattern: "Secretary"
11
+ - pattern: "Secretary-General"
12
+ - pattern: "Chief Justice"
13
+ - pattern: "Governor"
14
+ - pattern: "Senator"
15
+ - pattern: "Ambassador"
16
+ - pattern: "Mayor"
17
+ - pattern: "Chancellor"
18
+ - pattern: "Commissioner"
19
+ - pattern: "Chairman"
20
+ - pattern: "Chairwoman"
@@ -0,0 +1,26 @@
1
+ # Words written before a number to name an item of a series (Step 3, Example 4). At the head of a heading they are a
2
+ # label, which the text below never repeats, so heading-echo does not count them in the overlap. position is the side
3
+ # of the number the word is written on. Matched as written.
4
+ id: numbered-label
5
+ language: en
6
+ entries:
7
+ - pattern: Step
8
+ position: before
9
+ - pattern: Example
10
+ position: before
11
+ - pattern: Case
12
+ position: before
13
+ - pattern: Exercise
14
+ position: before
15
+ - pattern: Question
16
+ position: before
17
+ - pattern: Problem
18
+ position: before
19
+ - pattern: Lesson
20
+ position: before
21
+ - pattern: Stage
22
+ position: before
23
+ - pattern: Figure
24
+ position: before
25
+ - pattern: Table
26
+ position: before
@@ -0,0 +1,13 @@
1
+ # Words that take exactly two things, each with the conjunction that joins its pair: "between Provider and Customer",
2
+ # "both state and federal", "either email or post", "whether written or oral". When one stands in the item right
3
+ # before its own conjunction, oxford-comma-consistency reads that conjunction as part of the item, not as the list's
4
+ # last joint: "the Key Terms between Provider and Customer, and any policies". "between Acme, Beta and Gamma" is still
5
+ # a list: "between" is not in the last item. "the impact of either option and implementation" is still a list: "either"
6
+ # pairs with "or", not "and".
7
+ id: pair-opener
8
+ language: en
9
+ entries:
10
+ - pattern: between and
11
+ - pattern: both and
12
+ - pattern: either or
13
+ - pattern: whether or
@@ -0,0 +1,7 @@
1
+ # The marks and words written as the unit of a percentage.
2
+ id: percent-unit
3
+ language: en
4
+ entries:
5
+ - pattern: "%"
6
+ - pattern: percent
7
+ - pattern: per cent
@@ -0,0 +1,17 @@
1
+ # What a template writes inside the brackets of a blank: [Your Name], [Insert Date], [Company Name].
2
+ # A bracket counts when it holds one of these, alone or followed by more words.
3
+ id: placeholder-word
4
+ language: en
5
+ entries:
6
+ - pattern: your
7
+ - pattern: insert
8
+ - pattern: recipient
9
+ - pattern: name
10
+ - pattern: company name
11
+ - pattern: client name
12
+ - pattern: project name
13
+ - pattern: product name
14
+ - pattern: date
15
+ - pattern: describe
16
+ - pattern: specific
17
+ - pattern: add
@@ -0,0 +1,10 @@
1
+ # Determiners that count several things. A singular noun after one of these ("these new version") is a slip. They also
2
+ # stand alone ("These help us"), so a noun is counted only when an adjective comes between.
3
+ #
4
+ # "one of the" is left out. It takes a plural ("one of the most important features"), but in the corpus it only ever
5
+ # stood before a singular in "some one of the name of Cecily", which is right: "some one" is "someone".
6
+ id: plural-determiner
7
+ language: en
8
+ entries:
9
+ - pattern: these
10
+ - pattern: those
@@ -0,0 +1,12 @@
1
+ # Marks and words that, alone between two dates, make a period ("5 April – 9 April", "May 1 through May 5").
2
+ # "to" is not one: "moved from March 10 to March 3" changes a date rather than spanning a period.
3
+ id: range-connector
4
+ language: en
5
+ entries:
6
+ - pattern: "-"
7
+ - pattern: –
8
+ - pattern: —
9
+ - pattern: through
10
+ - pattern: thru
11
+ - pattern: until
12
+ - pattern: till
@@ -0,0 +1,10 @@
1
+ # Words that mark a count whose percentages need not add up to 100%. When the header or the text above has one, the
2
+ # percentages are not added.
3
+ id: share-exception
4
+ language: en
5
+ entries:
6
+ - pattern: multiple answers
7
+ - pattern: multiple responses
8
+ - pattern: more than one
9
+ - pattern: select all
10
+ - pattern: all that apply
@@ -0,0 +1,12 @@
1
+ # Words that name the parts of one whole (they add up to 100%). When a table's header, or the sentence or heading right
2
+ # above a list or a table, has one of these, its percentages are added and compared with 100%.
3
+ # "rate" and "percentage" are left out: they also name a rate per row. So is a bare "share" ("share price return").
4
+ id: share-label
5
+ language: en
6
+ entries:
7
+ - pattern: breakdown
8
+ - pattern: market share
9
+ - pattern: share of
10
+ - pattern: composition
11
+ - pattern: split
12
+ - pattern: allocation
@@ -0,0 +1,13 @@
1
+ # Determiners that count one thing. A plural noun after one of these ("a significant changes", "each new users") is a
2
+ # slip. "that" is left out: the tagger cannot tell the demonstrative from the conjunction ("so that users can").
3
+ # Only the articles never stand alone; after the others a noun is counted only when an adjective comes between, because
4
+ # "This results in a loss" is a pronoun and a verb, not a determiner and a noun.
5
+ id: singular-determiner
6
+ language: en
7
+ entries:
8
+ - pattern: a
9
+ - pattern: an
10
+ - pattern: another
11
+ - pattern: each
12
+ - pattern: every
13
+ - pattern: this
@@ -0,0 +1,17 @@
1
+ # Sentence openers that glue without saying how. Words that carry the argument (However, Therefore, Thus) are not here.
2
+ # "In addition," keeps its comma: "In addition to X" opens a phrase, not a transition.
3
+ id: stock-transition
4
+ language: en
5
+ entries:
6
+ - pattern: Additionally
7
+ - pattern: Moreover
8
+ - pattern: Furthermore
9
+ - pattern: In addition,
10
+ - pattern: Notably
11
+ - pattern: Importantly
12
+ - pattern: Crucially
13
+ - pattern: Ultimately
14
+ - pattern: Overall,
15
+ - pattern: In conclusion
16
+ - pattern: In summary
17
+ - pattern: That said
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@chaffjs/lang-en",
3
- "version": "0.14.0",
3
+ "version": "0.16.0",
4
4
  "description": "English language adapter for chaff",
5
5
  "license": "MIT",
6
6
  "author": "isamu",
package/src/citation.ts CHANGED
@@ -52,17 +52,18 @@ export const listMembers = (rest: string, plural: boolean): ListMember[] =>
52
52
  */
53
53
  const TAG = "(?<tag>[A-Z][A-Z0-9]{1,30})";
54
54
  /**
55
- * "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]". Only the document tells them apart, by listing
56
- * the tag or not, so this tag is a candidate that the core checks against the document (attrs.citedTag).
55
+ * "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]", and a numbered citation "[19]" like a blank to
56
+ * fill in. Only the document tells them apart, by listing the tag or not, so this tag is a candidate that the core checks
57
+ * against the document (attrs.citedTag).
57
58
  */
58
- const HYPHENATED_TAG = "(?<tag>[A-Z][A-Z0-9]{0,30}(?:-[A-Z0-9]{1,30}){1,4})";
59
+ const LISTED_TAG = "(?<tag>\\d{1,3}|[A-Z][A-Z0-9]{0,30}(?:-[A-Z0-9]{1,30}){1,4})";
59
60
  const tagAfter = (tag: string): RegExp => new RegExp(`^\\[${tag}\\]`, "u");
60
61
  /** "[HTTP], Section 12.1": the tag written just before the reference. */
61
62
  const tagBefore = (tag: string): RegExp => new RegExp(`\\[${tag}\\],?\\s?$`, "u");
62
63
  const TAG_AFTER = tagAfter(TAG);
63
64
  const TAG_BEFORE = tagBefore(TAG);
64
- const HYPHENATED_AFTER = tagAfter(HYPHENATED_TAG);
65
- const HYPHENATED_BEFORE = tagBefore(HYPHENATED_TAG);
65
+ const LISTED_AFTER = tagAfter(LISTED_TAG);
66
+ const LISTED_BEFORE = tagBefore(LISTED_TAG);
66
67
  const TAG_REACH = 40;
67
68
 
68
69
  const tagEndingAt = (pattern: RegExp, text: string, start: number): string | undefined =>
@@ -120,9 +121,12 @@ export const citedDocumentAfter = (text: string, end: number): string | undefine
120
121
  return words.length === 1 && SELF.has(name) ? undefined : name;
121
122
  };
122
123
 
123
- /** A hyphenated tag right after a reference ("Section 4.2.3 of [HTTP-CACHING]") or just before it ("[HTTP-CACHING], Section 4"). */
124
- export const hyphenatedTagAround = (text: string, start: number, end: number): string | undefined => {
124
+ /**
125
+ * A tag that names another document only if this document lists it, right after a reference ("Section 4.2.3 of
126
+ * [HTTP-CACHING]", "Section 4.2.2.17 of [19]") or just before it ("[HTTP-CACHING], Section 4", "[23], Section 2.17").
127
+ */
128
+ export const listedTagAround = (text: string, start: number, end: number): string | undefined => {
125
129
  const named = afterOf(text, end);
126
- const following = named === undefined ? undefined : HYPHENATED_AFTER.exec(named)?.groups?.["tag"];
127
- return following ?? tagEndingAt(HYPHENATED_BEFORE, text, start);
130
+ const following = named === undefined ? undefined : LISTED_AFTER.exec(named)?.groups?.["tag"];
131
+ return following ?? tagEndingAt(LISTED_BEFORE, text, start);
128
132
  };
@@ -0,0 +1,52 @@
1
+ import type { Lexicon, LexiconEntry } from "chaffjs/plugin";
2
+ import { escapeRegExp } from "./regexp.ts";
3
+
4
+ // "35 CFR §122", "42 U.S.C. § 1983", "RFC 1122, Section 3.3.4.2": a section of a code or a numbered document named
5
+ // right before the reference, not of this document. Which names are codes is the lexicon's (document-kind).
6
+
7
+ /** Built once from the lexicon; each pattern is undefined when the lexicon names no code of its shape. */
8
+ export type CodeVocabulary = {
9
+ readonly before: RegExp | undefined;
10
+ readonly numberedBefore: RegExp | undefined;
11
+ /** The name of a code that takes a title number, at the start of the text and ending at a word boundary. */
12
+ readonly titled: RegExp | undefined;
13
+ };
14
+
15
+ /** The code's title number ("35 CFR"), then the name, then at most a second "§" ("§§ 1981 and 1983") up to the reference. */
16
+ const TITLE_NUMBER = String.raw`(?:\d{1,3}\s+)?`;
17
+ const SECOND_SIGN = String.raw`\s*(?:§\s*)?$`;
18
+ /** The document's own number after its name, then a comma, an opening parenthesis or a space up to the reference. */
19
+ const OWN_NUMBER = String.raw`\s+\d{1,5}`;
20
+ const NUMBER_TO_REFERENCE = String.raw`(?:,\s*|\s*\(\s*|\s+)(?:§\s*)?$`;
21
+ const NOT_INSIDE_A_WORD = String.raw`(?<![\p{L}\p{N}_.])`;
22
+ const WORD_ENDS = String.raw`(?![\p{L}\p{N}_])`;
23
+
24
+ /** "35 CFR " is a few characters. Only this much before a reference is read, however long the line. */
25
+ const REACH = 40;
26
+
27
+ const namesOf = (entries: readonly LexiconEntry[]): string | undefined => {
28
+ const names = entries.map((entry) => entry.pattern).toSorted((left, right) => right.length - left.length);
29
+ return names.length === 0 ? undefined : `(?:${names.map(escapeRegExp).join("|")})`;
30
+ };
31
+
32
+ /** An entry with position "before" is a name written before its own number ("RFC 1122"); the others follow a title number. */
33
+ export const codeVocabulary = (lexicons: Readonly<Record<string, Lexicon>>): CodeVocabulary => {
34
+ const entries = lexicons["document-kind"] ?? [];
35
+ const titled = namesOf(entries.filter((entry) => entry.position !== "before"));
36
+ const numbered = namesOf(entries.filter((entry) => entry.position === "before"));
37
+ return {
38
+ before: titled === undefined ? undefined : new RegExp(String.raw`${NOT_INSIDE_A_WORD}(?<code>${TITLE_NUMBER}${titled})${SECOND_SIGN}`, "u"),
39
+ numberedBefore:
40
+ numbered === undefined ? undefined : new RegExp(String.raw`${NOT_INSIDE_A_WORD}(?<code>${numbered}${OWN_NUMBER})${NUMBER_TO_REFERENCE}`, "u"),
41
+ titled: titled === undefined ? undefined : new RegExp(String.raw`^${titled}${WORD_ENDS}`, "u"),
42
+ };
43
+ };
44
+
45
+ /** The code named right before the reference at `start`, or undefined when none is. */
46
+ export const citedCodeBefore = (text: string, start: number, vocabulary: CodeVocabulary): string | undefined => {
47
+ const before = text.slice(Math.max(0, start - REACH), start);
48
+ return vocabulary.before?.exec(before)?.groups?.["code"] ?? vocabulary.numberedBefore?.exec(before)?.groups?.["code"];
49
+ };
50
+
51
+ /** "40 CFR § 163.25" at the start of a heading: the text after the number starts with a code named after a title number. */
52
+ export const titledCodeAt = (rest: string, vocabulary: CodeVocabulary): boolean => vocabulary.titled?.test(rest) === true;
package/src/dates.ts CHANGED
@@ -3,7 +3,12 @@ import type { Mention } from "chaffjs/plugin";
3
3
  // Dates in English text, and the weekday written beside one.
4
4
 
5
5
  const MONTHS = ["january", "february", "march", "april", "may", "june", "july", "august", "september", "october", "november", "december"];
6
- const MONTH_WORD = /\b(?<month>[A-Z][a-z]{2,8})\b/gu;
6
+ /** "Sep", "Sept": a month cut short, which may carry a period ("Sept. 12, 2025"). */
7
+ const ABBREVIATED: ReadonlyMap<string, number> = new Map([...MONTHS.map((name, index) => [name.slice(0, 3), index + 1] as const), ["sept", 9]]);
8
+ /** Title case or capitals ("SEP 01, 2022" on a web page): a lower-case "mar" or "may" is never a month. */
9
+ const MONTH_WORD = /\b(?<month>[A-Z][a-z]{2,8}|[A-Z]{3,9})\b/gu;
10
+ /** "12-SEP-2025", the day-month-year form of labels and logs. Only a three-letter month is written this way. */
11
+ const HYPHENATED_DATE = /(?<![\w-])(?<d>\d{1,2})-(?<month>[A-Z][a-z]{2}|[A-Z]{3})-(?<y>\d{4})(?![\w-])/gu;
7
12
  const ISO_DATE = /\b(?<y>\d{4})-(?<m>\d{2})-(?<d>\d{2})\b/gu;
8
13
  const DAY_BEFORE = /(?<d>\d{1,2})(?:st|nd|rd|th)? $/u;
9
14
  const DAY_YEAR_AFTER = /^ (?<d>\d{1,2})(?:st|nd|rd|th)?,? (?<y>\d{4})\b/u;
@@ -41,12 +46,21 @@ const monthYear = (text: string, at: number, end: number, month: number): Mentio
41
46
  * The month is found first and its neighbours read with anchored patterns, never one long alternation.
42
47
  */
43
48
  const namedDate = (text: string, match: RegExpExecArray): Mention | undefined => {
44
- const month = MONTHS.indexOf((match.groups?.["month"] ?? "").toLowerCase()) + 1;
49
+ const word = (match.groups?.["month"] ?? "").toLowerCase();
50
+ const full = MONTHS.indexOf(word) + 1;
51
+ const month = full === 0 ? (ABBREVIATED.get(word) ?? 0) : full;
45
52
  if (month === 0) return undefined;
46
- const end = match.index + match[0].length;
53
+ const end = match.index + match[0].length + (full === 0 && text.charAt(match.index + match[0].length) === "." ? 1 : 0);
47
54
  return monthDayYear(text, match.index, end, month) ?? monthYear(text, match.index, end, month);
48
55
  };
49
56
 
57
+ const hyphenatedDate = (match: RegExpExecArray): Mention[] => {
58
+ const month = ABBREVIATED.get((match.groups?.["month"] ?? "").toLowerCase());
59
+ if (month === undefined) return [];
60
+ const value = `${match.groups?.["y"] ?? ""}-${pad(String(month))}-${pad(match.groups?.["d"] ?? "")}`;
61
+ return [{ start: match.index, end: match.index + match[0].length, attrs: { value } }];
62
+ };
63
+
50
64
  const isoDate = (match: RegExpExecArray): Mention => ({
51
65
  start: match.index,
52
66
  end: match.index + match[0].length,
@@ -88,6 +102,10 @@ const withWeekday = (text: string, date: Mention): Mention => {
88
102
  };
89
103
 
90
104
  export const dates = (text: string): Mention[] =>
91
- [...[...text.matchAll(MONTH_WORD)].flatMap((match) => namedDate(text, match) ?? []), ...[...text.matchAll(ISO_DATE)].map(isoDate)]
105
+ [
106
+ ...[...text.matchAll(MONTH_WORD)].flatMap((match) => namedDate(text, match) ?? []),
107
+ ...[...text.matchAll(HYPHENATED_DATE)].flatMap(hyphenatedDate),
108
+ ...[...text.matchAll(ISO_DATE)].map(isoDate),
109
+ ]
92
110
  .toSorted((left, right) => left.start - right.start)
93
111
  .map((date) => withWeekday(text, date));
@@ -0,0 +1,16 @@
1
+ // A word written in capitals for emphasis (the office will NEVER call you), told from an acronym by the dictionary.
2
+
3
+ const CAPITALS = /^\p{Lu}{2,}$/u;
4
+
5
+ const ADVERB_TAGS: ReadonlySet<string> = new Set(["RB", "RBR", "RBS"]);
6
+
7
+ /**
8
+ * The dictionary knows the lower-case word, and only as an adverb (never, always, also). An acronym names a thing, so a word
9
+ * with no noun or adjective reading cannot be one. A word that is also a noun or an adjective (fast, eagle, cheese) is kept:
10
+ * FAST and EAGLE are a telescope and a simulation, and the context tagger cannot tell them from emphasis.
11
+ */
12
+ export const isEmphasisedAdverb = (surface: string, tagsOf: (word: string) => readonly string[] | undefined): boolean => {
13
+ if (!CAPITALS.test(surface)) return false;
14
+ const tags = tagsOf(surface.toLowerCase()) ?? [];
15
+ return tags.length > 0 && tags.every((tag) => ADVERB_TAGS.has(tag));
16
+ };
package/src/index.ts CHANGED
@@ -1,14 +1,19 @@
1
1
  import { loadLexicons } from "./lexicons.ts";
2
2
  import { unmarkNumberStops } from "./number-stop.ts";
3
+ import { labelStops, unmarkLabelStops } from "./label-stop.ts";
3
4
  import { sentenceSpans } from "./sentence-split.ts";
4
5
  import { splitAtQuotedStops } from "./quoted-stop.ts";
5
6
  import { reattachClosingQuotes } from "./closing-quote.ts";
6
7
  import { structure } from "./structure.ts";
7
8
  import { isReady, prepare, tokenize } from "./pos.ts";
8
- import type { AdapterNeeds, LanguageAdapter, Segmentation, Sentence } from "chaffjs/plugin";
9
+ import { isJapaneseRun } from "./japanese-run.ts";
10
+ import type { AdapterNeeds, EmbeddedLanguage, LanguageAdapter, Segmentation, Sentence, Span } from "chaffjs/plugin";
9
11
 
10
12
  // chaff からは型だけを取る。実行時の値依存を作らない。アダプタは単体で動く。
11
13
 
14
+ const LEXICONS = loadLexicons();
15
+ const LABEL_STOPS = labelStops((LEXICONS["abbreviated-label"] ?? []).map((entry) => entry.pattern));
16
+
12
17
  const LATIN_LETTER = /[a-z]/giu;
13
18
  const COUNTABLE = /\S/gu;
14
19
 
@@ -25,9 +30,13 @@ const withTokens = (sentence: Sentence): Sentence => {
25
30
  };
26
31
  };
27
32
 
33
+ const JAPANESE: EmbeddedLanguage = { id: "ja", lengthUnit: "char" };
34
+
35
+ const withLanguage = (text: string, span: Span): Sentence => (isJapaneseRun(text) ? { span, text, embeddedLanguage: JAPANESE } : { span, text });
36
+
28
37
  /**
29
38
  * 英語は sentence-splitter の既定にほぼ任せる。"Dr." "e.g." "U.S." "$3.50" を
30
- * いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませること、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
39
+ * いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませることと、番号の前の略した名前(FIG. 1、Vol. XLIII)の点で切らないこと、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
31
40
  */
32
41
  export const adapter: LanguageAdapter = {
33
42
  kind: "language",
@@ -51,11 +60,11 @@ export const adapter: LanguageAdapter = {
51
60
  if (total === 0) return 0;
52
61
  return [...source.matchAll(LATIN_LETTER)].length / total;
53
62
  },
54
- lexicons: loadLexicons(),
63
+ lexicons: LEXICONS,
55
64
  structure,
56
65
  segment: (text: string): Segmentation => {
57
- const quotedStops = sentenceSpans(unmarkNumberStops(text)).flatMap((span) => splitAtQuotedStops(text, span));
58
- const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => ({ span, text: text.slice(span.start, span.end) }));
66
+ const quotedStops = sentenceSpans(unmarkNumberStops(unmarkLabelStops(text, LABEL_STOPS))).flatMap((span) => splitAtQuotedStops(text, span));
67
+ const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => withLanguage(text.slice(span.start, span.end), span));
59
68
  return { sentences: isReady() ? sentences.map(withTokens) : sentences };
60
69
  },
61
70
  };
@@ -0,0 +1,9 @@
1
+ /**
2
+ * A Japanese sentence in an English document (a quoted notice, a bilingual title): it has kana, and kana or kanji make
3
+ * up at least half of its letters (digits and symbols are not counted). Kanji alone do not count, as a Chinese name in English text is not Japanese.
4
+ */
5
+ const KANA = /[\p{Script=Hiragana}\p{Script=Katakana}]/u;
6
+ const JAPANESE = /[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}]/gu;
7
+ const LETTER = /\p{L}/gu;
8
+
9
+ export const isJapaneseRun = (text: string): boolean => KANA.test(text) && [...text.matchAll(JAPANESE)].length * 2 >= [...text.matchAll(LETTER)].length;
@@ -0,0 +1,25 @@
1
+ import { escapeRegExp } from "./regexp.ts";
2
+
3
+ /**
4
+ * sentence-splitter ends a sentence at the full stop of a label before its number: "FIG." and "1 illustrates …" were two
5
+ * sentences, and so were "Vol." and "XLIII (1979)". Where a number follows, that full stop is replaced with a letter
6
+ * before splitting. The length does not change, so the spans fit the original text.
7
+ * A lone "I" after a label is the pronoun ("He said No. I left."), and a word ("the last Fig. The tree") is not a number.
8
+ */
9
+
10
+ /** Built once from the lexicon; undefined when no label ends in a full stop. */
11
+ export type LabelStops = RegExp | undefined;
12
+
13
+ const NUMBER_AFTER = String.raw`(?=\s+(?:\d|(?:[IVXLCDM]{2,}|[VXLCDM])(?![\p{L}\p{N}_])))`;
14
+ const PLAIN_LETTER = "n";
15
+
16
+ /** The labels as written and in capitals, without their full stop ("Fig." → Fig, FIG). */
17
+ export const labelStops = (labels: readonly string[]): LabelStops => {
18
+ const stems = labels.filter((label) => label.endsWith(".")).flatMap((label) => [label.slice(0, -1), label.slice(0, -1).toUpperCase()]);
19
+ if (stems.length === 0) return undefined;
20
+ return new RegExp(String.raw`(?<![\p{L}\p{N}_.])(?:${[...new Set(stems)].map(escapeRegExp).join("|")})\.${NUMBER_AFTER}`, "gu");
21
+ };
22
+
23
+ /** The text with the full stop of every label before a number replaced by a letter. */
24
+ export const unmarkLabelStops = (text: string, stops: LabelStops): string =>
25
+ stops === undefined ? text : text.replace(stops, (label: string) => `${label.slice(0, -1)}${PLAIN_LETTER}`);