@chaffjs/lang-en 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/apostrophe.d.ts +2 -0
  2. package/dist/apostrophe.d.ts.map +1 -0
  3. package/dist/apostrophe.js +8 -0
  4. package/dist/apostrophe.js.map +1 -0
  5. package/dist/citation.d.ts +2 -0
  6. package/dist/citation.d.ts.map +1 -1
  7. package/dist/citation.js +37 -12
  8. package/dist/citation.js.map +1 -1
  9. package/dist/closing-quote.d.ts +6 -0
  10. package/dist/closing-quote.d.ts.map +1 -0
  11. package/dist/closing-quote.js +50 -0
  12. package/dist/closing-quote.js.map +1 -0
  13. package/dist/dates.js +1 -1
  14. package/dist/dates.js.map +1 -1
  15. package/dist/index.d.ts +1 -1
  16. package/dist/index.d.ts.map +1 -1
  17. package/dist/index.js +5 -2
  18. package/dist/index.js.map +1 -1
  19. package/dist/item-style.d.ts +29 -0
  20. package/dist/item-style.d.ts.map +1 -0
  21. package/dist/item-style.js +124 -0
  22. package/dist/item-style.js.map +1 -0
  23. package/dist/lexicons.d.ts.map +1 -1
  24. package/dist/lexicons.js +5 -6
  25. package/dist/lexicons.js.map +1 -1
  26. package/dist/long-runs.d.ts +6 -0
  27. package/dist/long-runs.d.ts.map +1 -0
  28. package/dist/long-runs.js +6 -0
  29. package/dist/long-runs.js.map +1 -0
  30. package/dist/pos.d.ts.map +1 -1
  31. package/dist/pos.js +40 -6
  32. package/dist/pos.js.map +1 -1
  33. package/dist/proper-noun.d.ts +21 -0
  34. package/dist/proper-noun.d.ts.map +1 -1
  35. package/dist/proper-noun.js +37 -0
  36. package/dist/proper-noun.js.map +1 -1
  37. package/dist/quoted-stop.d.ts +13 -0
  38. package/dist/quoted-stop.d.ts.map +1 -0
  39. package/dist/quoted-stop.js +45 -0
  40. package/dist/quoted-stop.js.map +1 -0
  41. package/dist/sentence-split.d.ts +2 -0
  42. package/dist/sentence-split.d.ts.map +1 -1
  43. package/dist/sentence-split.js +4 -2
  44. package/dist/sentence-split.js.map +1 -1
  45. package/dist/structure.d.ts.map +1 -1
  46. package/dist/structure.js +26 -67
  47. package/dist/structure.js.map +1 -1
  48. package/lexicons/common-acronym.yaml +4 -0
  49. package/lexicons/date-time-unit.yaml +6 -0
  50. package/lexicons/emphasis-word.yaml +1 -0
  51. package/lexicons/excessive-hedging.yaml +6 -0
  52. package/lexicons/fixed-phrase.yaml +14 -0
  53. package/lexicons/hedge-frame.yaml +9 -0
  54. package/lexicons/hedge-scope.yaml +7 -0
  55. package/lexicons/honorific.yaml +15 -0
  56. package/lexicons/http-method.yaml +14 -0
  57. package/lexicons/participle-word.yaml +8 -0
  58. package/lexicons/quantity-noun.yaml +6 -0
  59. package/lexicons/relative-word.yaml +10 -0
  60. package/lexicons/subject-pronoun.yaml +13 -0
  61. package/lexicons/superlative-amount.yaml +7 -0
  62. package/lexicons/superlative-bound.yaml +8 -0
  63. package/package.json +1 -1
  64. package/src/apostrophe.ts +8 -0
  65. package/src/citation.ts +41 -13
  66. package/src/closing-quote.ts +52 -0
  67. package/src/dates.ts +1 -1
  68. package/src/index.ts +5 -2
  69. package/src/item-style.ts +130 -0
  70. package/src/lexicons.ts +9 -8
  71. package/src/long-runs.ts +6 -0
  72. package/src/pos.ts +47 -6
  73. package/src/proper-noun.ts +43 -0
  74. package/src/quoted-stop.ts +52 -0
  75. package/src/sentence-split.ts +4 -2
  76. package/src/structure.ts +25 -69
@@ -0,0 +1,14 @@
1
+ # Fixed phrases, mostly borrowed, that open with a word the tagger reads as an article and act as one adjective or
2
+ # adverb: "the a priori approach", "an a fortiori case", "the a la carte menu". The "a" belongs to the phrase, so an
3
+ # article before it is not a doubled article. The tagger gives no sign of a foreign word here ("priori" is a noun to
4
+ # it), so the phrases are listed.
5
+ id: fixed-phrase
6
+ language: en
7
+ entries:
8
+ - pattern: a priori
9
+ - pattern: a posteriori
10
+ - pattern: a fortiori
11
+ - pattern: a la carte
12
+ - pattern: a la mode
13
+ - pattern: a cappella
14
+ - pattern: a capella
@@ -0,0 +1,9 @@
1
+ # Words that soften a claim without hedging it on their own: "may" is often permission or plain possibility.
2
+ # Next to a hedge in the same sentence they stack ("may possibly", "it could perhaps be argued").
3
+ # Used only by excessive-hedging's count of hedges in one sentence, never by its density.
4
+ id: hedge-frame
5
+ language: en
6
+ entries:
7
+ - pattern: may
8
+ - pattern: might
9
+ - pattern: could
@@ -0,0 +1,7 @@
1
+ # Entries of excessive-hedging that say where a claim holds, not how sure the writer is.
2
+ # "In some cases, the API may accept the key as a parameter" states a fact about some cases.
3
+ # They count towards the density, but never towards hedges stacked in one sentence.
4
+ id: hedge-scope
5
+ language: en
6
+ entries:
7
+ - pattern: in some cases
@@ -0,0 +1,15 @@
1
+ # Honorifics a transcript sets before a speaker's surname, which it prints in capitals (Mr. HAWLEY., Mrs. CAPITO.,
2
+ # Mr BLAKE in British style). A word in capitals right after one of these is a name, not an acronym. Matched as written.
3
+ id: honorific
4
+ language: en
5
+ entries:
6
+ - pattern: "Mr."
7
+ - pattern: "Mrs."
8
+ - pattern: "Ms."
9
+ - pattern: "Dr."
10
+ - pattern: "Mr"
11
+ - pattern: "Mrs"
12
+ - pattern: "Ms"
13
+ - pattern: "Dr"
14
+ - pattern: "Miss"
15
+ - pattern: "Madam"
@@ -0,0 +1,14 @@
1
+ # Names of HTTP request methods (RFC 9110, and PATCH from RFC 5789). They are written in capitals but are words, not
2
+ # abbreviations, so there is nothing to expand. Matched as written.
3
+ id: http-method
4
+ language: en
5
+ entries:
6
+ - pattern: GET
7
+ - pattern: HEAD
8
+ - pattern: POST
9
+ - pattern: PUT
10
+ - pattern: DELETE
11
+ - pattern: CONNECT
12
+ - pattern: OPTIONS
13
+ - pattern: TRACE
14
+ - pattern: PATCH
@@ -0,0 +1,8 @@
1
+ # Words that open a participle phrase after a comma but that the tagger reads as another part of speech:
2
+ # "It comes from Latin, meaning ship or boat." oxford-comma-consistency reads them as participles, so the phrase
3
+ # they open is not an item of the list before it. Words the tagger already reads as -ing verbs (including,
4
+ # regarding) need no entry.
5
+ id: participle-word
6
+ language: en
7
+ entries:
8
+ - pattern: meaning
@@ -0,0 +1,6 @@
1
+ # Nouns that name a measured quantity. A superlative joined to such a noun with no space names the quantity (最大風速
2
+ # in Japanese) rather than claiming anything is the greatest. English writes a space between a superlative and its noun,
3
+ # so there are none; the list exists so both languages name the same lists.
4
+ id: quantity-noun
5
+ language: en
6
+ entries: []
@@ -0,0 +1,10 @@
1
+ # Words that open a relative clause after a noun ("the best option which we chose", "the threats that we have seen").
2
+ # A superlative whose noun phrase such a clause restricts says what it is the most of.
3
+ # A language without this list puts no clause after a noun, and unqualified-superlative reads no clause or participle there.
4
+ id: relative-word
5
+ language: en
6
+ entries:
7
+ - pattern: that
8
+ - pattern: which
9
+ - pattern: who
10
+ - pattern: whom
@@ -0,0 +1,13 @@
1
+ # Pronouns that can be the subject of a clause ("the best we have measured", "the best option they could find").
2
+ # A reflexive or emphatic pronoun right after a noun ("The best players themselves get a bonus") is not a subject.
3
+ # A language without this list reads no clause that starts with a pronoun after a superlative.
4
+ id: subject-pronoun
5
+ language: en
6
+ entries:
7
+ - pattern: i
8
+ - pattern: you
9
+ - pattern: he
10
+ - pattern: she
11
+ - pattern: it
12
+ - pattern: we
13
+ - pattern: they
@@ -0,0 +1,7 @@
1
+ # Superlatives that also name an amount. With a noun right after them and nothing else joined to that noun ("the most
2
+ # work", "the most students", "for the most part"), they say which has the largest amount, not that anything is best.
3
+ # A language without this list reads no superlative as an amount.
4
+ id: superlative-amount
5
+ language: en
6
+ entries:
7
+ - pattern: the most
@@ -0,0 +1,8 @@
1
+ # Adjectives that bound a superlative to what can be had: right after it ("the best possible outcome") or after its noun
2
+ # ("the best education possible", "the best evidence available"). The superlative is the most of what is possible.
3
+ # A language without this list reads no superlative as bounded.
4
+ id: superlative-bound
5
+ language: en
6
+ entries:
7
+ - pattern: possible
8
+ - pattern: available
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@chaffjs/lang-en",
3
- "version": "0.12.0",
3
+ "version": "0.14.0",
4
4
  "description": "English language adapter for chaff",
5
5
  "license": "MIT",
6
6
  "author": "isamu",
@@ -0,0 +1,8 @@
1
+ /**
2
+ * 解析器(wink)は ' しかアポストロフィと読まず、that’s を that / ’ / s に割る。’ を ' に置き換えてから渡す。
3
+ * ’ は閉じの一重引用符でもあるので、字と字に挟まれたとき(don’t, team’s, 1990’s)だけ置き換える。閉じの引用符の後ろに字は続かない。
4
+ * ʼ(U+02BC)は引用符には使わないので、いつも置き換える。一字を一字に置き換えるので、位置は本文のまま。
5
+ */
6
+ const APOSTROPHE = /(?<=[\p{L}\p{N}])’(?=\p{L})|ʼ/gu;
7
+
8
+ export const straightApostrophes = (text: string): string => text.replace(APOSTROPHE, "'");
package/src/citation.ts CHANGED
@@ -4,8 +4,11 @@ const OF = /^,? of (?:the |that |those )?/u;
4
4
  /** "section 4(2)(a) (exception to liability …) of the Damages (Scotland) Act 2011": the gloss sits between the number and the name. */
5
5
  const GLOSS = /^ \([^()]{1,100}\)/u;
6
6
  const CONNECTOR = /^(?:,? (?:to|and|or)|,) /u;
7
- /** "9", "29(2)", "(g)", and a roman "V" for a list of Articles. The roman numeral must end the word: "VIII", not "Vendor". */
8
- const LISTED_NUMBERS = [/^\d{1,3}[A-Z]{0,2}(?:\([a-z0-9]{1,4}\))*/u, /^(?:\([a-z0-9]{1,4}\))+/u, /^[IVXLC]{1,7}\b/u];
7
+ /**
8
+ * "9", "29(2)", "(g)", and a roman "V" for a list of Articles. The roman numeral must end the word: "VIII", not "Vendor".
9
+ * A dotted "310.5" is not read: its "310" alone is not the section the list names.
10
+ */
11
+ const LISTED_NUMBERS = [/^\d{1,3}(?!\.?\d)[A-Z]{0,2}(?:\([a-z0-9]{1,4}\))*/u, /^(?:\([a-z0-9]{1,4}\))+/u, /^[IVXLC]{1,7}\b/u];
9
12
 
10
13
  const listedNumber = (text: string): string | undefined => LISTED_NUMBERS.map((pattern) => pattern.exec(text)?.[0]).find((found) => found !== undefined);
11
14
 
@@ -48,13 +51,25 @@ export const listMembers = (rest: string, plural: boolean): ListMember[] =>
48
51
  * word of a contract ("[Company]", "[Reserved]") is not taken for another document and still has to exist here.
49
52
  */
50
53
  const TAG = "(?<tag>[A-Z][A-Z0-9]{1,30})";
51
- const TAG_AFTER = new RegExp(`^\\[${TAG}\\]`, "u");
54
+ /**
55
+ * "[HTTP-CACHING]" is written like a contract's placeholder "[BUYER-1]". Only the document tells them apart, by listing
56
+ * the tag or not, so this tag is a candidate that the core checks against the document (attrs.citedTag).
57
+ */
58
+ const HYPHENATED_TAG = "(?<tag>[A-Z][A-Z0-9]{0,30}(?:-[A-Z0-9]{1,30}){1,4})";
59
+ const tagAfter = (tag: string): RegExp => new RegExp(`^\\[${tag}\\]`, "u");
52
60
  /** "[HTTP], Section 12.1": the tag written just before the reference. */
53
- const TAG_BEFORE = new RegExp(`\\[${TAG}\\],?\\s?$`, "u");
61
+ const tagBefore = (tag: string): RegExp => new RegExp(`\\[${tag}\\],?\\s?$`, "u");
62
+ const TAG_AFTER = tagAfter(TAG);
63
+ const TAG_BEFORE = tagBefore(TAG);
64
+ const HYPHENATED_AFTER = tagAfter(HYPHENATED_TAG);
65
+ const HYPHENATED_BEFORE = tagBefore(HYPHENATED_TAG);
66
+ const TAG_REACH = 40;
67
+
68
+ const tagEndingAt = (pattern: RegExp, text: string, start: number): string | undefined =>
69
+ pattern.exec(text.slice(Math.max(0, start - TAG_REACH), start))?.groups?.["tag"];
54
70
 
55
71
  /** The document cited by a tag just before a reference, as in "see [HTTP], Section 12.1". */
56
- export const citedDocumentBefore = (text: string, start: number): string | undefined =>
57
- TAG_BEFORE.exec(text.slice(Math.max(0, start - 40), start))?.groups?.["tag"];
72
+ export const citedDocumentBefore = (text: string, start: number): string | undefined => tagEndingAt(TAG_BEFORE, text, start);
58
73
 
59
74
  const CAPITALISED = /^[A-Z][\w'’-]*/u;
60
75
  /**
@@ -81,20 +96,33 @@ const titleWords = (rest: string): string[] => {
81
96
  return words;
82
97
  };
83
98
 
99
+ /** What follows the "of" after a reference and its list, or undefined when no "of" follows. */
100
+ const afterOf = (text: string, end: number): string | undefined => {
101
+ const rest = text.slice(end, end + 200);
102
+ const afterList = listEnd(rest);
103
+ const afterGloss = afterList + (GLOSS.exec(rest.slice(afterList))?.[0].length ?? 0);
104
+ const of = OF.exec(rest.slice(afterGloss));
105
+ return of === null ? undefined : rest.slice(afterGloss + of[0].length);
106
+ };
107
+
84
108
  /**
85
109
  * The document named right after a reference, or undefined when the reference is into this document.
86
110
  * "of this Agreement" and "of the Agreement" are this document; "of the Master Agreement" is another.
87
111
  */
88
112
  export const citedDocumentAfter = (text: string, end: number): string | undefined => {
89
- const rest = text.slice(end, end + 200);
90
- const afterList = listEnd(rest);
91
- const listed = afterList + (GLOSS.exec(rest.slice(afterList))?.[0].length ?? 0);
92
- const of = OF.exec(rest.slice(listed));
93
- if (of === null) return undefined;
94
- const tag = TAG_AFTER.exec(rest.slice(listed + of[0].length))?.groups?.["tag"];
113
+ const named = afterOf(text, end);
114
+ if (named === undefined) return undefined;
115
+ const tag = TAG_AFTER.exec(named)?.groups?.["tag"];
95
116
  if (tag !== undefined) return tag;
96
- const words = titleWords(rest.slice(listed + of[0].length));
117
+ const words = titleWords(named);
97
118
  if (words.length === 0) return undefined;
98
119
  const name = words.join(" ");
99
120
  return words.length === 1 && SELF.has(name) ? undefined : name;
100
121
  };
122
+
123
+ /** A hyphenated tag right after a reference ("Section 4.2.3 of [HTTP-CACHING]") or just before it ("[HTTP-CACHING], Section 4"). */
124
+ export const hyphenatedTagAround = (text: string, start: number, end: number): string | undefined => {
125
+ const named = afterOf(text, end);
126
+ const following = named === undefined ? undefined : HYPHENATED_AFTER.exec(named)?.groups?.["tag"];
127
+ return following ?? tagEndingAt(HYPHENATED_BEFORE, text, start);
128
+ };
@@ -0,0 +1,52 @@
1
+ import type { Span } from "chaffjs/plugin";
2
+ import { NEXT_SENTENCE_HEAD } from "./quoted-stop.ts";
3
+
4
+ /**
5
+ * sentence-splitter は疑問符・感嘆符の後の曲がった閉じ引用符(“Is it done?” Nobody …)と直線の一重引用符の手前で文を切る。
6
+ * 次の文が「”」で始まり、前の文は引用符が開いたまま終わる。
7
+ *
8
+ * 前の文が句点で終わり、次の文がそこから間を置かずに閉じ引用符(閉じ括弧も)で始まるとき、それを前の文の末尾へ戻す。
9
+ * 戻した後に空白と大文字が続けば文は二つのまま、何も続かなければ前の文だけ、
10
+ * 小文字・数字・句読点が続けば引用は文の途中にあるので一つの文につなぐ(直線の二重引用符で分割器がそうするのと同じ)。
11
+ * 開き引用符(“ ‘)は閉じの分類に入らない。語頭のアポストロフィ(’Tis ’90s)は後ろに字が続くので閉じと読まない。
12
+ */
13
+ const CLOSING_RUN = /^[\p{Pf}\p{Pe}"']+/u;
14
+ const WORD_CHARACTER = /^[\p{L}\p{N}]/u;
15
+ const STOP_AT_END = /[.?!]$/u;
16
+ // 語頭のアポストロフィの後の大文字(’Tis)も次の文の頭。
17
+ const NEXT_SENTENCE = new RegExp(String.raw`^(?:${NEXT_SENTENCE_HEAD}|\s+’\p{Lu})`, "u");
18
+ const LEADING_SPACE = /^\s*/u;
19
+
20
+ /** 文頭の閉じ引用符の並びの長さ。後ろに字が続けば開き引用符かアポストロフィなので 0。 */
21
+ export const closingRunLength = (sentence: string): number => {
22
+ const run = CLOSING_RUN.exec(sentence)?.[0].length ?? 0;
23
+ return run > 0 && !WORD_CHARACTER.test(sentence.slice(run)) ? run : 0;
24
+ };
25
+
26
+ const follows = (text: string, previous: Span, span: Span): boolean =>
27
+ previous.end === span.start && STOP_AT_END.test(text.slice(previous.start, previous.end));
28
+
29
+ /** previous と span を置き換える文。閉じ引用符が戻らなければ undefined。 */
30
+ const reattached = (text: string, previous: Span, span: Span): Span[] | undefined => {
31
+ const run = follows(text, previous, span) ? closingRunLength(text.slice(span.start, span.end)) : 0;
32
+ if (run === 0) return undefined;
33
+ const quoteEnd = span.start + run;
34
+ const rest = text.slice(quoteEnd, span.end);
35
+ if (rest.trim() === "") return [{ start: previous.start, end: quoteEnd }];
36
+ if (!NEXT_SENTENCE.test(rest)) return [{ start: previous.start, end: span.end }];
37
+ const next = quoteEnd + (LEADING_SPACE.exec(rest)?.[0].length ?? 0);
38
+ return [
39
+ { start: previous.start, end: quoteEnd },
40
+ { start: next, end: span.end },
41
+ ];
42
+ };
43
+
44
+ /** 文の span の並び(text 上の位置)で、文頭に取り残された閉じ引用符を前の文へ戻したもの。 */
45
+ export const reattachClosingQuotes = (text: string, spans: readonly Span[]): Span[] =>
46
+ spans.reduce<Span[]>((sentences, span) => {
47
+ const previous = sentences.at(-1);
48
+ const replaced = previous === undefined ? undefined : reattached(text, previous, span);
49
+ if (replaced === undefined) sentences.push(span);
50
+ else sentences.splice(-1, 1, ...replaced);
51
+ return sentences;
52
+ }, []);
package/src/dates.ts CHANGED
@@ -89,5 +89,5 @@ const withWeekday = (text: string, date: Mention): Mention => {
89
89
 
90
90
  export const dates = (text: string): Mention[] =>
91
91
  [...[...text.matchAll(MONTH_WORD)].flatMap((match) => namedDate(text, match) ?? []), ...[...text.matchAll(ISO_DATE)].map(isoDate)]
92
- .sort((left, right) => left.start - right.start)
92
+ .toSorted((left, right) => left.start - right.start)
93
93
  .map((date) => withWeekday(text, date));
package/src/index.ts CHANGED
@@ -1,6 +1,8 @@
1
1
  import { loadLexicons } from "./lexicons.ts";
2
2
  import { unmarkNumberStops } from "./number-stop.ts";
3
3
  import { sentenceSpans } from "./sentence-split.ts";
4
+ import { splitAtQuotedStops } from "./quoted-stop.ts";
5
+ import { reattachClosingQuotes } from "./closing-quote.ts";
4
6
  import { structure } from "./structure.ts";
5
7
  import { isReady, prepare, tokenize } from "./pos.ts";
6
8
  import type { AdapterNeeds, LanguageAdapter, Segmentation, Sentence } from "chaffjs/plugin";
@@ -25,7 +27,7 @@ const withTokens = (sentence: Sentence): Sentence => {
25
27
 
26
28
  /**
27
29
  * 英語は sentence-splitter の既定にほぼ任せる。"Dr." "e.g." "U.S." "$3.50" を
28
- * いずれも文末と誤認しない。前処理は、行の途中の番号を箇条書きと読ませないことだけ。spec §7.2。
30
+ * いずれも文末と誤認しない。前処理は行の途中の番号を箇条書きと読ませること、後処理は閉じ引用符の内側で閉じた文を切ることと、文頭に取り残された閉じ引用符を前の文へ戻すこと。spec §7.2。
29
31
  */
30
32
  export const adapter: LanguageAdapter = {
31
33
  kind: "language",
@@ -52,7 +54,8 @@ export const adapter: LanguageAdapter = {
52
54
  lexicons: loadLexicons(),
53
55
  structure,
54
56
  segment: (text: string): Segmentation => {
55
- const sentences: Sentence[] = sentenceSpans(unmarkNumberStops(text)).map((span) => ({ span, text: text.slice(span.start, span.end) }));
57
+ const quotedStops = sentenceSpans(unmarkNumberStops(text)).flatMap((span) => splitAtQuotedStops(text, span));
58
+ const sentences: Sentence[] = reattachClosingQuotes(text, quotedStops).map((span) => ({ span, text: text.slice(span.start, span.end) }));
56
59
  return { sentences: isReady() ? sentences.map(withTokens) : sentences };
57
60
  },
58
61
  };
@@ -0,0 +1,130 @@
1
+ import type { NumberedLine, NumberingContext } from "chaffjs/plugin";
2
+ import { parseRoman } from "./roman.ts";
3
+
4
+ /**
5
+ * An amendment inserts a subsection between two others and numbers it "(A1)" or "(2A)". It is a subsection, written
6
+ * outside the sequence: it has no ordinal, so "(1)" after "(A1)" is still the first.
7
+ */
8
+ export const INSERTED = "\\d{1,3}[A-Z]{1,2}|[A-Z]{1,2}\\d{1,3}";
9
+ const IS_INSERTED = new RegExp(`^(?:${INSERTED})$`, "u");
10
+ const MULTI_ROMAN = /^(?:ii|iii|iv|vi|vii|viii|ix)$/u;
11
+ const AMBIGUOUS = /^[ivx]$/u;
12
+ const DIGITS = /^\d+$/u;
13
+ const CAPITALS = /^[A-Z]+$/u;
14
+ const ONE_CAPITAL = /^[A-Z]$/u;
15
+
16
+ export type Style = "letter" | "roman" | "digit" | "capital" | "capital-roman";
17
+
18
+ /** 大文字で書いた同じ並び。"(B)" は英字の、"(II)" はローマ数字の大文字。 */
19
+ const CAPITAL_OF: Readonly<Record<"letter" | "roman", Style>> = { letter: "capital", roman: "capital-roman" };
20
+
21
+ /** 小文字にした番号を形で読む。一文字の "i" は、開いたときに付けた並びの位置で見分ける。ローマ数字なら 1、英字なら 9。 */
22
+ const shapeOf = (lower: string, ordinal: number | undefined): "letter" | "roman" => {
23
+ if (MULTI_ROMAN.test(lower)) return "roman";
24
+ return AMBIGUOUS.test(lower) && ordinal === parseRoman(lower) ? "roman" : "letter";
25
+ };
26
+
27
+ /** 開いている項目の書き方。"(ii)" はローマ数字、"(b)" は英字、"(B)" は大文字の英字。 */
28
+ export const styleOfOpen = (open: NumberedLine): Style | undefined => {
29
+ const inner = /^\((?<n>[A-Za-z0-9]{1,5})\)$/u.exec(open.label)?.groups?.["n"];
30
+ if (inner === undefined) return undefined;
31
+ if (DIGITS.test(inner) || IS_INSERTED.test(inner)) return "digit";
32
+ const shape = shapeOf(inner.toLowerCase(), open.ordinal);
33
+ return CAPITALS.test(inner) ? CAPITAL_OF[shape] : shape;
34
+ };
35
+
36
+ const LETTER_BEFORE_A = "a".charCodeAt(0) - 1;
37
+
38
+ /** "(b)" と "(B)" は 2 番目、"(ii)" と "(II)" も 2 番目。二文字以上の英字("(aa)")は並びが決まらないので付けない。 */
39
+ export const ordinalOf = (raw: string, style: Style): number | undefined => {
40
+ if (style === "digit") return IS_INSERTED.test(raw) ? undefined : Number(raw);
41
+ if (style === "roman" || style === "capital-roman") return parseRoman(raw);
42
+ return raw.length === 1 ? raw.toLowerCase().charCodeAt(0) - LETTER_BEFORE_A : undefined;
43
+ };
44
+
45
+ const styles = (context: NumberingContext): (Style | undefined)[] => context.open.map(styleOfOpen);
46
+
47
+ /** "(h)" の次の "(i)"、"(H)" の次の "(I)" は英字。同じ書き方で開いている英字の次の文字なら、ローマ数字とは読まない。 */
48
+ const follows = (raw: string, style: "letter" | "capital", context: NumberingContext): boolean =>
49
+ context.open.some((open, index) => styles(context)[index] === style && open.number.charCodeAt(0) + 1 === raw.charCodeAt(0));
50
+
51
+ /**
52
+ * 英字の並びは (a) から始まるので、"(a)" や "(1)" のすぐ下に来た "(i)" は一段深いローマ数字。
53
+ * 米国の規則は (a)(1)(i) の順に下る。見出しのすぐ下の "(i)" は、どちらとも決まらないので英字。
54
+ */
55
+ const OPENS_ROMAN: ReadonlySet<Style | undefined> = new Set(["letter", "digit"]);
56
+
57
+ /** "(I)" は米国の法典で (i) の一段下の大文字ローマ数字。"(H)" の次なら英字。 */
58
+ const capitalRoman = (raw: string, open: readonly (Style | undefined)[], context: NumberingContext): boolean => {
59
+ const lower = raw.toLowerCase();
60
+ if (MULTI_ROMAN.test(lower)) return open.includes("capital-roman");
61
+ if (!AMBIGUOUS.test(lower) || follows(raw, "capital", context)) return false;
62
+ return open.includes("capital-roman") || (raw === "I" && open.at(-1) === "roman");
63
+ };
64
+
65
+ /**
66
+ * 米国の規則は (a)(1)(i)(A) と下る。大文字の "(A)" は、開いているローマ数字のすぐ下で始まるか、開いている大文字の続きのときだけ項目。
67
+ * 英国の法令や契約書が一段目に使う "(A)" は、何の下とも決まらないので本文のまま読む。
68
+ */
69
+ const capitalStyleOf = (raw: string, context: NumberingContext): Style | undefined => {
70
+ const open = styles(context);
71
+ if (capitalRoman(raw, open, context)) return "capital-roman";
72
+ if (!ONE_CAPITAL.test(raw)) return undefined;
73
+ return open.includes("capital") || (raw === "A" && open.at(-1) === "roman") ? "capital" : undefined;
74
+ };
75
+
76
+ /**
77
+ * "(i)" is a roman numeral right under "(a)" or "(1)", or when a roman list is already open; the letter i otherwise.
78
+ * A capital "(A)" is an item only where a US regulation puts it, under a roman item; undefined elsewhere.
79
+ */
80
+ export const styleOf = (raw: string, context: NumberingContext): Style | undefined => {
81
+ if (DIGITS.test(raw) || IS_INSERTED.test(raw)) return "digit";
82
+ if (CAPITALS.test(raw)) return capitalStyleOf(raw, context);
83
+ if (MULTI_ROMAN.test(raw)) return "roman";
84
+ if (!AMBIGUOUS.test(raw) || follows(raw, "letter", context)) return "letter";
85
+ const open = styles(context);
86
+ return open.includes("roman") || OPENS_ROMAN.has(open.at(-1)) ? "roman" : "letter";
87
+ };
88
+
89
+ /** The nearest open item written the same way: "(b)" closes "(i)" and sits beside "(a)". */
90
+ const siblingOf = (style: Style, context: NumberingContext): NumberedLine | undefined => {
91
+ const open = styles(context);
92
+ return context.open.findLast((_item, index) => open[index] === style);
93
+ };
94
+
95
+ const IS_CAPITAL: ReadonlySet<Style | undefined> = new Set(["capital", "capital-roman"]);
96
+
97
+ /**
98
+ * "(A)" の下の "(1)" や "(i)"、"(i)" の下の "(A)" は、上に開いている同じ書き方の項目の続きではなく、一段深い並びの始まり。
99
+ * 米国の規則は (a)(1)(i)(A) の下を、もう一度 (1) や (i) で数える。
100
+ */
101
+ const startsBelow = (style: Style, ordinal: number | undefined, innermost: Style | undefined): boolean =>
102
+ ordinal === 1 && style !== innermost && (IS_CAPITAL.has(innermost) || (IS_CAPITAL.has(style) && innermost === "roman"));
103
+
104
+ /**
105
+ * A sibling has the depth of the open item written the same way. A new way of numbering goes one deeper than whatever
106
+ * is open, and so does a first item right under a capital, or a first capital right under a roman item.
107
+ */
108
+ export const depthFor = (style: Style, context: NumberingContext, ordinal?: number): number => {
109
+ const innermost = context.open.at(-1);
110
+ const deeper = (innermost?.depth ?? 0) + 1;
111
+ if (innermost !== undefined && startsBelow(style, ordinal, styleOfOpen(innermost))) return deeper;
112
+ return siblingOf(style, context)?.depth ?? deeper;
113
+ };
114
+
115
+ /**
116
+ * 参照の "(A)" や "(I)" は、ローマ数字の "(iii)" の次に書かれたときだけ番地の続き。木も、大文字はローマ数字の下でしか読まない。
117
+ * "section 5(A)" の "(A)" は番地に入れず、これまでどおり "section 5" を引く。"(h)(i)" や番号のすぐ後の "(i)" は、木と同じく英字と読む。
118
+ */
119
+ export const continuesAddress = (part: string, before: readonly string[]): boolean => {
120
+ if (!CAPITALS.test(part)) return true;
121
+ const [previous, earlier] = [before.at(-1) ?? "", before.at(-2) ?? ""];
122
+ if (MULTI_ROMAN.test(previous)) return true;
123
+ return AMBIGUOUS.test(previous) && earlier !== "" && earlier.charCodeAt(0) + 1 !== previous.charCodeAt(0);
124
+ };
125
+
126
+ /** 番号だけの行 "(5)" は、開いている "(4)" の次のときだけ項目。本文は次の行から始まる。 */
127
+ export const continuesOpen = (style: Style, ordinal: number | undefined, context: NumberingContext): boolean => {
128
+ const previous = siblingOf(style, context)?.ordinal;
129
+ return ordinal !== undefined && previous !== undefined && ordinal === previous + 1;
130
+ };
package/src/lexicons.ts CHANGED
@@ -27,11 +27,12 @@ const toEntry = (raw: unknown): LexiconEntry | undefined => {
27
27
  * 新しい言語のサポートは、ここを書くところから始まる。
28
28
  */
29
29
  export const loadLexicons = (dir: string = DIR): Record<string, Lexicon> =>
30
- readdirSync(dir)
31
- .filter((file) => file.endsWith(".yaml"))
32
- .reduce<Record<string, Lexicon>>((acc, file) => {
33
- const raw: unknown = parse(readFileSync(join(dir, file), "utf8"));
34
- if (!isRecord(raw) || typeof raw["id"] !== "string" || !Array.isArray(raw["entries"])) return acc;
35
- const entries = raw["entries"].map(toEntry).filter((entry) => entry !== undefined);
36
- return { ...acc, [raw["id"]]: entries };
37
- }, {});
30
+ Object.fromEntries(
31
+ readdirSync(dir)
32
+ .filter((file) => file.endsWith(".yaml"))
33
+ .flatMap((file): [string, Lexicon][] => {
34
+ const raw: unknown = parse(readFileSync(join(dir, file), "utf8"));
35
+ if (!isRecord(raw) || typeof raw["id"] !== "string" || !Array.isArray(raw["entries"])) return [];
36
+ return [[raw["id"], raw["entries"].map(toEntry).filter((entry) => entry !== undefined)]];
37
+ }),
38
+ );
@@ -0,0 +1,6 @@
1
+ /**
2
+ * 空白の無いまま limit 文字を超える並びを、同じ長さの空白で覆う。語ではなく(base64、ハッシュ、区切りの無い記号の列)、
3
+ * 解析器(wink)はその長さの二乗で遅くなる。長さを変えないので、残りの語の位置はそのまま。
4
+ */
5
+ export const blankLongRuns = (text: string, limit: number): string =>
6
+ text.replace(new RegExp(`\\S{${String(limit + 1)},}`, "gu"), (run) => " ".repeat(run.length));
package/src/pos.ts CHANGED
@@ -1,7 +1,9 @@
1
1
  import { createRequire } from "node:module";
2
2
  import type { Token } from "chaffjs/plugin";
3
+ import { straightApostrophes } from "./apostrophe.ts";
3
4
  import { loadLexicons } from "./lexicons.ts";
4
- import { properNounChecked } from "./proper-noun.ts";
5
+ import { blankLongRuns } from "./long-runs.ts";
6
+ import { lowercasedAt, properNounChecked, rereadAt, sentenceInitialCommonWord } from "./proper-noun.ts";
5
7
  import { isStativeParticiple, stativeVocabulary } from "./stative-participle.ts";
6
8
 
7
9
  const require = createRequire(import.meta.url);
@@ -157,18 +159,43 @@ const determinerFeatures = (entry: Tagged): Features => {
157
159
  return entry.pos === "DT" && ARTICLES.has(entry.value.toLowerCase()) ? { features: { PronType: "Art" } } : {};
158
160
  };
159
161
 
160
- /** 過去分詞は VerbForm=Part。Based on the review, のような分詞の導入句を、命令形の並び(fix the parser, ship it)と見分ける。 */
162
+ /** 複数形の名詞は UD の Number=Plur。数の語がそれを数えていれば(five minutes)、one of のような言い回しではなく量。 */
163
+ const PLURAL_TAG = new Set(["NNS", "NNPS"]);
164
+
165
+ const nounOrDeterminerFeatures = (entry: Tagged): Features => (PLURAL_TAG.has(entry.pos) ? { features: { Number: "Plur" } } : determinerFeatures(entry));
166
+
167
+ /**
168
+ * 過去分詞は VerbForm=Part。Based on the review, のような分詞の導入句を、命令形の並び(fix the parser, ship it)と見分ける。
169
+ * -ing 形は VerbForm=Ger。解析器は動名詞と現在分詞を分けないので、過去分詞を見る判断(Part)には混ぜない。
170
+ */
161
171
  const featuresOf = (tagged: readonly Tagged[], at: number): Features => {
162
172
  const entry = tagged[at];
163
173
  if (entry === undefined) return {};
164
- if (entry.pos !== "VBN") return determinerFeatures(entry);
174
+ if (entry.pos === "VBG") return { features: { VerbForm: "Ger" } };
175
+ if (entry.pos !== "VBN") return nounOrDeterminerFeatures(entry);
165
176
  return { features: isPassive(tagged, at) ? { VerbForm: "Part", Voice: "Pass" } : { VerbForm: "Part" } };
166
177
  };
167
178
 
168
- const state: { ready: Tagger | undefined } = { ready: undefined };
179
+ /** 解析器が引く語彙(語 → Penn Treebank の品詞の並び)。wink-pos-tagger が自分の依存から読むものと同じ一つを、同じ場所から読む。 */
180
+ type Vocabulary = (word: string) => readonly string[] | undefined;
181
+
182
+ const isTags = (value: unknown): value is readonly string[] => Array.isArray(value) && value.every((tag) => typeof tag === "string");
183
+
184
+ const buildVocabulary = (): Vocabulary => {
185
+ const words: unknown = createRequire(require.resolve("wink-pos-tagger"))("wink-lexicon/src/lexicon.js");
186
+ if (!isRecord(words)) throw new Error("wink-lexicon の語彙が object ではありません");
187
+ return (word) => {
188
+ const tags = Object.hasOwn(words, word) ? words[word] : undefined;
189
+ return isTags(tags) ? tags : undefined;
190
+ };
191
+ };
192
+
193
+ const state: { ready: Tagger | undefined; vocabulary: Vocabulary } = { ready: undefined, vocabulary: () => undefined };
169
194
 
170
195
  export const prepare = (): void => {
171
- state.ready ??= build();
196
+ if (state.ready !== undefined) return;
197
+ state.ready = build();
198
+ state.vocabulary = buildVocabulary();
172
199
  };
173
200
 
174
201
  export const isReady = (): boolean => state.ready !== undefined;
@@ -195,8 +222,22 @@ const locate = (text: string, tagged: readonly Tagged[]): Token[] =>
195
222
  { tokens: [], cursor: 0 },
196
223
  ).tokens;
197
224
 
225
+ /** 英語の語はこれより長くならない。超える並びは語として読まない。 */
226
+ const RUN_LIMIT = 1000;
227
+
228
+ const tagged = (tagger: Tagger, text: string): readonly Tagged[] => toArray(callMethod(tagger, "tagSentence", [text])).filter(isTagged);
229
+
230
+ /** 文頭で大文字になっただけの普通の語を、小文字で書いたときの品詞に戻す。前後の語による判断も効くよう、文ごと解析し直す。 */
231
+ const withSentenceInitialCase = (tagger: Tagger, text: string, entries: readonly Tagged[]): readonly Tagged[] => {
232
+ const at = sentenceInitialCommonWord(entries, state.vocabulary);
233
+ const lowered = lowercasedAt(text, entries[at]?.value);
234
+ return lowered === undefined ? entries : rereadAt(entries, at, tagged(tagger, lowered));
235
+ };
236
+
198
237
  export const tokenize = (text: string): Token[] | undefined => {
199
238
  const tagger = state.ready;
200
239
  if (tagger === undefined) return undefined;
201
- return locate(text, toArray(callMethod(tagger, "tagSentence", [text])).filter(isTagged));
240
+ // 語は解析させた字(don't)で返す。語彙表の語と同じ字で比べられる。span は本文を指したまま。
241
+ const words = straightApostrophes(blankLongRuns(text, RUN_LIMIT));
242
+ return locate(words, withSentenceInitialCase(tagger, words, tagged(tagger, words)));
202
243
  };
@@ -19,3 +19,46 @@ export const properNounChecked = (surface: string, pos: string): string => {
19
19
  if (!LETTER.test(surface)) return nonWord(surface);
20
20
  return SMALL.test(surface) && !CAPITAL.test(surface) ? "NOUN" : "PROPN";
21
21
  };
22
+
23
+ /** 解析器が返す 1 語。pos は Penn Treebank。 */
24
+ type TaggedWord = { readonly value: string; readonly pos: string; readonly lemma?: string };
25
+
26
+ const PROPER_TAG = new Set(["NNP", "NNPS"]);
27
+
28
+ /** 頭の一字だけが大文字の語(Containers、Such)。文頭の大文字はこの形になる。API や iPhone は名前の書き方なので含めない。 */
29
+ const CAPITALISED = /^\p{Lu}\p{Ll}+$/u;
30
+
31
+ /** 解析器の語彙に、その語が普通の語として載っているか。一度でも固有名詞として載っていれば(may の May)、名前かもしれない。 */
32
+ const isCommonWord = (tags: readonly string[] | undefined): boolean => tags !== undefined && tags.length > 0 && !tags.some((tag) => PROPER_TAG.has(tag));
33
+
34
+ /**
35
+ * 文頭で大文字になっただけの普通の語の位置。無ければ -1。
36
+ * wink は大文字で始まる名詞と形容詞をすべて固有名詞にするので、Containers are … の Containers も Traditional servers … の Traditional も固有名詞になる。
37
+ * 文頭の大文字は英語の書き方で、名前の印ではない。解析器の語彙が小文字の形を普通の語として知っていれば、名前ではなくその語として読む。
38
+ * 語彙に無い語(Kubernetes、Congress)と、文の途中の大文字の語は名前のまま。
39
+ */
40
+ export const sentenceInitialCommonWord = (tagged: readonly TaggedWord[], tagsOf: (word: string) => readonly string[] | undefined): number => {
41
+ const at = tagged.findIndex((entry) => LETTER.test(entry.value));
42
+ const first = tagged[at];
43
+ if (first === undefined || !PROPER_TAG.has(first.pos) || !CAPITALISED.test(first.value)) return -1;
44
+ return isCommonWord(tagsOf(first.value.toLowerCase())) ? at : -1;
45
+ };
46
+
47
+ /** 文の中の最初の word を小文字にした文。word が文に無ければ undefined。 */
48
+ export const lowercasedAt = (text: string, word: string | undefined): string | undefined => {
49
+ const start = word === undefined ? -1 : text.indexOf(word);
50
+ if (word === undefined || start === -1) return undefined;
51
+ return `${text.slice(0, start)}${word.toLowerCase()}${text.slice(start + word.length)}`;
52
+ };
53
+
54
+ /**
55
+ * at の語の品詞と原形を、小文字にして解析し直した結果(again)から取る。表層は元のまま。
56
+ * 解析し直した文の同じ位置に、同じ語の小文字が無ければ、語の切り方が変わったので元のまま。
57
+ */
58
+ export const rereadAt = (entries: readonly TaggedWord[], at: number, again: readonly TaggedWord[]): readonly TaggedWord[] => {
59
+ const entry = entries[at];
60
+ const reread = again[at];
61
+ if (entry === undefined || reread === undefined || reread.value !== entry.value.toLowerCase()) return entries;
62
+ const word: TaggedWord = { value: entry.value, pos: reread.pos, ...(reread.lemma === undefined ? {} : { lemma: reread.lemma }) };
63
+ return entries.map((original, index) => (index === at ? word : original));
64
+ };