@dxos/nlp 0.9.1-staging.ee54ba693a → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/lib/chunk-align.mjs +56 -0
  2. package/dist/lib/chunk-align.mjs.map +1 -0
  3. package/dist/lib/index.mjs +59 -0
  4. package/dist/lib/index.mjs.map +1 -0
  5. package/dist/lib/testing.mjs +92 -0
  6. package/dist/lib/testing.mjs.map +1 -0
  7. package/dist/types/src/align.d.ts.map +1 -1
  8. package/dist/types/src/index.d.ts +1 -1
  9. package/dist/types/src/index.d.ts.map +1 -1
  10. package/dist/types/src/parse.d.ts +0 -2
  11. package/dist/types/src/parse.d.ts.map +1 -1
  12. package/dist/types/src/testing/index.d.ts +1 -1
  13. package/dist/types/src/testing/index.d.ts.map +1 -1
  14. package/dist/types/src/{stub.d.ts → testing/stub.d.ts} +1 -1
  15. package/dist/types/src/testing/stub.d.ts.map +1 -0
  16. package/dist/types/src/testing/stub.test.d.ts.map +1 -0
  17. package/dist/types/tsconfig.tsbuildinfo +1 -1
  18. package/package.json +4 -4
  19. package/src/align.ts +7 -1
  20. package/src/index.ts +1 -1
  21. package/src/parse.test.ts +2 -1
  22. package/src/parse.ts +0 -3
  23. package/src/testing/index.ts +1 -1
  24. package/src/{stub.ts → testing/stub.ts} +9 -5
  25. package/dist/lib/neutral/chunk-D4MHDU46.mjs +0 -156
  26. package/dist/lib/neutral/chunk-D4MHDU46.mjs.map +0 -7
  27. package/dist/lib/neutral/index.mjs +0 -78
  28. package/dist/lib/neutral/index.mjs.map +0 -7
  29. package/dist/lib/neutral/meta.json +0 -1
  30. package/dist/lib/neutral/testing/index.mjs +0 -7
  31. package/dist/lib/neutral/testing/index.mjs.map +0 -7
  32. package/dist/types/src/stub.d.ts.map +0 -1
  33. package/dist/types/src/stub.test.d.ts.map +0 -1
  34. /package/dist/types/src/{stub.test.d.ts → testing/stub.test.d.ts} +0 -0
  35. /package/src/{stub.test.ts → testing/stub.test.ts} +0 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dxos/nlp",
3
- "version": "0.9.1-staging.ee54ba693a",
3
+ "version": "0.11.0",
4
4
  "description": "Natural-language parsing: per-word UPOS token structure for intention/transcript analysis.",
5
5
  "homepage": "https://dxos.org",
6
6
  "bugs": "https://github.com/dxos/dxos/issues",
@@ -16,12 +16,12 @@
16
16
  ".": {
17
17
  "source": "./src/index.ts",
18
18
  "types": "./dist/types/src/index.d.ts",
19
- "default": "./dist/lib/neutral/index.mjs"
19
+ "import": "./dist/lib/index.mjs"
20
20
  },
21
21
  "./testing": {
22
22
  "source": "./src/testing/index.ts",
23
23
  "types": "./dist/types/src/testing/index.d.ts",
24
- "default": "./dist/lib/neutral/testing/index.mjs"
24
+ "import": "./dist/lib/testing.mjs"
25
25
  }
26
26
  },
27
27
  "types": "dist/types/src/index.d.ts",
@@ -31,7 +31,7 @@
31
31
  ],
32
32
  "dependencies": {
33
33
  "@effect/ai": "0.36.0",
34
- "@dxos/ai": "0.9.1-staging.ee54ba693a"
34
+ "@dxos/ai": "0.11.0"
35
35
  },
36
36
  "devDependencies": {
37
37
  "effect": "3.21.4"
package/src/align.ts CHANGED
@@ -22,14 +22,20 @@ export const assembleDocument = (sourceText: string, rawSentences: readonly RawS
22
22
  if (start < 0) {
23
23
  continue;
24
24
  }
25
+
25
26
  const end = start + text.length;
26
27
  tokens.push({ index: tokens.length, text, upos, start, end });
27
28
  cursor = end;
28
29
  }
30
+
29
31
  if (tokens.length > 0) {
30
32
  sentences.push({ index: sentenceIndex, start: tokens[0].start, end: tokens[tokens.length - 1].end, tokens });
31
33
  }
32
34
  });
33
35
 
34
- return { sourceHash: sourceHash(sourceText), sentences, timestamp: undefined };
36
+ return {
37
+ sourceHash: sourceHash(sourceText),
38
+ sentences,
39
+ timestamp: undefined,
40
+ };
35
41
  };
package/src/index.ts CHANGED
@@ -5,4 +5,4 @@
5
5
  export * from './Document';
6
6
  export * from './align';
7
7
  export * from './hash';
8
- export { type Parser, parseText, stubParse } from './parse';
8
+ export { type Parser, parseText } from './parse';
package/src/parse.test.ts CHANGED
@@ -5,7 +5,8 @@
5
5
  import { describe, test } from 'vitest';
6
6
 
7
7
  import { assembleDocument } from './align';
8
- import { type Parser, stubParse } from './parse';
8
+ import { type Parser } from './parse';
9
+ import { stubParse } from './testing';
9
10
 
10
11
  describe('parser seam', () => {
11
12
  test('stubParse satisfies the Parser contract', async ({ expect }) => {
package/src/parse.ts CHANGED
@@ -10,7 +10,6 @@ import { AiService } from '@dxos/ai';
10
10
 
11
11
  import { assembleDocument } from './align';
12
12
  import { type Document, Upos } from './Document';
13
- import { stubParse } from './stub';
14
13
 
15
14
  const PARSE_MODEL = 'com.anthropic.model.claude-haiku-4-5.default';
16
15
 
@@ -53,5 +52,3 @@ export const parseText = (text: string) =>
53
52
 
54
53
  /** The pluggable parser contract consumed by the editor extension and pipeline. */
55
54
  export type Parser = (text: string) => Promise<Document>;
56
-
57
- export { stubParse };
@@ -4,4 +4,4 @@
4
4
 
5
5
  // Import directly from `stub` (not `parse`) so the testing entrypoint stays isolated to the
6
6
  // offline tagger and does not pull the live-parser/AI stack into story and test bundles.
7
- export { stubParse } from '../stub';
7
+ export { stubParse } from './stub';
@@ -2,11 +2,11 @@
2
2
  // Copyright 2026 DXOS.org
3
3
  //
4
4
 
5
- import { assembleDocument } from './align';
6
- import { type Document, type RawSentence, type Upos } from './Document';
5
+ import { assembleDocument } from '../align';
6
+ import { type Document, type RawSentence, type Upos } from '../Document';
7
7
 
8
- // Closed-class lexicon: small, deterministic, language-is-English assumption (the stub is a demo
9
- // fallback, not the production tagger). Lowercased keys.
8
+ // Closed-class lexicon: small, deterministic, language-is-English assumption
9
+ // (the stub is a demo fallback, not the production tagger). Lowercased keys.
10
10
  const LEXICON: Record<string, Upos> = {
11
11
  the: 'DET',
12
12
  a: 'DET',
@@ -58,7 +58,9 @@ const LEXICON: Record<string, Upos> = {
58
58
 
59
59
  const WORD_RE = /[A-Za-z]+(?:'[A-Za-z]+)?|[0-9]+|[.!?,;:]/g;
60
60
 
61
- /** Tag one token by lexicon → number → suffix heuristic → capitalization. `initial` = sentence start. */
61
+ /**
62
+ * Fake tags one token by lexicon → number → suffix heuristic → capitalization. `initial` = sentence start.
63
+ */
62
64
  const tagWord = (raw: string, initial: boolean): Upos => {
63
65
  if (/^[.!?,;:]$/.test(raw)) {
64
66
  return 'PUNCT';
@@ -100,9 +102,11 @@ export const stubTag = (text: string): RawSentence[] => {
100
102
  initial = true;
101
103
  }
102
104
  }
105
+
103
106
  if (tokens.length > 0) {
104
107
  sentences.push({ tokens });
105
108
  }
109
+
106
110
  return sentences;
107
111
  };
108
112
 
@@ -1,156 +0,0 @@
1
- // src/hash.ts
2
- var sourceHash = (text) => {
3
- let hash = 2166136261;
4
- for (let index = 0; index < text.length; index++) {
5
- hash ^= text.charCodeAt(index);
6
- hash = Math.imul(hash, 16777619);
7
- }
8
- return (hash >>> 0).toString(16).padStart(8, "0");
9
- };
10
-
11
- // src/align.ts
12
- var assembleDocument = (sourceText, rawSentences) => {
13
- let cursor = 0;
14
- const sentences = [];
15
- rawSentences.forEach((raw, sentenceIndex) => {
16
- const tokens = [];
17
- for (const { text, upos } of raw.tokens) {
18
- const start = sourceText.indexOf(text, cursor);
19
- if (start < 0) {
20
- continue;
21
- }
22
- const end = start + text.length;
23
- tokens.push({
24
- index: tokens.length,
25
- text,
26
- upos,
27
- start,
28
- end
29
- });
30
- cursor = end;
31
- }
32
- if (tokens.length > 0) {
33
- sentences.push({
34
- index: sentenceIndex,
35
- start: tokens[0].start,
36
- end: tokens[tokens.length - 1].end,
37
- tokens
38
- });
39
- }
40
- });
41
- return {
42
- sourceHash: sourceHash(sourceText),
43
- sentences,
44
- timestamp: void 0
45
- };
46
- };
47
-
48
- // src/stub.ts
49
- var LEXICON = {
50
- the: "DET",
51
- a: "DET",
52
- an: "DET",
53
- this: "DET",
54
- that: "DET",
55
- these: "DET",
56
- those: "DET",
57
- i: "PRON",
58
- you: "PRON",
59
- he: "PRON",
60
- she: "PRON",
61
- it: "PRON",
62
- we: "PRON",
63
- they: "PRON",
64
- is: "AUX",
65
- am: "AUX",
66
- are: "AUX",
67
- was: "AUX",
68
- were: "AUX",
69
- be: "AUX",
70
- been: "AUX",
71
- do: "AUX",
72
- did: "AUX",
73
- in: "ADP",
74
- on: "ADP",
75
- at: "ADP",
76
- of: "ADP",
77
- to: "ADP",
78
- over: "ADP",
79
- under: "ADP",
80
- with: "ADP",
81
- for: "ADP",
82
- and: "CCONJ",
83
- or: "CCONJ",
84
- but: "CCONJ",
85
- because: "SCONJ",
86
- if: "SCONJ",
87
- while: "SCONJ",
88
- although: "SCONJ",
89
- not: "PART",
90
- very: "ADV",
91
- quickly: "ADV",
92
- well: "ADV",
93
- oh: "INTJ",
94
- yes: "INTJ",
95
- no: "INTJ"
96
- };
97
- var WORD_RE = /[A-Za-z]+(?:'[A-Za-z]+)?|[0-9]+|[.!?,;:]/g;
98
- var tagWord = (raw, initial) => {
99
- if (/^[.!?,;:]$/.test(raw)) {
100
- return "PUNCT";
101
- }
102
- if (/^[0-9]+$/.test(raw)) {
103
- return "NUM";
104
- }
105
- const lower = raw.toLowerCase();
106
- if (LEXICON[lower]) {
107
- return LEXICON[lower];
108
- }
109
- if (/^[A-Z]/.test(raw) && !initial) {
110
- return "PROPN";
111
- }
112
- if (/(ing|ed|ize|ise)$/.test(lower)) {
113
- return "VERB";
114
- }
115
- if (/(ly)$/.test(lower)) {
116
- return "ADV";
117
- }
118
- if (/(ous|ful|ive|able|al)$/.test(lower)) {
119
- return "ADJ";
120
- }
121
- return "NOUN";
122
- };
123
- var stubTag = (text) => {
124
- const sentences = [];
125
- let tokens = [];
126
- let initial = true;
127
- for (const match of text.matchAll(WORD_RE)) {
128
- const raw = match[0];
129
- tokens.push({
130
- text: raw,
131
- upos: tagWord(raw, initial)
132
- });
133
- initial = false;
134
- if (/^[.!?]$/.test(raw)) {
135
- sentences.push({
136
- tokens
137
- });
138
- tokens = [];
139
- initial = true;
140
- }
141
- }
142
- if (tokens.length > 0) {
143
- sentences.push({
144
- tokens
145
- });
146
- }
147
- return sentences;
148
- };
149
- var stubParse = async (text) => assembleDocument(text, stubTag(text));
150
-
151
- export {
152
- sourceHash,
153
- assembleDocument,
154
- stubParse
155
- };
156
- //# sourceMappingURL=chunk-D4MHDU46.mjs.map
@@ -1,7 +0,0 @@
1
- {
2
- "version": 3,
3
- "sources": ["../../../src/hash.ts", "../../../src/align.ts", "../../../src/stub.ts"],
4
- "sourcesContent": ["//\n// Copyright 2026 DXOS.org\n//\n\n/**\n * Fast non-cryptographic hash (FNV-1a, 32-bit) of source text. Used purely to detect whether an\n * analyzed span still matches the current editor text — change detection, not security.\n */\nexport const sourceHash = (text: string): string => {\n let hash = 0x811c9dc5;\n for (let index = 0; index < text.length; index++) {\n hash ^= text.charCodeAt(index);\n hash = Math.imul(hash, 0x01000193);\n }\n return (hash >>> 0).toString(16).padStart(8, '0');\n};\n", "//\n// Copyright 2026 DXOS.org\n//\n\nimport { type Document, type RawSentence, type Sentence, type Token } from './Document';\nimport { sourceHash } from './hash';\n\n/**\n * Align offset-free tagger output against the source text to compute exact character offsets.\n * A single forward cursor guarantees repeated surface forms map to successive occurrences rather\n * than re-matching the first. Tokens whose surface form cannot be located ahead of the cursor are\n * dropped (the tagger hallucinated a token), keeping offsets internally consistent.\n */\nexport const assembleDocument = (sourceText: string, rawSentences: readonly RawSentence[]): Document => {\n let cursor = 0;\n const sentences: Sentence[] = [];\n\n rawSentences.forEach((raw, sentenceIndex) => {\n const tokens: Token[] = [];\n for (const { text, upos } of raw.tokens) {\n const start = sourceText.indexOf(text, cursor);\n if (start < 0) {\n continue;\n }\n const end = start + text.length;\n tokens.push({ index: tokens.length, text, upos, start, end });\n cursor = end;\n }\n if (tokens.length > 0) {\n sentences.push({ index: sentenceIndex, start: tokens[0].start, end: tokens[tokens.length - 1].end, tokens });\n }\n });\n\n return { sourceHash: sourceHash(sourceText), sentences, timestamp: undefined };\n};\n", "//\n// Copyright 2026 DXOS.org\n//\n\nimport { assembleDocument } from './align';\nimport { type Document, type RawSentence, type Upos } from './Document';\n\n// Closed-class lexicon: small, deterministic, language-is-English assumption (the stub is a demo\n// fallback, not the production tagger). Lowercased keys.\nconst LEXICON: Record<string, Upos> = {\n the: 'DET',\n a: 'DET',\n an: 'DET',\n this: 'DET',\n that: 'DET',\n these: 'DET',\n those: 'DET',\n i: 'PRON',\n you: 'PRON',\n he: 'PRON',\n she: 'PRON',\n it: 'PRON',\n we: 'PRON',\n they: 'PRON',\n is: 'AUX',\n am: 'AUX',\n are: 'AUX',\n was: 'AUX',\n were: 'AUX',\n be: 'AUX',\n been: 'AUX',\n do: 'AUX',\n did: 'AUX',\n in: 'ADP',\n on: 'ADP',\n at: 'ADP',\n of: 'ADP',\n to: 'ADP',\n over: 'ADP',\n under: 'ADP',\n with: 'ADP',\n for: 'ADP',\n and: 'CCONJ',\n or: 'CCONJ',\n but: 'CCONJ',\n because: 'SCONJ',\n if: 'SCONJ',\n while: 'SCONJ',\n although: 'SCONJ',\n not: 'PART',\n very: 'ADV',\n quickly: 'ADV',\n well: 'ADV',\n oh: 'INTJ',\n yes: 'INTJ',\n no: 'INTJ',\n};\n\nconst WORD_RE = /[A-Za-z]+(?:'[A-Za-z]+)?|[0-9]+|[.!?,;:]/g;\n\n/** Tag one token by lexicon → number → suffix heuristic → capitalization. `initial` = sentence start. */\nconst tagWord = (raw: string, initial: boolean): Upos => {\n if (/^[.!?,;:]$/.test(raw)) {\n return 'PUNCT';\n }\n if (/^[0-9]+$/.test(raw)) {\n return 'NUM';\n }\n const lower = raw.toLowerCase();\n if (LEXICON[lower]) {\n return LEXICON[lower];\n }\n if (/^[A-Z]/.test(raw) && !initial) {\n return 'PROPN';\n }\n if (/(ing|ed|ize|ise)$/.test(lower)) {\n return 'VERB';\n }\n if (/(ly)$/.test(lower)) {\n return 'ADV';\n }\n if (/(ous|ful|ive|able|al)$/.test(lower)) {\n return 'ADJ';\n }\n return 'NOUN';\n};\n\n/** Deterministic UPOS tagger: splits on sentence-final punctuation, tags each token. */\nexport const stubTag = (text: string): RawSentence[] => {\n const sentences: RawSentence[] = [];\n let tokens: { text: string; upos: Upos }[] = [];\n let initial = true;\n for (const match of text.matchAll(WORD_RE)) {\n const raw = match[0];\n tokens.push({ text: raw, upos: tagWord(raw, initial) });\n initial = false;\n if (/^[.!?]$/.test(raw)) {\n sentences.push({ tokens });\n tokens = [];\n initial = true;\n }\n }\n if (tokens.length > 0) {\n sentences.push({ tokens });\n }\n return sentences;\n};\n\n/** Parser-shaped wrapper around the stub tagger (async to match the `Parser` contract). */\nexport const stubParse = async (text: string): Promise<Document> => assembleDocument(text, stubTag(text));\n"],
5
- "mappings": ";AAQO,IAAMA,aAAa,CAACC,SAAAA;AACzB,MAAIC,OAAO;AACX,WAASC,QAAQ,GAAGA,QAAQF,KAAKG,QAAQD,SAAS;AAChDD,YAAQD,KAAKI,WAAWF,KAAAA;AACxBD,WAAOI,KAAKC,KAAKL,MAAM,QAAA;EACzB;AACA,UAAQA,SAAS,GAAGM,SAAS,EAAA,EAAIC,SAAS,GAAG,GAAA;AAC/C;;;ACFO,IAAMC,mBAAmB,CAACC,YAAoBC,iBAAAA;AACnD,MAAIC,SAAS;AACb,QAAMC,YAAwB,CAAA;AAE9BF,eAAaG,QAAQ,CAACC,KAAKC,kBAAAA;AACzB,UAAMC,SAAkB,CAAA;AACxB,eAAW,EAAEC,MAAMC,KAAI,KAAMJ,IAAIE,QAAQ;AACvC,YAAMG,QAAQV,WAAWW,QAAQH,MAAMN,MAAAA;AACvC,UAAIQ,QAAQ,GAAG;AACb;MACF;AACA,YAAME,MAAMF,QAAQF,KAAKK;AACzBN,aAAOO,KAAK;QAAEC,OAAOR,OAAOM;QAAQL;QAAMC;QAAMC;QAAOE;MAAI,CAAA;AAC3DV,eAASU;IACX;AACA,QAAIL,OAAOM,SAAS,GAAG;AACrBV,gBAAUW,KAAK;QAAEC,OAAOT;QAAeI,OAAOH,OAAO,CAAA,EAAGG;QAAOE,KAAKL,OAAOA,OAAOM,SAAS,CAAA,EAAGD;QAAKL;MAAO,CAAA;IAC5G;EACF,CAAA;AAEA,SAAO;IAAES,YAAYA,WAAWhB,UAAAA;IAAaG;IAAWc,WAAWC;EAAU;AAC/E;;;ACzBA,IAAMC,UAAgC;EACpCC,KAAK;EACLC,GAAG;EACHC,IAAI;EACJC,MAAM;EACNC,MAAM;EACNC,OAAO;EACPC,OAAO;EACPC,GAAG;EACHC,KAAK;EACLC,IAAI;EACJC,KAAK;EACLC,IAAI;EACJC,IAAI;EACJC,MAAM;EACNC,IAAI;EACJC,IAAI;EACJC,KAAK;EACLC,KAAK;EACLC,MAAM;EACNC,IAAI;EACJC,MAAM;EACNC,IAAI;EACJC,KAAK;EACLC,IAAI;EACJC,IAAI;EACJC,IAAI;EACJC,IAAI;EACJC,IAAI;EACJC,MAAM;EACNC,OAAO;EACPC,MAAM;EACNC,KAAK;EACLC,KAAK;EACLC,IAAI;EACJC,KAAK;EACLC,SAAS;EACTC,IAAI;EACJC,OAAO;EACPC,UAAU;EACVC,KAAK;EACLC,MAAM;EACNC,SAAS;EACTC,MAAM;EACNC,IAAI;EACJC,KAAK;EACLC,IAAI;AACN;AAEA,IAAMC,UAAU;AAGhB,IAAMC,UAAU,CAACC,KAAaC,YAAAA;AAC5B,MAAI,aAAaC,KAAKF,GAAAA,GAAM;AAC1B,WAAO;EACT;AACA,MAAI,WAAWE,KAAKF,GAAAA,GAAM;AACxB,WAAO;EACT;AACA,QAAMG,QAAQH,IAAII,YAAW;AAC7B,MAAIrD,QAAQoD,KAAAA,GAAQ;AAClB,WAAOpD,QAAQoD,KAAAA;EACjB;AACA,MAAI,SAASD,KAAKF,GAAAA,KAAQ,CAACC,SAAS;AAClC,WAAO;EACT;AACA,MAAI,oBAAoBC,KAAKC,KAAAA,GAAQ;AACnC,WAAO;EACT;AACA,MAAI,QAAQD,KAAKC,KAAAA,GAAQ;AACvB,WAAO;EACT;AACA,MAAI,yBAAyBD,KAAKC,KAAAA,GAAQ;AACxC,WAAO;EACT;AACA,SAAO;AACT;AAGO,IAAME,UAAU,CAACC,SAAAA;AACtB,QAAMC,YAA2B,CAAA;AACjC,MAAIC,SAAyC,CAAA;AAC7C,MAAIP,UAAU;AACd,aAAWQ,SAASH,KAAKI,SAASZ,OAAAA,GAAU;AAC1C,UAAME,MAAMS,MAAM,CAAA;AAClBD,WAAOG,KAAK;MAAEL,MAAMN;MAAKY,MAAMb,QAAQC,KAAKC,OAAAA;IAAS,CAAA;AACrDA,cAAU;AACV,QAAI,UAAUC,KAAKF,GAAAA,GAAM;AACvBO,gBAAUI,KAAK;QAAEH;MAAO,CAAA;AACxBA,eAAS,CAAA;AACTP,gBAAU;IACZ;EACF;AACA,MAAIO,OAAOK,SAAS,GAAG;AACrBN,cAAUI,KAAK;MAAEH;IAAO,CAAA;EAC1B;AACA,SAAOD;AACT;AAGO,IAAMO,YAAY,OAAOR,SAAoCS,iBAAiBT,MAAMD,QAAQC,IAAAA,CAAAA;",
6
- "names": ["sourceHash", "text", "hash", "index", "length", "charCodeAt", "Math", "imul", "toString", "padStart", "assembleDocument", "sourceText", "rawSentences", "cursor", "sentences", "forEach", "raw", "sentenceIndex", "tokens", "text", "upos", "start", "indexOf", "end", "length", "push", "index", "sourceHash", "timestamp", "undefined", "LEXICON", "the", "a", "an", "this", "that", "these", "those", "i", "you", "he", "she", "it", "we", "they", "is", "am", "are", "was", "were", "be", "been", "do", "did", "in", "on", "at", "of", "to", "over", "under", "with", "for", "and", "or", "but", "because", "if", "while", "although", "not", "very", "quickly", "well", "oh", "yes", "no", "WORD_RE", "tagWord", "raw", "initial", "test", "lower", "toLowerCase", "stubTag", "text", "sentences", "tokens", "match", "matchAll", "push", "upos", "length", "stubParse", "assembleDocument"]
7
- }
@@ -1,78 +0,0 @@
1
- import {
2
- assembleDocument,
3
- sourceHash,
4
- stubParse
5
- } from "./chunk-D4MHDU46.mjs";
6
-
7
- // src/Document.ts
8
- import * as Schema from "effect/Schema";
9
- var Upos = Schema.Literal("ADJ", "ADP", "ADV", "AUX", "CCONJ", "DET", "INTJ", "NOUN", "NUM", "PART", "PRON", "PROPN", "PUNCT", "SCONJ", "SYM", "VERB", "X");
10
- var Token = Schema.Struct({
11
- index: Schema.Number.annotations({
12
- description: "Position of the token within its sentence."
13
- }),
14
- text: Schema.String.annotations({
15
- description: "Surface form exactly as it appears in the source."
16
- }),
17
- upos: Upos.annotations({
18
- description: "Universal part-of-speech tag."
19
- }),
20
- start: Schema.Number,
21
- end: Schema.Number
22
- });
23
- var Sentence = Schema.Struct({
24
- index: Schema.Number,
25
- start: Schema.Number,
26
- end: Schema.Number,
27
- tokens: Schema.Array(Token)
28
- });
29
- var Document = Schema.Struct({
30
- sourceHash: Schema.String,
31
- sentences: Schema.Array(Sentence),
32
- timestamp: Schema.optional(Schema.Number)
33
- });
34
-
35
- // src/parse.ts
36
- import * as LanguageModel from "@effect/ai/LanguageModel";
37
- import * as Effect from "effect/Effect";
38
- import * as Schema2 from "effect/Schema";
39
- import { AiService } from "@dxos/ai";
40
- var PARSE_MODEL = "com.anthropic.model.claude-haiku-4-5.default";
41
- var TaggedSentences = Schema2.Struct({
42
- sentences: Schema2.Array(Schema2.Struct({
43
- tokens: Schema2.Array(Schema2.Struct({
44
- text: Schema2.String.annotations({
45
- description: "Token surface form exactly as in the source."
46
- }),
47
- upos: Upos.annotations({
48
- description: "Universal POS tag for the token."
49
- })
50
- }))
51
- }))
52
- });
53
- var parseText = (text) => Effect.gen(function* () {
54
- const { value } = yield* Effect.scoped(LanguageModel.generateObject({
55
- schema: TaggedSentences,
56
- prompt: [
57
- "Tokenize the text below into sentences and tokens, and tag each token with its",
58
- "Universal POS tag (UPOS): ADJ ADP ADV AUX CCONJ DET INTJ NOUN NUM PART PRON PROPN",
59
- "PUNCT SCONJ SYM VERB X. Return each token surface form exactly as it appears, in order,",
60
- "including punctuation as its own PUNCT token. Do not add or omit tokens.",
61
- "",
62
- "Text:",
63
- text
64
- ].join("\n")
65
- }));
66
- return assembleDocument(text, value.sentences);
67
- }).pipe(Effect.provide(AiService.model(PARSE_MODEL)));
68
- export {
69
- Document,
70
- Sentence,
71
- Token,
72
- Upos,
73
- assembleDocument,
74
- parseText,
75
- sourceHash,
76
- stubParse
77
- };
78
- //# sourceMappingURL=index.mjs.map
@@ -1,7 +0,0 @@
1
- {
2
- "version": 3,
3
- "sources": ["../../../src/Document.ts", "../../../src/parse.ts"],
4
- "sourcesContent": ["//\n// Copyright 2026 DXOS.org\n//\n\nimport * as Schema from 'effect/Schema';\n\n/** Universal POS tagset (17 tags). https://universaldependencies.org/u/pos/ */\nexport const Upos = Schema.Literal(\n 'ADJ',\n 'ADP',\n 'ADV',\n 'AUX',\n 'CCONJ',\n 'DET',\n 'INTJ',\n 'NOUN',\n 'NUM',\n 'PART',\n 'PRON',\n 'PROPN',\n 'PUNCT',\n 'SCONJ',\n 'SYM',\n 'VERB',\n 'X',\n);\nexport type Upos = Schema.Schema.Type<typeof Upos>;\n\n/** A single word/punctuation token. `start`/`end` are character offsets within the source text. */\nexport const Token = Schema.Struct({\n index: Schema.Number.annotations({ description: 'Position of the token within its sentence.' }),\n text: Schema.String.annotations({ description: 'Surface form exactly as it appears in the source.' }),\n upos: Upos.annotations({ description: 'Universal part-of-speech tag.' }),\n start: Schema.Number,\n end: Schema.Number,\n});\nexport type Token = Schema.Schema.Type<typeof Token>;\n\nexport const Sentence = Schema.Struct({\n index: Schema.Number,\n start: Schema.Number,\n end: Schema.Number,\n tokens: Schema.Array(Token),\n});\nexport type Sentence = Schema.Schema.Type<typeof Sentence>;\n\n/** A parsed document. `sourceHash` is the divergence signal; `timestamp` is debug-only. */\nexport const Document = Schema.Struct({\n sourceHash: Schema.String,\n sentences: Schema.Array(Sentence),\n timestamp: Schema.optional(Schema.Number),\n});\nexport type Document = Schema.Schema.Type<typeof Document>;\n\n/** Raw, offset-free output of a tagger before alignment. */\nexport type RawSentence = { readonly tokens: readonly { readonly text: string; readonly upos: Upos }[] };\n", "//\n// Copyright 2026 DXOS.org\n//\n\nimport * as LanguageModel from '@effect/ai/LanguageModel';\nimport * as Effect from 'effect/Effect';\nimport * as Schema from 'effect/Schema';\n\nimport { AiService } from '@dxos/ai';\n\nimport { assembleDocument } from './align';\nimport { type Document, Upos } from './Document';\nimport { stubParse } from './stub';\n\nconst PARSE_MODEL = 'com.anthropic.model.claude-haiku-4-5.default';\n\n/** LLM output schema: sentences → tokens, no offsets (alignment computes those). */\nconst TaggedSentences = Schema.Struct({\n sentences: Schema.Array(\n Schema.Struct({\n tokens: Schema.Array(\n Schema.Struct({\n text: Schema.String.annotations({ description: 'Token surface form exactly as in the source.' }),\n upos: Upos.annotations({ description: 'Universal POS tag for the token.' }),\n }),\n ),\n }),\n ),\n});\n\n/**\n * Tag `text` with UPOS via a small LLM, then deterministically align tokens to source offsets.\n * Provides the LanguageModel internally; residual requirement is {@link AiService.AiService}.\n */\nexport const parseText = (text: string) =>\n Effect.gen(function* () {\n const { value } = yield* Effect.scoped(\n LanguageModel.generateObject({\n schema: TaggedSentences,\n prompt: [\n 'Tokenize the text below into sentences and tokens, and tag each token with its',\n 'Universal POS tag (UPOS): ADJ ADP ADV AUX CCONJ DET INTJ NOUN NUM PART PRON PROPN',\n 'PUNCT SCONJ SYM VERB X. Return each token surface form exactly as it appears, in order,',\n 'including punctuation as its own PUNCT token. Do not add or omit tokens.',\n '',\n 'Text:',\n text,\n ].join('\\n'),\n }),\n );\n return assembleDocument(text, value.sentences);\n }).pipe(Effect.provide(AiService.model(PARSE_MODEL)));\n\n/** The pluggable parser contract consumed by the editor extension and pipeline. */\nexport type Parser = (text: string) => Promise<Document>;\n\nexport { stubParse };\n"],
5
- "mappings": ";;;;;;;AAIA,YAAYA,YAAY;AAGjB,IAAMC,OAAcC,eACzB,OACA,OACA,OACA,OACA,SACA,OACA,QACA,QACA,OACA,QACA,QACA,SACA,SACA,SACA,OACA,QACA,GAAA;AAKK,IAAMC,QAAeC,cAAO;EACjCC,OAAcC,cAAOC,YAAY;IAAEC,aAAa;EAA6C,CAAA;EAC7FC,MAAaC,cAAOH,YAAY;IAAEC,aAAa;EAAoD,CAAA;EACnGG,MAAMV,KAAKM,YAAY;IAAEC,aAAa;EAAgC,CAAA;EACtEI,OAAcN;EACdO,KAAYP;AACd,CAAA;AAGO,IAAMQ,WAAkBV,cAAO;EACpCC,OAAcC;EACdM,OAAcN;EACdO,KAAYP;EACZS,QAAeC,aAAMb,KAAAA;AACvB,CAAA;AAIO,IAAMc,WAAkBb,cAAO;EACpCc,YAAmBR;EACnBS,WAAkBH,aAAMF,QAAAA;EACxBM,WAAkBC,gBAAgBf,aAAM;AAC1C,CAAA;;;AC/CA,YAAYgB,mBAAmB;AAC/B,YAAYC,YAAY;AACxB,YAAYC,aAAY;AAExB,SAASC,iBAAiB;AAM1B,IAAMC,cAAc;AAGpB,IAAMC,kBAAyBC,eAAO;EACpCC,WAAkBC,cACTF,eAAO;IACZG,QAAeD,cACNF,eAAO;MACZI,MAAaC,eAAOC,YAAY;QAAEC,aAAa;MAA+C,CAAA;MAC9FC,MAAMC,KAAKH,YAAY;QAAEC,aAAa;MAAmC,CAAA;IAC3E,CAAA,CAAA;EAEJ,CAAA,CAAA;AAEJ,CAAA;AAMO,IAAMG,YAAY,CAACN,SACjBO,WAAI,aAAA;AACT,QAAM,EAAEC,MAAK,IAAK,OAAcC,cAChBC,6BAAe;IAC3BC,QAAQhB;IACRiB,QAAQ;MACN;MACA;MACA;MACA;MACA;MACA;MACAZ;MACAa,KAAK,IAAA;EACT,CAAA,CAAA;AAEF,SAAOC,iBAAiBd,MAAMQ,MAAMX,SAAS;AAC/C,CAAA,EAAGkB,KAAYC,eAAQC,UAAUC,MAAMxB,WAAAA,CAAAA,CAAAA;",
6
- "names": ["Schema", "Upos", "Literal", "Token", "Struct", "index", "Number", "annotations", "description", "text", "String", "upos", "start", "end", "Sentence", "tokens", "Array", "Document", "sourceHash", "sentences", "timestamp", "optional", "LanguageModel", "Effect", "Schema", "AiService", "PARSE_MODEL", "TaggedSentences", "Struct", "sentences", "Array", "tokens", "text", "String", "annotations", "description", "upos", "Upos", "parseText", "gen", "value", "scoped", "generateObject", "schema", "prompt", "join", "assembleDocument", "pipe", "provide", "AiService", "model"]
7
- }
@@ -1 +0,0 @@
1
- {"inputs":{"src/Document.ts":{"bytes":4951,"imports":[{"path":"effect/Schema","kind":"import-statement","external":true}],"format":"esm"},"src/hash.ts":{"bytes":1857,"imports":[],"format":"esm"},"src/align.ts":{"bytes":4637,"imports":[{"path":"src/hash.ts","kind":"import-statement","original":"./hash"}],"format":"esm"},"src/stub.ts":{"bytes":9098,"imports":[{"path":"src/align.ts","kind":"import-statement","original":"./align"}],"format":"esm"},"src/parse.ts":{"bytes":6393,"imports":[{"path":"@effect/ai/LanguageModel","kind":"import-statement","external":true},{"path":"effect/Effect","kind":"import-statement","external":true},{"path":"effect/Schema","kind":"import-statement","external":true},{"path":"@dxos/ai","kind":"import-statement","external":true},{"path":"src/align.ts","kind":"import-statement","original":"./align"},{"path":"src/Document.ts","kind":"import-statement","original":"./Document"},{"path":"src/stub.ts","kind":"import-statement","original":"./stub"}],"format":"esm"},"src/index.ts":{"bytes":749,"imports":[{"path":"src/Document.ts","kind":"import-statement","original":"./Document"},{"path":"src/align.ts","kind":"import-statement","original":"./align"},{"path":"src/hash.ts","kind":"import-statement","original":"./hash"},{"path":"src/parse.ts","kind":"import-statement","original":"./parse"}],"format":"esm"},"src/testing/index.ts":{"bytes":892,"imports":[{"path":"src/stub.ts","kind":"import-statement","original":"../stub"}],"format":"esm"}},"outputs":{"dist/lib/neutral/index.mjs.map":{"imports":[],"exports":[],"inputs":{},"bytes":5555},"dist/lib/neutral/index.mjs":{"imports":[{"path":"dist/lib/neutral/chunk-D4MHDU46.mjs","kind":"import-statement"},{"path":"effect/Schema","kind":"import-statement","external":true},{"path":"@effect/ai/LanguageModel","kind":"import-statement","external":true},{"path":"effect/Effect","kind":"import-statement","external":true},{"path":"effect/Schema","kind":"import-statement","external":true},{"path":"@dxos/ai","kind":"import-statement","external":true}],"exports":["Document","Sentence","Token","Upos","assembleDocument","parseText","sourceHash","stubParse"],"entryPoint":"src/index.ts","inputs":{"src/Document.ts":{"bytesInOutput":853},"src/index.ts":{"bytesInOutput":0},"src/parse.ts":{"bytesInOutput":1295}},"bytes":2418},"dist/lib/neutral/testing/index.mjs.map":{"imports":[],"exports":[],"inputs":{},"bytes":93},"dist/lib/neutral/testing/index.mjs":{"imports":[{"path":"dist/lib/neutral/chunk-D4MHDU46.mjs","kind":"import-statement"}],"exports":["stubParse"],"entryPoint":"src/testing/index.ts","inputs":{"src/testing/index.ts":{"bytesInOutput":0}},"bytes":112},"dist/lib/neutral/chunk-D4MHDU46.mjs.map":{"imports":[],"exports":[],"inputs":{},"bytes":7870},"dist/lib/neutral/chunk-D4MHDU46.mjs":{"imports":[],"exports":["assembleDocument","sourceHash","stubParse"],"inputs":{"src/hash.ts":{"bytesInOutput":242},"src/align.ts":{"bytesInOutput":790},"src/stub.ts":{"bytesInOutput":1814}},"bytes":2997}}}
@@ -1,7 +0,0 @@
1
- import {
2
- stubParse
3
- } from "../chunk-D4MHDU46.mjs";
4
- export {
5
- stubParse
6
- };
7
- //# sourceMappingURL=index.mjs.map
@@ -1,7 +0,0 @@
1
- {
2
- "version": 3,
3
- "sources": [],
4
- "sourcesContent": [],
5
- "mappings": "",
6
- "names": []
7
- }
@@ -1 +0,0 @@
1
- {"version":3,"file":"stub.d.ts","sourceRoot":"","sources":["../../../src/stub.ts"],"names":[],"mappings":"AAKA,OAAO,EAAE,KAAK,QAAQ,EAAE,KAAK,WAAW,EAAa,MAAM,YAAY,CAAC;AAkFxE,wFAAwF;AACxF,eAAO,MAAM,OAAO,SAAU,MAAM,KAAG,WAAW,EAkBjD,CAAC;AAEF,2FAA2F;AAC3F,eAAO,MAAM,SAAS,SAAgB,MAAM,KAAG,OAAO,CAAC,QAAQ,CAA0C,CAAC"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"stub.test.d.ts","sourceRoot":"","sources":["../../../src/stub.test.ts"],"names":[],"mappings":""}
File without changes