@dxos/nlp 0.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/dist/lib/neutral/chunk-D4MHDU46.mjs +156 -0
  2. package/dist/lib/neutral/chunk-D4MHDU46.mjs.map +7 -0
  3. package/dist/lib/neutral/index.mjs +78 -0
  4. package/dist/lib/neutral/index.mjs.map +7 -0
  5. package/dist/lib/neutral/meta.json +1 -0
  6. package/dist/lib/neutral/testing/index.mjs +7 -0
  7. package/dist/lib/neutral/testing/index.mjs.map +7 -0
  8. package/dist/types/src/Document.d.ts +52 -0
  9. package/dist/types/src/Document.d.ts.map +1 -0
  10. package/dist/types/src/align.d.ts +9 -0
  11. package/dist/types/src/align.d.ts.map +1 -0
  12. package/dist/types/src/align.test.d.ts +2 -0
  13. package/dist/types/src/align.test.d.ts.map +1 -0
  14. package/dist/types/src/hash.d.ts +6 -0
  15. package/dist/types/src/hash.d.ts.map +1 -0
  16. package/dist/types/src/hash.test.d.ts +2 -0
  17. package/dist/types/src/hash.test.d.ts.map +1 -0
  18. package/dist/types/src/index.d.ts +5 -0
  19. package/dist/types/src/index.d.ts.map +1 -0
  20. package/dist/types/src/parse.d.ts +28 -0
  21. package/dist/types/src/parse.d.ts.map +1 -0
  22. package/dist/types/src/parse.test.d.ts +2 -0
  23. package/dist/types/src/parse.test.d.ts.map +1 -0
  24. package/dist/types/src/stub.d.ts +6 -0
  25. package/dist/types/src/stub.d.ts.map +1 -0
  26. package/dist/types/src/stub.test.d.ts +2 -0
  27. package/dist/types/src/stub.test.d.ts.map +1 -0
  28. package/dist/types/src/testing/index.d.ts +2 -0
  29. package/dist/types/src/testing/index.d.ts.map +1 -0
  30. package/dist/types/tsconfig.tsbuildinfo +1 -0
  31. package/package.json +46 -0
  32. package/src/Document.ts +56 -0
  33. package/src/align.test.ts +58 -0
  34. package/src/align.ts +35 -0
  35. package/src/hash.test.ts +20 -0
  36. package/src/hash.ts +16 -0
  37. package/src/index.ts +8 -0
  38. package/src/parse.test.ts +31 -0
  39. package/src/parse.ts +57 -0
  40. package/src/stub.test.ts +31 -0
  41. package/src/stub.ts +110 -0
  42. package/src/testing/index.ts +7 -0
package/package.json ADDED
@@ -0,0 +1,46 @@
1
+ {
2
+ "name": "@dxos/nlp",
3
+ "version": "0.0.0",
4
+ "description": "Natural-language parsing: per-word UPOS token structure for intention/transcript analysis.",
5
+ "homepage": "https://dxos.org",
6
+ "bugs": "https://github.com/dxos/dxos/issues",
7
+ "repository": {
8
+ "type": "git",
9
+ "url": "https://github.com/dxos/dxos"
10
+ },
11
+ "license": "FSL-1.1-Apache-2.0",
12
+ "author": "info@dxos.org",
13
+ "sideEffects": false,
14
+ "type": "module",
15
+ "exports": {
16
+ ".": {
17
+ "source": "./src/index.ts",
18
+ "types": "./dist/types/src/index.d.ts",
19
+ "default": "./dist/lib/neutral/index.mjs"
20
+ },
21
+ "./testing": {
22
+ "source": "./src/testing/index.ts",
23
+ "types": "./dist/types/src/testing/index.d.ts",
24
+ "default": "./dist/lib/neutral/testing/index.mjs"
25
+ }
26
+ },
27
+ "types": "dist/types/src/index.d.ts",
28
+ "files": [
29
+ "dist",
30
+ "src"
31
+ ],
32
+ "dependencies": {
33
+ "@dxos/ai": "workspace:*",
34
+ "@effect/ai": "catalog:"
35
+ },
36
+ "devDependencies": {
37
+ "effect": "catalog:"
38
+ },
39
+ "peerDependencies": {
40
+ "effect": "catalog:"
41
+ },
42
+ "publishConfig": {
43
+ "access": "public"
44
+ },
45
+ "beast": {}
46
+ }
@@ -0,0 +1,56 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import * as Schema from 'effect/Schema';
6
+
7
+ /** Universal POS tagset (17 tags). https://universaldependencies.org/u/pos/ */
8
+ export const Upos = Schema.Literal(
9
+ 'ADJ',
10
+ 'ADP',
11
+ 'ADV',
12
+ 'AUX',
13
+ 'CCONJ',
14
+ 'DET',
15
+ 'INTJ',
16
+ 'NOUN',
17
+ 'NUM',
18
+ 'PART',
19
+ 'PRON',
20
+ 'PROPN',
21
+ 'PUNCT',
22
+ 'SCONJ',
23
+ 'SYM',
24
+ 'VERB',
25
+ 'X',
26
+ );
27
+ export type Upos = Schema.Schema.Type<typeof Upos>;
28
+
29
+ /** A single word/punctuation token. `start`/`end` are character offsets within the source text. */
30
+ export const Token = Schema.Struct({
31
+ index: Schema.Number.annotations({ description: 'Position of the token within its sentence.' }),
32
+ text: Schema.String.annotations({ description: 'Surface form exactly as it appears in the source.' }),
33
+ upos: Upos.annotations({ description: 'Universal part-of-speech tag.' }),
34
+ start: Schema.Number,
35
+ end: Schema.Number,
36
+ });
37
+ export type Token = Schema.Schema.Type<typeof Token>;
38
+
39
+ export const Sentence = Schema.Struct({
40
+ index: Schema.Number,
41
+ start: Schema.Number,
42
+ end: Schema.Number,
43
+ tokens: Schema.Array(Token),
44
+ });
45
+ export type Sentence = Schema.Schema.Type<typeof Sentence>;
46
+
47
+ /** A parsed document. `sourceHash` is the divergence signal; `timestamp` is debug-only. */
48
+ export const Document = Schema.Struct({
49
+ sourceHash: Schema.String,
50
+ sentences: Schema.Array(Sentence),
51
+ timestamp: Schema.optional(Schema.Number),
52
+ });
53
+ export type Document = Schema.Schema.Type<typeof Document>;
54
+
55
+ /** Raw, offset-free output of a tagger before alignment. */
56
+ export type RawSentence = { readonly tokens: readonly { readonly text: string; readonly upos: Upos }[] };
@@ -0,0 +1,58 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import { describe, test } from 'vitest';
6
+
7
+ import { assembleDocument } from './align';
8
+ import { sourceHash } from './hash';
9
+
10
+ describe('assembleDocument', () => {
11
+ test('assigns exact offsets to each token', ({ expect }) => {
12
+ const source = 'The dog barks.';
13
+ const doc = assembleDocument(source, [
14
+ {
15
+ tokens: [
16
+ { text: 'The', upos: 'DET' },
17
+ { text: 'dog', upos: 'NOUN' },
18
+ { text: 'barks', upos: 'VERB' },
19
+ { text: '.', upos: 'PUNCT' },
20
+ ],
21
+ },
22
+ ]);
23
+
24
+ const [sentence] = doc.sentences;
25
+ expect(sentence.tokens.map((t) => source.slice(t.start, t.end))).toEqual(['The', 'dog', 'barks', '.']);
26
+ expect(sentence.tokens.map((t) => t.index)).toEqual([0, 1, 2, 3]);
27
+ expect(sentence.start).toBe(0);
28
+ expect(sentence.end).toBe(14);
29
+ expect(doc.sourceHash).toBe(sourceHash(source));
30
+ });
31
+
32
+ test('handles repeated words by scanning forward (no re-match of earlier occurrence)', ({ expect }) => {
33
+ const source = 'dog dog';
34
+ const doc = assembleDocument(source, [
35
+ {
36
+ tokens: [
37
+ { text: 'dog', upos: 'NOUN' },
38
+ { text: 'dog', upos: 'NOUN' },
39
+ ],
40
+ },
41
+ ]);
42
+ expect(doc.sentences[0].tokens.map((t) => t.start)).toEqual([0, 4]);
43
+ });
44
+
45
+ test('skips tokens not found in source rather than throwing', ({ expect }) => {
46
+ const source = 'hello world';
47
+ const doc = assembleDocument(source, [
48
+ {
49
+ tokens: [
50
+ { text: 'hello', upos: 'INTJ' },
51
+ { text: 'GHOST', upos: 'X' },
52
+ { text: 'world', upos: 'NOUN' },
53
+ ],
54
+ },
55
+ ]);
56
+ expect(doc.sentences[0].tokens.map((t) => t.text)).toEqual(['hello', 'world']);
57
+ });
58
+ });
package/src/align.ts ADDED
@@ -0,0 +1,35 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import { type Document, type RawSentence, type Sentence, type Token } from './Document';
6
+ import { sourceHash } from './hash';
7
+
8
+ /**
9
+ * Align offset-free tagger output against the source text to compute exact character offsets.
10
+ * A single forward cursor guarantees repeated surface forms map to successive occurrences rather
11
+ * than re-matching the first. Tokens whose surface form cannot be located ahead of the cursor are
12
+ * dropped (the tagger hallucinated a token), keeping offsets internally consistent.
13
+ */
14
+ export const assembleDocument = (sourceText: string, rawSentences: readonly RawSentence[]): Document => {
15
+ let cursor = 0;
16
+ const sentences: Sentence[] = [];
17
+
18
+ rawSentences.forEach((raw, sentenceIndex) => {
19
+ const tokens: Token[] = [];
20
+ for (const { text, upos } of raw.tokens) {
21
+ const start = sourceText.indexOf(text, cursor);
22
+ if (start < 0) {
23
+ continue;
24
+ }
25
+ const end = start + text.length;
26
+ tokens.push({ index: tokens.length, text, upos, start, end });
27
+ cursor = end;
28
+ }
29
+ if (tokens.length > 0) {
30
+ sentences.push({ index: sentenceIndex, start: tokens[0].start, end: tokens[tokens.length - 1].end, tokens });
31
+ }
32
+ });
33
+
34
+ return { sourceHash: sourceHash(sourceText), sentences, timestamp: undefined };
35
+ };
@@ -0,0 +1,20 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import { describe, test } from 'vitest';
6
+
7
+ import { sourceHash } from './hash';
8
+
9
+ describe('sourceHash', () => {
10
+ test('returns the expected FNV-1a digest for known inputs', ({ expect }) => {
11
+ // Pinned so a change to the algorithm or output format is caught; downstream code compares the
12
+ // stored hash string verbatim across the StateEffect boundary.
13
+ expect(sourceHash('')).toBe('811c9dc5');
14
+ expect(sourceHash('a')).toBe('e40c292c');
15
+ });
16
+
17
+ test('is order-sensitive', ({ expect }) => {
18
+ expect(sourceHash('the lazy dog')).not.toBe(sourceHash('the dog lazy'));
19
+ });
20
+ });
package/src/hash.ts ADDED
@@ -0,0 +1,16 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ /**
6
+ * Fast non-cryptographic hash (FNV-1a, 32-bit) of source text. Used purely to detect whether an
7
+ * analyzed span still matches the current editor text — change detection, not security.
8
+ */
9
+ export const sourceHash = (text: string): string => {
10
+ let hash = 0x811c9dc5;
11
+ for (let index = 0; index < text.length; index++) {
12
+ hash ^= text.charCodeAt(index);
13
+ hash = Math.imul(hash, 0x01000193);
14
+ }
15
+ return (hash >>> 0).toString(16).padStart(8, '0');
16
+ };
package/src/index.ts ADDED
@@ -0,0 +1,8 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ export * from './Document';
6
+ export * from './align';
7
+ export * from './hash';
8
+ export { type Parser, parseText, stubParse } from './parse';
@@ -0,0 +1,31 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import { describe, test } from 'vitest';
6
+
7
+ import { assembleDocument } from './align';
8
+ import { type Parser, stubParse } from './parse';
9
+
10
+ describe('parser seam', () => {
11
+ test('stubParse satisfies the Parser contract', async ({ expect }) => {
12
+ const parser: Parser = stubParse;
13
+ const doc = await parser('The cat sleeps.');
14
+ expect(doc.sentences[0].tokens.length).toBeGreaterThan(0);
15
+ });
16
+
17
+ test('alignment over model-shaped output yields exact offsets', ({ expect }) => {
18
+ const source = 'Alice runs fast.';
19
+ const doc = assembleDocument(source, [
20
+ {
21
+ tokens: [
22
+ { text: 'Alice', upos: 'PROPN' },
23
+ { text: 'runs', upos: 'VERB' },
24
+ { text: 'fast', upos: 'ADV' },
25
+ { text: '.', upos: 'PUNCT' },
26
+ ],
27
+ },
28
+ ]);
29
+ expect(doc.sentences[0].tokens.map((t) => source.slice(t.start, t.end))).toEqual(['Alice', 'runs', 'fast', '.']);
30
+ });
31
+ });
package/src/parse.ts ADDED
@@ -0,0 +1,57 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import * as LanguageModel from '@effect/ai/LanguageModel';
6
+ import * as Effect from 'effect/Effect';
7
+ import * as Schema from 'effect/Schema';
8
+
9
+ import { AiService } from '@dxos/ai';
10
+
11
+ import { assembleDocument } from './align';
12
+ import { type Document, Upos } from './Document';
13
+ import { stubParse } from './stub';
14
+
15
+ const PARSE_MODEL = 'com.anthropic.model.claude-haiku-4-5.default';
16
+
17
+ /** LLM output schema: sentences → tokens, no offsets (alignment computes those). */
18
+ const TaggedSentences = Schema.Struct({
19
+ sentences: Schema.Array(
20
+ Schema.Struct({
21
+ tokens: Schema.Array(
22
+ Schema.Struct({
23
+ text: Schema.String.annotations({ description: 'Token surface form exactly as in the source.' }),
24
+ upos: Upos.annotations({ description: 'Universal POS tag for the token.' }),
25
+ }),
26
+ ),
27
+ }),
28
+ ),
29
+ });
30
+
31
+ /**
32
+ * Tag `text` with UPOS via a small LLM, then deterministically align tokens to source offsets.
33
+ * Provides the LanguageModel internally; residual requirement is {@link AiService.AiService}.
34
+ */
35
+ export const parseText = (text: string) =>
36
+ Effect.gen(function* () {
37
+ const { value } = yield* Effect.scoped(
38
+ LanguageModel.generateObject({
39
+ schema: TaggedSentences,
40
+ prompt: [
41
+ 'Tokenize the text below into sentences and tokens, and tag each token with its',
42
+ 'Universal POS tag (UPOS): ADJ ADP ADV AUX CCONJ DET INTJ NOUN NUM PART PRON PROPN',
43
+ 'PUNCT SCONJ SYM VERB X. Return each token surface form exactly as it appears, in order,',
44
+ 'including punctuation as its own PUNCT token. Do not add or omit tokens.',
45
+ '',
46
+ 'Text:',
47
+ text,
48
+ ].join('\n'),
49
+ }),
50
+ );
51
+ return assembleDocument(text, value.sentences);
52
+ }).pipe(Effect.provide(AiService.model(PARSE_MODEL)));
53
+
54
+ /** The pluggable parser contract consumed by the editor extension and pipeline. */
55
+ export type Parser = (text: string) => Promise<Document>;
56
+
57
+ export { stubParse };
@@ -0,0 +1,31 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import { describe, test } from 'vitest';
6
+
7
+ import { stubParse, stubTag } from './stub';
8
+
9
+ describe('stubTag', () => {
10
+ test('tags closed-class words from the lexicon and splits sentences', ({ expect }) => {
11
+ const sentences = stubTag('The dog runs. It barks!');
12
+ expect(sentences).toHaveLength(2);
13
+ const first = sentences[0].tokens;
14
+ expect(first.find((t) => t.text === 'The')?.upos).toBe('DET');
15
+ expect(first.find((t) => t.text === '.')?.upos).toBe('PUNCT');
16
+ });
17
+
18
+ test('tags capitalized non-initial words as PROPN', ({ expect }) => {
19
+ const [sentence] = stubTag('I met Alice today');
20
+ expect(sentence.tokens.find((t) => t.text === 'Alice')?.upos).toBe('PROPN');
21
+ });
22
+ });
23
+
24
+ describe('stubParse', () => {
25
+ test('returns an aligned Document with exact offsets', async ({ expect }) => {
26
+ const source = 'The dog runs.';
27
+ const doc = await stubParse(source);
28
+ expect(doc.sourceHash).toBeTypeOf('string');
29
+ expect(doc.sentences[0].tokens.map((t) => source.slice(t.start, t.end))).toContain('dog');
30
+ });
31
+ });
package/src/stub.ts ADDED
@@ -0,0 +1,110 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ import { assembleDocument } from './align';
6
+ import { type Document, type RawSentence, type Upos } from './Document';
7
+
8
+ // Closed-class lexicon: small, deterministic, language-is-English assumption (the stub is a demo
9
+ // fallback, not the production tagger). Lowercased keys.
10
+ const LEXICON: Record<string, Upos> = {
11
+ the: 'DET',
12
+ a: 'DET',
13
+ an: 'DET',
14
+ this: 'DET',
15
+ that: 'DET',
16
+ these: 'DET',
17
+ those: 'DET',
18
+ i: 'PRON',
19
+ you: 'PRON',
20
+ he: 'PRON',
21
+ she: 'PRON',
22
+ it: 'PRON',
23
+ we: 'PRON',
24
+ they: 'PRON',
25
+ is: 'AUX',
26
+ am: 'AUX',
27
+ are: 'AUX',
28
+ was: 'AUX',
29
+ were: 'AUX',
30
+ be: 'AUX',
31
+ been: 'AUX',
32
+ do: 'AUX',
33
+ did: 'AUX',
34
+ in: 'ADP',
35
+ on: 'ADP',
36
+ at: 'ADP',
37
+ of: 'ADP',
38
+ to: 'ADP',
39
+ over: 'ADP',
40
+ under: 'ADP',
41
+ with: 'ADP',
42
+ for: 'ADP',
43
+ and: 'CCONJ',
44
+ or: 'CCONJ',
45
+ but: 'CCONJ',
46
+ because: 'SCONJ',
47
+ if: 'SCONJ',
48
+ while: 'SCONJ',
49
+ although: 'SCONJ',
50
+ not: 'PART',
51
+ very: 'ADV',
52
+ quickly: 'ADV',
53
+ well: 'ADV',
54
+ oh: 'INTJ',
55
+ yes: 'INTJ',
56
+ no: 'INTJ',
57
+ };
58
+
59
+ const WORD_RE = /[A-Za-z]+(?:'[A-Za-z]+)?|[0-9]+|[.!?,;:]/g;
60
+
61
+ /** Tag one token by lexicon → number → suffix heuristic → capitalization. `initial` = sentence start. */
62
+ const tagWord = (raw: string, initial: boolean): Upos => {
63
+ if (/^[.!?,;:]$/.test(raw)) {
64
+ return 'PUNCT';
65
+ }
66
+ if (/^[0-9]+$/.test(raw)) {
67
+ return 'NUM';
68
+ }
69
+ const lower = raw.toLowerCase();
70
+ if (LEXICON[lower]) {
71
+ return LEXICON[lower];
72
+ }
73
+ if (/^[A-Z]/.test(raw) && !initial) {
74
+ return 'PROPN';
75
+ }
76
+ if (/(ing|ed|ize|ise)$/.test(lower)) {
77
+ return 'VERB';
78
+ }
79
+ if (/(ly)$/.test(lower)) {
80
+ return 'ADV';
81
+ }
82
+ if (/(ous|ful|ive|able|al)$/.test(lower)) {
83
+ return 'ADJ';
84
+ }
85
+ return 'NOUN';
86
+ };
87
+
88
+ /** Deterministic UPOS tagger: splits on sentence-final punctuation, tags each token. */
89
+ export const stubTag = (text: string): RawSentence[] => {
90
+ const sentences: RawSentence[] = [];
91
+ let tokens: { text: string; upos: Upos }[] = [];
92
+ let initial = true;
93
+ for (const match of text.matchAll(WORD_RE)) {
94
+ const raw = match[0];
95
+ tokens.push({ text: raw, upos: tagWord(raw, initial) });
96
+ initial = false;
97
+ if (/^[.!?]$/.test(raw)) {
98
+ sentences.push({ tokens });
99
+ tokens = [];
100
+ initial = true;
101
+ }
102
+ }
103
+ if (tokens.length > 0) {
104
+ sentences.push({ tokens });
105
+ }
106
+ return sentences;
107
+ };
108
+
109
+ /** Parser-shaped wrapper around the stub tagger (async to match the `Parser` contract). */
110
+ export const stubParse = async (text: string): Promise<Document> => assembleDocument(text, stubTag(text));
@@ -0,0 +1,7 @@
1
+ //
2
+ // Copyright 2026 DXOS.org
3
+ //
4
+
5
+ // Import directly from `stub` (not `parse`) so the testing entrypoint stays isolated to the
6
+ // offline tagger and does not pull the live-parser/AI stack into story and test bundles.
7
+ export { stubParse } from '../stub';