@dxos/nlp 0.9.1-staging.ee54ba693a → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/lib/chunk-align.mjs +56 -0
  2. package/dist/lib/chunk-align.mjs.map +1 -0
  3. package/dist/lib/index.mjs +59 -0
  4. package/dist/lib/index.mjs.map +1 -0
  5. package/dist/lib/testing.mjs +92 -0
  6. package/dist/lib/testing.mjs.map +1 -0
  7. package/dist/types/src/align.d.ts.map +1 -1
  8. package/dist/types/src/index.d.ts +1 -1
  9. package/dist/types/src/index.d.ts.map +1 -1
  10. package/dist/types/src/parse.d.ts +0 -2
  11. package/dist/types/src/parse.d.ts.map +1 -1
  12. package/dist/types/src/testing/index.d.ts +1 -1
  13. package/dist/types/src/testing/index.d.ts.map +1 -1
  14. package/dist/types/src/{stub.d.ts → testing/stub.d.ts} +1 -1
  15. package/dist/types/src/testing/stub.d.ts.map +1 -0
  16. package/dist/types/src/testing/stub.test.d.ts.map +1 -0
  17. package/dist/types/tsconfig.tsbuildinfo +1 -1
  18. package/package.json +4 -4
  19. package/src/align.ts +7 -1
  20. package/src/index.ts +1 -1
  21. package/src/parse.test.ts +2 -1
  22. package/src/parse.ts +0 -3
  23. package/src/testing/index.ts +1 -1
  24. package/src/{stub.ts → testing/stub.ts} +9 -5
  25. package/dist/lib/neutral/chunk-D4MHDU46.mjs +0 -156
  26. package/dist/lib/neutral/chunk-D4MHDU46.mjs.map +0 -7
  27. package/dist/lib/neutral/index.mjs +0 -78
  28. package/dist/lib/neutral/index.mjs.map +0 -7
  29. package/dist/lib/neutral/meta.json +0 -1
  30. package/dist/lib/neutral/testing/index.mjs +0 -7
  31. package/dist/lib/neutral/testing/index.mjs.map +0 -7
  32. package/dist/types/src/stub.d.ts.map +0 -1
  33. package/dist/types/src/stub.test.d.ts.map +0 -1
  34. /package/dist/types/src/{stub.test.d.ts → testing/stub.test.d.ts} +0 -0
  35. /package/src/{stub.test.ts → testing/stub.test.ts} +0 -0
@@ -0,0 +1,56 @@
1
+ //#region src/hash.ts
2
+ /**
3
+ * Fast non-cryptographic hash (FNV-1a, 32-bit) of source text. Used purely to detect whether an
4
+ * analyzed span still matches the current editor text — change detection, not security.
5
+ */
6
+ var sourceHash = (text) => {
7
+ let hash = 2166136261;
8
+ for (let index = 0; index < text.length; index++) {
9
+ hash ^= text.charCodeAt(index);
10
+ hash = Math.imul(hash, 16777619);
11
+ }
12
+ return (hash >>> 0).toString(16).padStart(8, "0");
13
+ };
14
+ //#endregion
15
+ //#region src/align.ts
16
+ /**
17
+ * Align offset-free tagger output against the source text to compute exact character offsets.
18
+ * A single forward cursor guarantees repeated surface forms map to successive occurrences rather
19
+ * than re-matching the first. Tokens whose surface form cannot be located ahead of the cursor are
20
+ * dropped (the tagger hallucinated a token), keeping offsets internally consistent.
21
+ */
22
+ var assembleDocument = (sourceText, rawSentences) => {
23
+ let cursor = 0;
24
+ const sentences = [];
25
+ rawSentences.forEach((raw, sentenceIndex) => {
26
+ const tokens = [];
27
+ for (const { text, upos } of raw.tokens) {
28
+ const start = sourceText.indexOf(text, cursor);
29
+ if (start < 0) continue;
30
+ const end = start + text.length;
31
+ tokens.push({
32
+ index: tokens.length,
33
+ text,
34
+ upos,
35
+ start,
36
+ end
37
+ });
38
+ cursor = end;
39
+ }
40
+ if (tokens.length > 0) sentences.push({
41
+ index: sentenceIndex,
42
+ start: tokens[0].start,
43
+ end: tokens[tokens.length - 1].end,
44
+ tokens
45
+ });
46
+ });
47
+ return {
48
+ sourceHash: sourceHash(sourceText),
49
+ sentences,
50
+ timestamp: void 0
51
+ };
52
+ };
53
+ //#endregion
54
+ export { sourceHash as n, assembleDocument as t };
55
+
56
+ //# sourceMappingURL=chunk-align.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"chunk-align.mjs","names":[],"sources":["../../src/hash.ts","../../src/align.ts"],"sourcesContent":["//\n// Copyright 2026 DXOS.org\n//\n\n/**\n * Fast non-cryptographic hash (FNV-1a, 32-bit) of source text. Used purely to detect whether an\n * analyzed span still matches the current editor text — change detection, not security.\n */\nexport const sourceHash = (text: string): string => {\n let hash = 0x811c9dc5;\n for (let index = 0; index < text.length; index++) {\n hash ^= text.charCodeAt(index);\n hash = Math.imul(hash, 0x01000193);\n }\n return (hash >>> 0).toString(16).padStart(8, '0');\n};\n","//\n// Copyright 2026 DXOS.org\n//\n\nimport { type Document, type RawSentence, type Sentence, type Token } from './Document';\nimport { sourceHash } from './hash';\n\n/**\n * Align offset-free tagger output against the source text to compute exact character offsets.\n * A single forward cursor guarantees repeated surface forms map to successive occurrences rather\n * than re-matching the first. Tokens whose surface form cannot be located ahead of the cursor are\n * dropped (the tagger hallucinated a token), keeping offsets internally consistent.\n */\nexport const assembleDocument = (sourceText: string, rawSentences: readonly RawSentence[]): Document => {\n let cursor = 0;\n const sentences: Sentence[] = [];\n\n rawSentences.forEach((raw, sentenceIndex) => {\n const tokens: Token[] = [];\n for (const { text, upos } of raw.tokens) {\n const start = sourceText.indexOf(text, cursor);\n if (start < 0) {\n continue;\n }\n\n const end = start + text.length;\n tokens.push({ index: tokens.length, text, upos, start, end });\n cursor = end;\n }\n\n if (tokens.length > 0) {\n sentences.push({ index: sentenceIndex, start: tokens[0].start, end: tokens[tokens.length - 1].end, tokens });\n }\n });\n\n return {\n sourceHash: sourceHash(sourceText),\n sentences,\n timestamp: undefined,\n };\n};\n"],"mappings":";;;;;AAQA,IAAa,cAAc,SAAyB;CAClD,IAAI,OAAO;CACX,KAAK,IAAI,QAAQ,GAAG,QAAQ,KAAK,QAAQ,SAAS;EAChD,QAAQ,KAAK,WAAW,KAAK;EAC7B,OAAO,KAAK,KAAK,MAAM,QAAU;CACnC;CACA,QAAQ,SAAS,EAAA,CAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG;AAClD;;;;;;;;;ACFA,IAAa,oBAAoB,YAAoB,iBAAmD;CACtG,IAAI,SAAS;CACb,MAAM,YAAwB,CAAC;CAE/B,aAAa,SAAS,KAAK,kBAAkB;EAC3C,MAAM,SAAkB,CAAC;EACzB,KAAK,MAAM,EAAE,MAAM,UAAU,IAAI,QAAQ;GACvC,MAAM,QAAQ,WAAW,QAAQ,MAAM,MAAM;GAC7C,IAAI,QAAQ,GACV;GAGF,MAAM,MAAM,QAAQ,KAAK;GACzB,OAAO,KAAK;IAAE,OAAO,OAAO;IAAQ;IAAM;IAAM;IAAO;GAAI,CAAC;GAC5D,SAAS;EACX;EAEA,IAAI,OAAO,SAAS,GAClB,UAAU,KAAK;GAAE,OAAO;GAAe,OAAO,OAAO,EAAE,CAAC;GAAO,KAAK,OAAO,OAAO,SAAS,EAAE,CAAC;GAAK;EAAO,CAAC;CAE/G,CAAC;CAED,OAAO;EACL,YAAY,WAAW,UAAU;EACjC;EACA,WAAW,KAAA;CACb;AACF"}
@@ -0,0 +1,59 @@
1
+ import { n as sourceHash, t as assembleDocument } from "./chunk-align.mjs";
2
+ import * as Schema from "effect/Schema";
3
+ import * as LanguageModel from "@effect/ai/LanguageModel";
4
+ import * as Effect from "effect/Effect";
5
+ import { AiService } from "@dxos/ai";
6
+ //#region src/Document.ts
7
+ /** Universal POS tagset (17 tags). https://universaldependencies.org/u/pos/ */
8
+ var Upos = Schema.Literal("ADJ", "ADP", "ADV", "AUX", "CCONJ", "DET", "INTJ", "NOUN", "NUM", "PART", "PRON", "PROPN", "PUNCT", "SCONJ", "SYM", "VERB", "X");
9
+ /** A single word/punctuation token. `start`/`end` are character offsets within the source text. */
10
+ var Token = Schema.Struct({
11
+ index: Schema.Number.annotations({ description: "Position of the token within its sentence." }),
12
+ text: Schema.String.annotations({ description: "Surface form exactly as it appears in the source." }),
13
+ upos: Upos.annotations({ description: "Universal part-of-speech tag." }),
14
+ start: Schema.Number,
15
+ end: Schema.Number
16
+ });
17
+ var Sentence = Schema.Struct({
18
+ index: Schema.Number,
19
+ start: Schema.Number,
20
+ end: Schema.Number,
21
+ tokens: Schema.Array(Token)
22
+ });
23
+ /** A parsed document. `sourceHash` is the divergence signal; `timestamp` is debug-only. */
24
+ var Document = Schema.Struct({
25
+ sourceHash: Schema.String,
26
+ sentences: Schema.Array(Sentence),
27
+ timestamp: Schema.optional(Schema.Number)
28
+ });
29
+ //#endregion
30
+ //#region src/parse.ts
31
+ var PARSE_MODEL = "com.anthropic.model.claude-haiku-4-5.default";
32
+ /** LLM output schema: sentences → tokens, no offsets (alignment computes those). */
33
+ var TaggedSentences = Schema.Struct({ sentences: Schema.Array(Schema.Struct({ tokens: Schema.Array(Schema.Struct({
34
+ text: Schema.String.annotations({ description: "Token surface form exactly as in the source." }),
35
+ upos: Upos.annotations({ description: "Universal POS tag for the token." })
36
+ })) })) });
37
+ /**
38
+ * Tag `text` with UPOS via a small LLM, then deterministically align tokens to source offsets.
39
+ * Provides the LanguageModel internally; residual requirement is {@link AiService.AiService}.
40
+ */
41
+ var parseText = (text) => Effect.gen(function* () {
42
+ const { value } = yield* Effect.scoped(LanguageModel.generateObject({
43
+ schema: TaggedSentences,
44
+ prompt: [
45
+ "Tokenize the text below into sentences and tokens, and tag each token with its",
46
+ "Universal POS tag (UPOS): ADJ ADP ADV AUX CCONJ DET INTJ NOUN NUM PART PRON PROPN",
47
+ "PUNCT SCONJ SYM VERB X. Return each token surface form exactly as it appears, in order,",
48
+ "including punctuation as its own PUNCT token. Do not add or omit tokens.",
49
+ "",
50
+ "Text:",
51
+ text
52
+ ].join("\n")
53
+ }));
54
+ return assembleDocument(text, value.sentences);
55
+ }).pipe(Effect.provide(AiService.model(PARSE_MODEL)));
56
+ //#endregion
57
+ export { Document, Sentence, Token, Upos, assembleDocument, parseText, sourceHash };
58
+
59
+ //# sourceMappingURL=index.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.mjs","names":[],"sources":["../../src/Document.ts","../../src/parse.ts"],"sourcesContent":["//\n// Copyright 2026 DXOS.org\n//\n\nimport * as Schema from 'effect/Schema';\n\n/** Universal POS tagset (17 tags). https://universaldependencies.org/u/pos/ */\nexport const Upos = Schema.Literal(\n 'ADJ',\n 'ADP',\n 'ADV',\n 'AUX',\n 'CCONJ',\n 'DET',\n 'INTJ',\n 'NOUN',\n 'NUM',\n 'PART',\n 'PRON',\n 'PROPN',\n 'PUNCT',\n 'SCONJ',\n 'SYM',\n 'VERB',\n 'X',\n);\nexport type Upos = Schema.Schema.Type<typeof Upos>;\n\n/** A single word/punctuation token. `start`/`end` are character offsets within the source text. */\nexport const Token = Schema.Struct({\n index: Schema.Number.annotations({ description: 'Position of the token within its sentence.' }),\n text: Schema.String.annotations({ description: 'Surface form exactly as it appears in the source.' }),\n upos: Upos.annotations({ description: 'Universal part-of-speech tag.' }),\n start: Schema.Number,\n end: Schema.Number,\n});\nexport type Token = Schema.Schema.Type<typeof Token>;\n\nexport const Sentence = Schema.Struct({\n index: Schema.Number,\n start: Schema.Number,\n end: Schema.Number,\n tokens: Schema.Array(Token),\n});\nexport type Sentence = Schema.Schema.Type<typeof Sentence>;\n\n/** A parsed document. `sourceHash` is the divergence signal; `timestamp` is debug-only. */\nexport const Document = Schema.Struct({\n sourceHash: Schema.String,\n sentences: Schema.Array(Sentence),\n timestamp: Schema.optional(Schema.Number),\n});\nexport type Document = Schema.Schema.Type<typeof Document>;\n\n/** Raw, offset-free output of a tagger before alignment. */\nexport type RawSentence = { readonly tokens: readonly { readonly text: string; readonly upos: Upos }[] };\n","//\n// Copyright 2026 DXOS.org\n//\n\nimport * as LanguageModel from '@effect/ai/LanguageModel';\nimport * as Effect from 'effect/Effect';\nimport * as Schema from 'effect/Schema';\n\nimport { AiService } from '@dxos/ai';\n\nimport { assembleDocument } from './align';\nimport { type Document, Upos } from './Document';\n\nconst PARSE_MODEL = 'com.anthropic.model.claude-haiku-4-5.default';\n\n/** LLM output schema: sentences → tokens, no offsets (alignment computes those). */\nconst TaggedSentences = Schema.Struct({\n sentences: Schema.Array(\n Schema.Struct({\n tokens: Schema.Array(\n Schema.Struct({\n text: Schema.String.annotations({ description: 'Token surface form exactly as in the source.' }),\n upos: Upos.annotations({ description: 'Universal POS tag for the token.' }),\n }),\n ),\n }),\n ),\n});\n\n/**\n * Tag `text` with UPOS via a small LLM, then deterministically align tokens to source offsets.\n * Provides the LanguageModel internally; residual requirement is {@link AiService.AiService}.\n */\nexport const parseText = (text: string) =>\n Effect.gen(function* () {\n const { value } = yield* Effect.scoped(\n LanguageModel.generateObject({\n schema: TaggedSentences,\n prompt: [\n 'Tokenize the text below into sentences and tokens, and tag each token with its',\n 'Universal POS tag (UPOS): ADJ ADP ADV AUX CCONJ DET INTJ NOUN NUM PART PRON PROPN',\n 'PUNCT SCONJ SYM VERB X. Return each token surface form exactly as it appears, in order,',\n 'including punctuation as its own PUNCT token. Do not add or omit tokens.',\n '',\n 'Text:',\n text,\n ].join('\\n'),\n }),\n );\n return assembleDocument(text, value.sentences);\n }).pipe(Effect.provide(AiService.model(PARSE_MODEL)));\n\n/** The pluggable parser contract consumed by the editor extension and pipeline. */\nexport type Parser = (text: string) => Promise<Document>;\n"],"mappings":";;;;;;;AAOA,IAAa,OAAO,OAAO,QACzB,OACA,OACA,OACA,OACA,SACA,OACA,QACA,QACA,OACA,QACA,QACA,SACA,SACA,SACA,OACA,QACA,GACF;;AAIA,IAAa,QAAQ,OAAO,OAAO;CACjC,OAAO,OAAO,OAAO,YAAY,EAAE,aAAa,6CAA6C,CAAC;CAC9F,MAAM,OAAO,OAAO,YAAY,EAAE,aAAa,oDAAoD,CAAC;CACpG,MAAM,KAAK,YAAY,EAAE,aAAa,gCAAgC,CAAC;CACvE,OAAO,OAAO;CACd,KAAK,OAAO;AACd,CAAC;AAGD,IAAa,WAAW,OAAO,OAAO;CACpC,OAAO,OAAO;CACd,OAAO,OAAO;CACd,KAAK,OAAO;CACZ,QAAQ,OAAO,MAAM,KAAK;AAC5B,CAAC;;AAID,IAAa,WAAW,OAAO,OAAO;CACpC,YAAY,OAAO;CACnB,WAAW,OAAO,MAAM,QAAQ;CAChC,WAAW,OAAO,SAAS,OAAO,MAAM;AAC1C,CAAC;;;ACtCD,IAAM,cAAc;;AAGpB,IAAM,kBAAkB,OAAO,OAAO,EACpC,WAAW,OAAO,MAChB,OAAO,OAAO,EACZ,QAAQ,OAAO,MACb,OAAO,OAAO;CACZ,MAAM,OAAO,OAAO,YAAY,EAAE,aAAa,+CAA+C,CAAC;CAC/F,MAAM,KAAK,YAAY,EAAE,aAAa,mCAAmC,CAAC;AAC5E,CAAC,CACH,EACF,CAAC,CACH,EACF,CAAC;;;;;AAMD,IAAa,aAAa,SACxB,OAAO,IAAI,aAAa;CACtB,MAAM,EAAE,UAAU,OAAO,OAAO,OAC9B,cAAc,eAAe;EAC3B,QAAQ;EACR,QAAQ;GACN;GACA;GACA;GACA;GACA;GACA;GACA;EACF,CAAC,CAAC,KAAK,IAAI;CACb,CAAC,CACH;CACA,OAAO,iBAAiB,MAAM,MAAM,SAAS;AAC/C,CAAC,CAAC,CAAC,KAAK,OAAO,QAAQ,UAAU,MAAM,WAAW,CAAC,CAAC"}
@@ -0,0 +1,92 @@
1
+ import { t as assembleDocument } from "./chunk-align.mjs";
2
+ //#region src/testing/stub.ts
3
+ var LEXICON = {
4
+ the: "DET",
5
+ a: "DET",
6
+ an: "DET",
7
+ this: "DET",
8
+ that: "DET",
9
+ these: "DET",
10
+ those: "DET",
11
+ i: "PRON",
12
+ you: "PRON",
13
+ he: "PRON",
14
+ she: "PRON",
15
+ it: "PRON",
16
+ we: "PRON",
17
+ they: "PRON",
18
+ is: "AUX",
19
+ am: "AUX",
20
+ are: "AUX",
21
+ was: "AUX",
22
+ were: "AUX",
23
+ be: "AUX",
24
+ been: "AUX",
25
+ do: "AUX",
26
+ did: "AUX",
27
+ in: "ADP",
28
+ on: "ADP",
29
+ at: "ADP",
30
+ of: "ADP",
31
+ to: "ADP",
32
+ over: "ADP",
33
+ under: "ADP",
34
+ with: "ADP",
35
+ for: "ADP",
36
+ and: "CCONJ",
37
+ or: "CCONJ",
38
+ but: "CCONJ",
39
+ because: "SCONJ",
40
+ if: "SCONJ",
41
+ while: "SCONJ",
42
+ although: "SCONJ",
43
+ not: "PART",
44
+ very: "ADV",
45
+ quickly: "ADV",
46
+ well: "ADV",
47
+ oh: "INTJ",
48
+ yes: "INTJ",
49
+ no: "INTJ"
50
+ };
51
+ var WORD_RE = /[A-Za-z]+(?:'[A-Za-z]+)?|[0-9]+|[.!?,;:]/g;
52
+ /**
53
+ * Fake tags one token by lexicon → number → suffix heuristic → capitalization. `initial` = sentence start.
54
+ */
55
+ var tagWord = (raw, initial) => {
56
+ if (/^[.!?,;:]$/.test(raw)) return "PUNCT";
57
+ if (/^[0-9]+$/.test(raw)) return "NUM";
58
+ const lower = raw.toLowerCase();
59
+ if (LEXICON[lower]) return LEXICON[lower];
60
+ if (/^[A-Z]/.test(raw) && !initial) return "PROPN";
61
+ if (/(ing|ed|ize|ise)$/.test(lower)) return "VERB";
62
+ if (/(ly)$/.test(lower)) return "ADV";
63
+ if (/(ous|ful|ive|able|al)$/.test(lower)) return "ADJ";
64
+ return "NOUN";
65
+ };
66
+ /** Deterministic UPOS tagger: splits on sentence-final punctuation, tags each token. */
67
+ var stubTag = (text) => {
68
+ const sentences = [];
69
+ let tokens = [];
70
+ let initial = true;
71
+ for (const match of text.matchAll(WORD_RE)) {
72
+ const raw = match[0];
73
+ tokens.push({
74
+ text: raw,
75
+ upos: tagWord(raw, initial)
76
+ });
77
+ initial = false;
78
+ if (/^[.!?]$/.test(raw)) {
79
+ sentences.push({ tokens });
80
+ tokens = [];
81
+ initial = true;
82
+ }
83
+ }
84
+ if (tokens.length > 0) sentences.push({ tokens });
85
+ return sentences;
86
+ };
87
+ /** Parser-shaped wrapper around the stub tagger (async to match the `Parser` contract). */
88
+ var stubParse = async (text) => assembleDocument(text, stubTag(text));
89
+ //#endregion
90
+ export { stubParse };
91
+
92
+ //# sourceMappingURL=testing.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"testing.mjs","names":[],"sources":["../../src/testing/stub.ts"],"sourcesContent":["//\n// Copyright 2026 DXOS.org\n//\n\nimport { assembleDocument } from '../align';\nimport { type Document, type RawSentence, type Upos } from '../Document';\n\n// Closed-class lexicon: small, deterministic, language-is-English assumption\n// (the stub is a demo fallback, not the production tagger). Lowercased keys.\nconst LEXICON: Record<string, Upos> = {\n the: 'DET',\n a: 'DET',\n an: 'DET',\n this: 'DET',\n that: 'DET',\n these: 'DET',\n those: 'DET',\n i: 'PRON',\n you: 'PRON',\n he: 'PRON',\n she: 'PRON',\n it: 'PRON',\n we: 'PRON',\n they: 'PRON',\n is: 'AUX',\n am: 'AUX',\n are: 'AUX',\n was: 'AUX',\n were: 'AUX',\n be: 'AUX',\n been: 'AUX',\n do: 'AUX',\n did: 'AUX',\n in: 'ADP',\n on: 'ADP',\n at: 'ADP',\n of: 'ADP',\n to: 'ADP',\n over: 'ADP',\n under: 'ADP',\n with: 'ADP',\n for: 'ADP',\n and: 'CCONJ',\n or: 'CCONJ',\n but: 'CCONJ',\n because: 'SCONJ',\n if: 'SCONJ',\n while: 'SCONJ',\n although: 'SCONJ',\n not: 'PART',\n very: 'ADV',\n quickly: 'ADV',\n well: 'ADV',\n oh: 'INTJ',\n yes: 'INTJ',\n no: 'INTJ',\n};\n\nconst WORD_RE = /[A-Za-z]+(?:'[A-Za-z]+)?|[0-9]+|[.!?,;:]/g;\n\n/**\n * Fake tags one token by lexicon → number → suffix heuristic → capitalization. `initial` = sentence start.\n */\nconst tagWord = (raw: string, initial: boolean): Upos => {\n if (/^[.!?,;:]$/.test(raw)) {\n return 'PUNCT';\n }\n if (/^[0-9]+$/.test(raw)) {\n return 'NUM';\n }\n const lower = raw.toLowerCase();\n if (LEXICON[lower]) {\n return LEXICON[lower];\n }\n if (/^[A-Z]/.test(raw) && !initial) {\n return 'PROPN';\n }\n if (/(ing|ed|ize|ise)$/.test(lower)) {\n return 'VERB';\n }\n if (/(ly)$/.test(lower)) {\n return 'ADV';\n }\n if (/(ous|ful|ive|able|al)$/.test(lower)) {\n return 'ADJ';\n }\n return 'NOUN';\n};\n\n/** Deterministic UPOS tagger: splits on sentence-final punctuation, tags each token. */\nexport const stubTag = (text: string): RawSentence[] => {\n const sentences: RawSentence[] = [];\n let tokens: { text: string; upos: Upos }[] = [];\n let initial = true;\n for (const match of text.matchAll(WORD_RE)) {\n const raw = match[0];\n tokens.push({ text: raw, upos: tagWord(raw, initial) });\n initial = false;\n if (/^[.!?]$/.test(raw)) {\n sentences.push({ tokens });\n tokens = [];\n initial = true;\n }\n }\n\n if (tokens.length > 0) {\n sentences.push({ tokens });\n }\n\n return sentences;\n};\n\n/** Parser-shaped wrapper around the stub tagger (async to match the `Parser` contract). */\nexport const stubParse = async (text: string): Promise<Document> => assembleDocument(text, stubTag(text));\n"],"mappings":";;AASA,IAAM,UAAgC;CACpC,KAAK;CACL,GAAG;CACH,IAAI;CACJ,MAAM;CACN,MAAM;CACN,OAAO;CACP,OAAO;CACP,GAAG;CACH,KAAK;CACL,IAAI;CACJ,KAAK;CACL,IAAI;CACJ,IAAI;CACJ,MAAM;CACN,IAAI;CACJ,IAAI;CACJ,KAAK;CACL,KAAK;CACL,MAAM;CACN,IAAI;CACJ,MAAM;CACN,IAAI;CACJ,KAAK;CACL,IAAI;CACJ,IAAI;CACJ,IAAI;CACJ,IAAI;CACJ,IAAI;CACJ,MAAM;CACN,OAAO;CACP,MAAM;CACN,KAAK;CACL,KAAK;CACL,IAAI;CACJ,KAAK;CACL,SAAS;CACT,IAAI;CACJ,OAAO;CACP,UAAU;CACV,KAAK;CACL,MAAM;CACN,SAAS;CACT,MAAM;CACN,IAAI;CACJ,KAAK;CACL,IAAI;AACN;AAEA,IAAM,UAAU;;;;AAKhB,IAAM,WAAW,KAAa,YAA2B;CACvD,IAAI,aAAa,KAAK,GAAG,GACvB,OAAO;CAET,IAAI,WAAW,KAAK,GAAG,GACrB,OAAO;CAET,MAAM,QAAQ,IAAI,YAAY;CAC9B,IAAI,QAAQ,QACV,OAAO,QAAQ;CAEjB,IAAI,SAAS,KAAK,GAAG,KAAK,CAAC,SACzB,OAAO;CAET,IAAI,oBAAoB,KAAK,KAAK,GAChC,OAAO;CAET,IAAI,QAAQ,KAAK,KAAK,GACpB,OAAO;CAET,IAAI,yBAAyB,KAAK,KAAK,GACrC,OAAO;CAET,OAAO;AACT;;AAGA,IAAa,WAAW,SAAgC;CACtD,MAAM,YAA2B,CAAC;CAClC,IAAI,SAAyC,CAAC;CAC9C,IAAI,UAAU;CACd,KAAK,MAAM,SAAS,KAAK,SAAS,OAAO,GAAG;EAC1C,MAAM,MAAM,MAAM;EAClB,OAAO,KAAK;GAAE,MAAM;GAAK,MAAM,QAAQ,KAAK,OAAO;EAAE,CAAC;EACtD,UAAU;EACV,IAAI,UAAU,KAAK,GAAG,GAAG;GACvB,UAAU,KAAK,EAAE,OAAO,CAAC;GACzB,SAAS,CAAC;GACV,UAAU;EACZ;CACF;CAEA,IAAI,OAAO,SAAS,GAClB,UAAU,KAAK,EAAE,OAAO,CAAC;CAG3B,OAAO;AACT;;AAGA,IAAa,YAAY,OAAO,SAAoC,iBAAiB,MAAM,QAAQ,IAAI,CAAC"}
@@ -1 +1 @@
1
- {"version":3,"file":"align.d.ts","sourceRoot":"","sources":["../../../src/align.ts"],"names":[],"mappings":"AAIA,OAAO,EAAE,KAAK,QAAQ,EAAE,KAAK,WAAW,EAA6B,MAAM,YAAY,CAAC;AAGxF;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,eAAgB,MAAM,gBAAgB,SAAS,WAAW,EAAE,KAAG,QAqB3F,CAAC"}
1
+ {"version":3,"file":"align.d.ts","sourceRoot":"","sources":["../../../src/align.ts"],"names":[],"mappings":"AAIA,OAAO,EAAE,KAAK,QAAQ,EAAE,KAAK,WAAW,EAA6B,MAAM,YAAY,CAAC;AAGxF;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,eAAgB,MAAM,gBAAgB,SAAS,WAAW,EAAE,KAAG,QA2B3F,CAAC"}
@@ -1,5 +1,5 @@
1
1
  export * from './Document';
2
2
  export * from './align';
3
3
  export * from './hash';
4
- export { type Parser, parseText, stubParse } from './parse';
4
+ export { type Parser, parseText } from './parse';
5
5
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../src/index.ts"],"names":[],"mappings":"AAIA,cAAc,YAAY,CAAC;AAC3B,cAAc,SAAS,CAAC;AACxB,cAAc,QAAQ,CAAC;AACvB,OAAO,EAAE,KAAK,MAAM,EAAE,SAAS,EAAE,SAAS,EAAE,MAAM,SAAS,CAAC"}
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../src/index.ts"],"names":[],"mappings":"AAIA,cAAc,YAAY,CAAC;AAC3B,cAAc,SAAS,CAAC;AACxB,cAAc,QAAQ,CAAC;AACvB,OAAO,EAAE,KAAK,MAAM,EAAE,SAAS,EAAE,MAAM,SAAS,CAAC"}
@@ -1,7 +1,6 @@
1
1
  import * as Effect from 'effect/Effect';
2
2
  import { AiService } from '@dxos/ai';
3
3
  import { type Document } from './Document';
4
- import { stubParse } from './stub';
5
4
  /**
6
5
  * Tag `text` with UPOS via a small LLM, then deterministically align tokens to source offsets.
7
6
  * Provides the LanguageModel internally; residual requirement is {@link AiService.AiService}.
@@ -24,5 +23,4 @@ export declare const parseText: (text: string) => Effect.Effect<{
24
23
  }, import("@dxos/ai").AiModelNotAvailableError | import("@effect/ai/AiError").AiError, AiService.AiService>;
25
24
  /** The pluggable parser contract consumed by the editor extension and pipeline. */
26
25
  export type Parser = (text: string) => Promise<Document>;
27
- export { stubParse };
28
26
  //# sourceMappingURL=parse.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"parse.d.ts","sourceRoot":"","sources":["../../../src/parse.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,MAAM,MAAM,eAAe,CAAC;AAGxC,OAAO,EAAE,SAAS,EAAE,MAAM,UAAU,CAAC;AAGrC,OAAO,EAAE,KAAK,QAAQ,EAAQ,MAAM,YAAY,CAAC;AACjD,OAAO,EAAE,SAAS,EAAE,MAAM,QAAQ,CAAC;AAkBnC;;;GAGG;AACH,eAAO,MAAM,SAAS,SAAU,MAAM;;;;;;;;;;;;;;;2GAiBiB,CAAC;AAExD,mFAAmF;AACnF,MAAM,MAAM,MAAM,GAAG,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,QAAQ,CAAC,CAAC;AAEzD,OAAO,EAAE,SAAS,EAAE,CAAC"}
1
+ {"version":3,"file":"parse.d.ts","sourceRoot":"","sources":["../../../src/parse.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,MAAM,MAAM,eAAe,CAAC;AAGxC,OAAO,EAAE,SAAS,EAAE,MAAM,UAAU,CAAC;AAGrC,OAAO,EAAE,KAAK,QAAQ,EAAQ,MAAM,YAAY,CAAC;AAkBjD;;;GAGG;AACH,eAAO,MAAM,SAAS,SAAU,MAAM;;;;;;;;;;;;;;;2GAiBiB,CAAC;AAExD,mFAAmF;AACnF,MAAM,MAAM,MAAM,GAAG,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,QAAQ,CAAC,CAAC"}
@@ -1,2 +1,2 @@
1
- export { stubParse } from '../stub';
1
+ export { stubParse } from './stub';
2
2
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/testing/index.ts"],"names":[],"mappings":"AAMA,OAAO,EAAE,SAAS,EAAE,MAAM,SAAS,CAAC"}
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/testing/index.ts"],"names":[],"mappings":"AAMA,OAAO,EAAE,SAAS,EAAE,MAAM,QAAQ,CAAC"}
@@ -1,4 +1,4 @@
1
- import { type Document, type RawSentence } from './Document';
1
+ import { type Document, type RawSentence } from '../Document';
2
2
  /** Deterministic UPOS tagger: splits on sentence-final punctuation, tags each token. */
3
3
  export declare const stubTag: (text: string) => RawSentence[];
4
4
  /** Parser-shaped wrapper around the stub tagger (async to match the `Parser` contract). */
@@ -0,0 +1 @@
1
+ {"version":3,"file":"stub.d.ts","sourceRoot":"","sources":["../../../../src/testing/stub.ts"],"names":[],"mappings":"AAKA,OAAO,EAAE,KAAK,QAAQ,EAAE,KAAK,WAAW,EAAa,MAAM,aAAa,CAAC;AAoFzE,wFAAwF;AACxF,eAAO,MAAM,OAAO,SAAU,MAAM,KAAG,WAAW,EAoBjD,CAAC;AAEF,2FAA2F;AAC3F,eAAO,MAAM,SAAS,SAAgB,MAAM,KAAG,OAAO,CAAC,QAAQ,CAA0C,CAAC"}
@@ -0,0 +1 @@
1
+ {"version":3,"file":"stub.test.d.ts","sourceRoot":"","sources":["../../../../src/testing/stub.test.ts"],"names":[],"mappings":""}