@stratta/mcp 0.5.0 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -13
- package/dist/errors.d.ts +6 -0
- package/dist/errors.js +55 -0
- package/dist/index.js +69 -119
- package/dist/tools/define.d.ts +45 -0
- package/dist/tools/define.js +4 -0
- package/dist/tools/ingest.d.ts +140 -243
- package/dist/tools/ingest.js +225 -238
- package/dist/tools/read.d.ts +51 -0
- package/dist/tools/read.js +247 -0
- package/package.json +9 -9
- package/skills/ingest-norm/SKILL.md +140 -128
- package/dist/tools/get-cross-refs.d.ts +0 -24
- package/dist/tools/get-cross-refs.js +0 -23
- package/dist/tools/get-figure.d.ts +0 -24
- package/dist/tools/get-figure.js +0 -50
- package/dist/tools/get-methodology.d.ts +0 -18
- package/dist/tools/get-methodology.js +0 -21
- package/dist/tools/get-section.d.ts +0 -24
- package/dist/tools/get-section.js +0 -34
- package/dist/tools/get-subtree.d.ts +0 -29
- package/dist/tools/get-subtree.js +0 -39
- package/dist/tools/get-toc.d.ts +0 -24
- package/dist/tools/get-toc.js +0 -32
- package/dist/tools/list-norms.d.ts +0 -11
- package/dist/tools/list-norms.js +0 -17
- package/dist/tools/search-in-norm.d.ts +0 -29
- package/dist/tools/search-in-norm.js +0 -31
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { api } from '../client.js';
|
|
3
|
+
import { requireApiKey } from '../auth.js';
|
|
4
|
+
import { defineTool } from './define.js';
|
|
5
|
+
/**
|
|
6
|
+
* The eight read tools: how an agent navigates a norm.
|
|
7
|
+
*
|
|
8
|
+
* All of them are `readOnlyHint: true` — they never write, so a client is free
|
|
9
|
+
* to retry them and an agent is free to explore.
|
|
10
|
+
*/
|
|
11
|
+
const norm = z
|
|
12
|
+
.string()
|
|
13
|
+
.min(1)
|
|
14
|
+
.describe('Norm code, e.g. "SIA 261", "SIA 263", "EN 1992-1-1".');
|
|
15
|
+
const sectionPath = z
|
|
16
|
+
.string()
|
|
17
|
+
.min(1)
|
|
18
|
+
.describe('Section path as it appears in the document, e.g. "4.2.1" or "Annexe A".');
|
|
19
|
+
const readOnly = { readOnlyHint: true, openWorldHint: false };
|
|
20
|
+
export const getMethodology = defineTool({
|
|
21
|
+
name: 'get_methodology',
|
|
22
|
+
title: 'Get methodology',
|
|
23
|
+
description: 'MANDATORY FIRST CALL when answering any technical question about Swiss civil engineering norms via Stratta. Returns the canonical persona, navigation workflow, meta-routing hints (which SIA norms cover which topics), tree-navigation rules (where to look for formulas vs coefficients vs definitions), citation format and answer rules. Adopt these rules verbatim for the rest of the consultation. If you skip this call you WILL produce lower-quality answers (wrong citation format, missing cross-norm dependencies, hallucinated values).',
|
|
24
|
+
inputSchema: {
|
|
25
|
+
norm: norm
|
|
26
|
+
.optional()
|
|
27
|
+
.describe('Optional: norm code the user is asking about, if already known. Used to scope methodology hints (currently informational only).'),
|
|
28
|
+
},
|
|
29
|
+
annotations: readOnly,
|
|
30
|
+
run: (client, args) => client.query(api._mcp.getMethodology, { norm: args.norm }),
|
|
31
|
+
});
|
|
32
|
+
export const listNorms = defineTool({
|
|
33
|
+
name: 'list_norms',
|
|
34
|
+
title: 'List norms',
|
|
35
|
+
description: "List all engineering norms (SIA, Eurocodes, etc.) available in YOUR workspace. Returns code, year, title, and language for each. Workflow: call get_methodology FIRST (persona + rules), then list_norms to know what is queryable, then meta-route the user's question to the relevant norms before drilling in.",
|
|
36
|
+
inputSchema: {},
|
|
37
|
+
annotations: readOnly,
|
|
38
|
+
run: async (client) => ({
|
|
39
|
+
norms: await client.action(api._mcp.listPublishedNorms, {
|
|
40
|
+
apiKey: requireApiKey(),
|
|
41
|
+
}),
|
|
42
|
+
}),
|
|
43
|
+
});
|
|
44
|
+
export const getToc = defineTool({
|
|
45
|
+
name: 'get_toc',
|
|
46
|
+
title: 'Get table of contents',
|
|
47
|
+
description: "Get the high-level table of contents for a norm. By default returns only top-level chapters (depth=1) to stay light. Call get_subtree on a specific chapter's path to drill into sections + subsections. Increase maxDepth if you need a wider overview (cost: response size grows fast). Each node has nodeId, path, title, summary, pageStart, pageEnd, depth, and children (empty at the maxDepth boundary).",
|
|
48
|
+
inputSchema: {
|
|
49
|
+
norm,
|
|
50
|
+
maxDepth: z
|
|
51
|
+
.number()
|
|
52
|
+
.int()
|
|
53
|
+
.min(1)
|
|
54
|
+
.max(6)
|
|
55
|
+
.optional()
|
|
56
|
+
.describe('Maximum nesting depth to include. Defaults to 1 (chapters only). depth=2 includes sections X.Y. depth=3 includes sub-subsections X.Y.Z (may exceed response size limit on large norms).'),
|
|
57
|
+
},
|
|
58
|
+
annotations: readOnly,
|
|
59
|
+
run: async (client, args) => {
|
|
60
|
+
const tree = await client.action(api._mcp.getDocumentToc, {
|
|
61
|
+
apiKey: requireApiKey(),
|
|
62
|
+
code: args.norm,
|
|
63
|
+
maxDepth: args.maxDepth,
|
|
64
|
+
});
|
|
65
|
+
// An unknown or unpublished norm comes back empty. Saying so beats
|
|
66
|
+
// handing the agent an empty array it will read as "this norm has no
|
|
67
|
+
// chapters" and then answer from memory.
|
|
68
|
+
if (!tree || (Array.isArray(tree) && tree.length === 0)) {
|
|
69
|
+
return {
|
|
70
|
+
error: `Norm "${args.norm}" not found in your workspace, or not published yet. Call list_norms to see what is available.`,
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
return { norm: args.norm, tree };
|
|
74
|
+
},
|
|
75
|
+
});
|
|
76
|
+
export const getSubtree = defineTool({
|
|
77
|
+
name: 'get_subtree',
|
|
78
|
+
title: 'Get subtree',
|
|
79
|
+
description: 'Drill down into a specific chapter or section. Returns the subtree rooted at `path` with optional depth limit (relative to the root). Use this after get_toc to explore one chapter in detail without fetching the entire TOC. Each node has nodeId, path, title, summary, depth, pageStart, pageEnd, and recursive children.',
|
|
80
|
+
inputSchema: {
|
|
81
|
+
norm,
|
|
82
|
+
path: sectionPath.describe('Section path to root the subtree at, e.g. "14" or "14.2".'),
|
|
83
|
+
maxDepth: z
|
|
84
|
+
.number()
|
|
85
|
+
.int()
|
|
86
|
+
.min(1)
|
|
87
|
+
.max(6)
|
|
88
|
+
.optional()
|
|
89
|
+
.describe('Max nesting depth relative to the root. Default: unlimited. depth=1 returns the root + direct children only.'),
|
|
90
|
+
},
|
|
91
|
+
annotations: readOnly,
|
|
92
|
+
run: async (client, args) => {
|
|
93
|
+
const tree = await client.action(api._mcp.getSubtree, {
|
|
94
|
+
apiKey: requireApiKey(),
|
|
95
|
+
code: args.norm,
|
|
96
|
+
path: args.path,
|
|
97
|
+
maxDepth: args.maxDepth,
|
|
98
|
+
});
|
|
99
|
+
if (tree === null) {
|
|
100
|
+
return {
|
|
101
|
+
error: `Section "${args.path}" not found in norm "${args.norm}".`,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
return { norm: args.norm, root: args.path, tree };
|
|
105
|
+
},
|
|
106
|
+
});
|
|
107
|
+
export const getSection = defineTool({
|
|
108
|
+
name: 'get_section',
|
|
109
|
+
title: 'Get section',
|
|
110
|
+
description: "Fetch the full enriched content of a section: markdown text with formulas in LaTeX and tables inline, pageStart/pageEnd, figures (call get_figure for the image), attached tables/formulas, and cross-references to other norms. ALL technical claims in your answer MUST be backed by a [<norm> <path>, p. <pageStart>] citation pointing to a section you actually fetched via get_section — never cite from memory. When a section's crossRefs list non-empty targets, follow them with another get_section call if the answer depends on them.",
|
|
111
|
+
inputSchema: { norm, path: sectionPath },
|
|
112
|
+
annotations: readOnly,
|
|
113
|
+
run: async (client, args) => {
|
|
114
|
+
const section = await client.action(api._mcp.getSection, {
|
|
115
|
+
apiKey: requireApiKey(),
|
|
116
|
+
code: args.norm,
|
|
117
|
+
path: args.path,
|
|
118
|
+
});
|
|
119
|
+
if (section === null) {
|
|
120
|
+
return {
|
|
121
|
+
error: `Section "${args.path}" not found in norm "${args.norm}".`,
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
return { norm: args.norm, section };
|
|
125
|
+
},
|
|
126
|
+
});
|
|
127
|
+
export const searchInNorm = defineTool({
|
|
128
|
+
name: 'search_in_norm',
|
|
129
|
+
title: 'Search in norm',
|
|
130
|
+
description: 'Full-text search within a norm. Returns up to `limit` matches in relevance order, with path, title and a snippet. Matching is by term, not substring: "charges variables" finds sections containing those words, and a hit in a section TITLE ranks above one in its body. Use this when you do not yet know which section to read.',
|
|
131
|
+
inputSchema: {
|
|
132
|
+
norm,
|
|
133
|
+
keyword: z
|
|
134
|
+
.string()
|
|
135
|
+
.min(1)
|
|
136
|
+
.max(200)
|
|
137
|
+
.describe('Search terms. Searches both section titles and content.'),
|
|
138
|
+
limit: z
|
|
139
|
+
.number()
|
|
140
|
+
.int()
|
|
141
|
+
.min(1)
|
|
142
|
+
.max(50)
|
|
143
|
+
.optional()
|
|
144
|
+
.describe('Max results to return (default 20).'),
|
|
145
|
+
},
|
|
146
|
+
annotations: readOnly,
|
|
147
|
+
run: async (client, args) => ({
|
|
148
|
+
norm: args.norm,
|
|
149
|
+
keyword: args.keyword,
|
|
150
|
+
hits: await client.action(api._mcp.searchInNorm, {
|
|
151
|
+
apiKey: requireApiKey(),
|
|
152
|
+
code: args.norm,
|
|
153
|
+
keyword: args.keyword,
|
|
154
|
+
limit: args.limit,
|
|
155
|
+
}),
|
|
156
|
+
}),
|
|
157
|
+
});
|
|
158
|
+
export const getCrossRefs = defineTool({
|
|
159
|
+
name: 'get_cross_refs',
|
|
160
|
+
title: 'Get cross-references',
|
|
161
|
+
description: 'List outgoing cross-references from a section to other norms (e.g. SIA 261 §4.2 → SIA 263). For compound questions you MUST follow these refs: fetch each target via get_section before concluding. Skipping cross-refs is the #1 cause of incomplete answers — SIA norms intentionally distribute the rule, the coefficient, and the action across separate norms (260 / 261 / domain).',
|
|
162
|
+
inputSchema: { norm, path: sectionPath },
|
|
163
|
+
annotations: readOnly,
|
|
164
|
+
run: async (client, args) => ({
|
|
165
|
+
norm: args.norm,
|
|
166
|
+
path: args.path,
|
|
167
|
+
crossRefs: await client.action(api._mcp.getCrossRefs, {
|
|
168
|
+
apiKey: requireApiKey(),
|
|
169
|
+
code: args.norm,
|
|
170
|
+
path: args.path,
|
|
171
|
+
}),
|
|
172
|
+
}),
|
|
173
|
+
});
|
|
174
|
+
export const getFigure = defineTool({
|
|
175
|
+
name: 'get_figure',
|
|
176
|
+
title: 'Get figure',
|
|
177
|
+
description: 'Retrieve a figure (image) referenced in a section. Returns the image inline (base64) so you can see and reason about it. Use the `id` returned by get_section in its `figures` array.',
|
|
178
|
+
inputSchema: {
|
|
179
|
+
norm,
|
|
180
|
+
figureId: z
|
|
181
|
+
.string()
|
|
182
|
+
.min(1)
|
|
183
|
+
.describe('Figure ID from get_section response (figures[].id).'),
|
|
184
|
+
},
|
|
185
|
+
annotations: readOnly,
|
|
186
|
+
run: async (client, args) => {
|
|
187
|
+
const apiKey = requireApiKey();
|
|
188
|
+
const meta = (await client.action(api._mcp.getFigureMeta, {
|
|
189
|
+
apiKey,
|
|
190
|
+
code: args.norm,
|
|
191
|
+
figureId: args.figureId,
|
|
192
|
+
}));
|
|
193
|
+
if (!meta)
|
|
194
|
+
return { error: 'Figure not found.' };
|
|
195
|
+
// The storage URL is used server-side, here, only to download the bytes —
|
|
196
|
+
// it is never returned to the agent. Convex `getUrl` hands back a bare,
|
|
197
|
+
// unsigned, non-expiring public link, and the norm figures are licensed,
|
|
198
|
+
// per-org content: surfacing that link would drop a permanent, tenant-
|
|
199
|
+
// bypassing capability into the model context (and thus the provider's
|
|
200
|
+
// logs). The agent gets the image inline instead.
|
|
201
|
+
const url = (await client.action(api._mcp.getStorageUrl, {
|
|
202
|
+
apiKey,
|
|
203
|
+
code: args.norm,
|
|
204
|
+
figureId: args.figureId,
|
|
205
|
+
}));
|
|
206
|
+
if (!url)
|
|
207
|
+
return { error: 'Figure storage is unavailable.' };
|
|
208
|
+
const response = await fetch(url);
|
|
209
|
+
if (!response.ok) {
|
|
210
|
+
return { error: `Failed to download figure: HTTP ${response.status}` };
|
|
211
|
+
}
|
|
212
|
+
const buffer = await response.arrayBuffer();
|
|
213
|
+
return {
|
|
214
|
+
caption: meta.caption,
|
|
215
|
+
figureNumber: meta.figureNumber,
|
|
216
|
+
image: {
|
|
217
|
+
base64: Buffer.from(buffer).toString('base64'),
|
|
218
|
+
mimeType: response.headers.get('content-type') ?? 'image/png',
|
|
219
|
+
},
|
|
220
|
+
};
|
|
221
|
+
},
|
|
222
|
+
toContent: (result) => {
|
|
223
|
+
const r = result;
|
|
224
|
+
if (r.error || !r.image) {
|
|
225
|
+
return {
|
|
226
|
+
content: [{ type: 'text', text: r.error ?? 'Figure unavailable.' }],
|
|
227
|
+
isError: true,
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
return {
|
|
231
|
+
content: [
|
|
232
|
+
{ type: 'image', data: r.image.base64, mimeType: r.image.mimeType },
|
|
233
|
+
{ type: 'text', text: `${r.figureNumber} — ${r.caption}` },
|
|
234
|
+
],
|
|
235
|
+
};
|
|
236
|
+
},
|
|
237
|
+
});
|
|
238
|
+
export const readTools = [
|
|
239
|
+
getMethodology,
|
|
240
|
+
listNorms,
|
|
241
|
+
getToc,
|
|
242
|
+
getSubtree,
|
|
243
|
+
getSection,
|
|
244
|
+
searchInNorm,
|
|
245
|
+
getCrossRefs,
|
|
246
|
+
getFigure,
|
|
247
|
+
];
|
package/package.json
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stratta/mcp",
|
|
3
3
|
"mcpName": "io.github.hugogebel-boop/stratta",
|
|
4
|
-
"version": "0.
|
|
5
|
-
"description": "MCP server exposing
|
|
4
|
+
"version": "0.7.1",
|
|
5
|
+
"description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
|
|
6
6
|
"license": "UNLICENSED",
|
|
7
|
-
"author": "
|
|
7
|
+
"author": "SmartFlow <hello@stratta.ch>",
|
|
8
8
|
"homepage": "https://stratta.ch",
|
|
9
9
|
"repository": {
|
|
10
10
|
"type": "git",
|
|
@@ -33,7 +33,7 @@
|
|
|
33
33
|
"README.md"
|
|
34
34
|
],
|
|
35
35
|
"scripts": {
|
|
36
|
-
"build": "tsc",
|
|
36
|
+
"build": "node -e \"require('fs').rmSync('dist',{recursive:true,force:true})\" && tsc",
|
|
37
37
|
"dev": "tsc --watch",
|
|
38
38
|
"start": "node dist/index.js",
|
|
39
39
|
"test": "vitest run",
|
|
@@ -41,14 +41,14 @@
|
|
|
41
41
|
"prepublishOnly": "npm run build"
|
|
42
42
|
},
|
|
43
43
|
"dependencies": {
|
|
44
|
-
"@modelcontextprotocol/sdk": "^1.
|
|
45
|
-
"convex": "^1.
|
|
44
|
+
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
45
|
+
"convex": "^1.42.3",
|
|
46
46
|
"zod": "^4.0.0"
|
|
47
47
|
},
|
|
48
48
|
"devDependencies": {
|
|
49
|
-
"@types/node": "^
|
|
50
|
-
"typescript": "^
|
|
51
|
-
"vitest": "^
|
|
49
|
+
"@types/node": "^26.1.1",
|
|
50
|
+
"typescript": "^6.0.3",
|
|
51
|
+
"vitest": "^4.1.10"
|
|
52
52
|
},
|
|
53
53
|
"engines": {
|
|
54
54
|
"node": ">=20"
|
|
@@ -1,128 +1,140 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: ingest-norm
|
|
3
|
-
description: Use when the user wants to add an engineering norm (SIA, Eurocode, etc.) they are licensed for into THEIR Stratta workspace. A Python pre-pass (PyMuPDF) extracts the hierarchical tree and rasterizes figures; an agentic pass enriches sections (LaTeX formulas, tables, cross-references, summaries) and writes everything via the MCP `ingest_*` tools — scoped to the user's own organization. Trigger phrases: "ingère cette norme", "/ingest-norm", "ingest SIA", "ajoute la norme X à Stratta".
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Ingest a norm into your Stratta workspace
|
|
7
|
-
|
|
8
|
-
This skill turns a norm PDF **you are licensed to use** into a queryable
|
|
9
|
-
TreeRAG inside **your own** Stratta workspace. A Python pre-pass (PyMuPDF)
|
|
10
|
-
does the deterministic heavy lifting — TOC tree, per-section raw text, figure
|
|
11
|
-
captions and full-page renders. Then an agentic pass enriches sections with
|
|
12
|
-
summaries, LaTeX formulas, structured tables, and cross-references. Final
|
|
13
|
-
writes go through the Stratta MCP `ingest_*` tools, scoped to your org.
|
|
14
|
-
|
|
15
|
-
> ⚠️ **Licence**: only ingest norms your organization holds a valid licence
|
|
16
|
-
> for. You are responsible for your usage rights (see Stratta's Terms).
|
|
17
|
-
|
|
18
|
-
## Prerequisites
|
|
19
|
-
- The Stratta MCP server is installed and your `STRATTA_API_KEY` resolves
|
|
20
|
-
(via env, `~/.stratta/config.json`, or first-call elicitation).
|
|
21
|
-
- **Python ≥ 3.10 with PyMuPDF**. Install once:
|
|
22
|
-
`python -m pip install --user pymupdf` (or `uv pip install pymupdf`).
|
|
23
|
-
- The norm PDF is available locally.
|
|
24
|
-
|
|
25
|
-
## Tools used (all scoped to your workspace)
|
|
26
|
-
`ingest_status` · `ingest_create_document` · `ingest_create_sections` ·
|
|
27
|
-
`ingest_attach_formula` · `ingest_attach_table` · `ingest_attach_cross_ref` ·
|
|
28
|
-
`ingest_upload_figure` · `ingest_normalize_cross_refs` · `ingest_publish` ·
|
|
29
|
-
`ingest_delete`.
|
|
30
|
-
|
|
31
|
-
##
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
```bash
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
`
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
`
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
- **
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
- `
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
`
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
1
|
+
---
|
|
2
|
+
name: ingest-norm
|
|
3
|
+
description: Use when the user wants to add an engineering norm (SIA, Eurocode, etc.) they are licensed for into THEIR Stratta workspace. A Python pre-pass (PyMuPDF) extracts the hierarchical tree and rasterizes figures; an agentic pass enriches sections (LaTeX formulas, tables, cross-references, summaries) and writes everything via the MCP `ingest_*` tools — scoped to the user's own organization. Trigger phrases: "ingère cette norme", "/ingest-norm", "ingest SIA", "ajoute la norme X à Stratta".
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Ingest a norm into your Stratta workspace
|
|
7
|
+
|
|
8
|
+
This skill turns a norm PDF **you are licensed to use** into a queryable
|
|
9
|
+
TreeRAG inside **your own** Stratta workspace. A Python pre-pass (PyMuPDF)
|
|
10
|
+
does the deterministic heavy lifting — TOC tree, per-section raw text, figure
|
|
11
|
+
captions and full-page renders. Then an agentic pass enriches sections with
|
|
12
|
+
summaries, LaTeX formulas, structured tables, and cross-references. Final
|
|
13
|
+
writes go through the Stratta MCP `ingest_*` tools, scoped to your org.
|
|
14
|
+
|
|
15
|
+
> ⚠️ **Licence**: only ingest norms your organization holds a valid licence
|
|
16
|
+
> for. You are responsible for your usage rights (see Stratta's Terms).
|
|
17
|
+
|
|
18
|
+
## Prerequisites
|
|
19
|
+
- The Stratta MCP server is installed and your `STRATTA_API_KEY` resolves
|
|
20
|
+
(via env, `~/.stratta/config.json`, or first-call elicitation).
|
|
21
|
+
- **Python ≥ 3.10 with PyMuPDF**. Install once:
|
|
22
|
+
`python -m pip install --user pymupdf` (or `uv pip install pymupdf`).
|
|
23
|
+
- The norm PDF is available locally.
|
|
24
|
+
|
|
25
|
+
## Tools used (all scoped to your workspace)
|
|
26
|
+
`ingest_status` · `ingest_create_document` · `ingest_create_sections` ·
|
|
27
|
+
`ingest_attach_formula` · `ingest_attach_table` · `ingest_attach_cross_ref` ·
|
|
28
|
+
`ingest_upload_figure` · `ingest_normalize_cross_refs` · `ingest_publish` ·
|
|
29
|
+
`ingest_delete`.
|
|
30
|
+
|
|
31
|
+
## Quotas — read before you start
|
|
32
|
+
The workspace has server-enforced limits on norms, sections and figures. A
|
|
33
|
+
`QUOTA_EXCEEDED` error names the dimension, the current count and the limit.
|
|
34
|
+
|
|
35
|
+
**Do not retry it.** The limit will not move on its own. Stop the ingestion,
|
|
36
|
+
report the numbers to the user, and offer the two ways out: delete a norm that is
|
|
37
|
+
no longer needed (`ingest_delete`), or raise the plan.
|
|
38
|
+
|
|
39
|
+
Check the budget up front on a large norm: the pre-pass output tells you how many
|
|
40
|
+
sections and figures you are about to write. Failing at step 8 of 10 leaves a
|
|
41
|
+
half-ingested document behind, which the user then has to delete by hand.
|
|
42
|
+
|
|
43
|
+
## Workflow
|
|
44
|
+
|
|
45
|
+
### 1. Locate the pre-pass script
|
|
46
|
+
It ships inside this package at `scripts/ingest-prepass.py`. Resolve its path:
|
|
47
|
+
```bash
|
|
48
|
+
node -e "console.log(require.resolve('@stratta/mcp/package.json'))"
|
|
49
|
+
# → <root>/package.json → <root>/scripts/ingest-prepass.py
|
|
50
|
+
```
|
|
51
|
+
If the user is working in the Stratta monorepo, the script also lives at
|
|
52
|
+
`packages/mcp/scripts/ingest-prepass.py`.
|
|
53
|
+
|
|
54
|
+
### 2. Check for an existing copy
|
|
55
|
+
`ingest_status { code }` (e.g. `"SIA 261"`). To re-ingest, call
|
|
56
|
+
`ingest_delete { documentId }` first.
|
|
57
|
+
|
|
58
|
+
### 3. Run the pre-pass
|
|
59
|
+
```bash
|
|
60
|
+
python <pkg-root>/scripts/ingest-prepass.py \
|
|
61
|
+
--pdf <path-to-pdf> \
|
|
62
|
+
--output .stratta-ingest/<code-slug>
|
|
63
|
+
```
|
|
64
|
+
Output under `.stratta-ingest/<code-slug>/`:
|
|
65
|
+
- `prepass.json` — full manifest (see below).
|
|
66
|
+
- `figures/page-NNN.png` — one PNG per page that contains a `Figure N` caption.
|
|
67
|
+
|
|
68
|
+
`prepass.json` structure:
|
|
69
|
+
- `doc` — `pageCount`, detected `language`, `tocSource`, raw `metadata`.
|
|
70
|
+
- `stats` — `sectionCount`, `byDepth`, `figureCount`.
|
|
71
|
+
- `sections[]` — full hierarchical tree (depth **0 = chapter**, 1+ = sub-sections),
|
|
72
|
+
each with `nodeId`, `parentNodeId`, `path` (`"4.2.1"` or `"Annexe B"`),
|
|
73
|
+
`title`, `depth`, `pageStart`, `pageEnd`, `orderIndex`, `rawText` (concat
|
|
74
|
+
of the pages the node spans).
|
|
75
|
+
- `figures[]` — one entry per `Figure N` caption: `figureNumber`, `caption`,
|
|
76
|
+
`page`, `fileName`, `mimeType`.
|
|
77
|
+
|
|
78
|
+
The pre-pass is **exhaustive** (e.g. ~550 nodes on SIA 261). You decide what
|
|
79
|
+
to keep in the next step.
|
|
80
|
+
|
|
81
|
+
### 4. Create the document
|
|
82
|
+
Read `prepass.json`, then:
|
|
83
|
+
`ingest_create_document { code, year, title, language: <doc.language>, totalPages: <doc.pageCount> }`
|
|
84
|
+
→ returns `documentId`. Keep it for every subsequent call.
|
|
85
|
+
|
|
86
|
+
### 5. Decide section granularity + generate summaries
|
|
87
|
+
Iterate `sections[]` and decide what to keep. Two viable strategies:
|
|
88
|
+
- **Keep all** — most faithful, ~500 sections on a typical SIA norm. Great
|
|
89
|
+
for fine-grained navigation but verbose.
|
|
90
|
+
- **Aggregate trivial leaves** — fold paragraph-level nodes (`6.1.1`...`6.1.11`)
|
|
91
|
+
into their parent (`6.1`), concatenating their `rawText`. Typical result:
|
|
92
|
+
100-150 sections. Recommended unless the user asks for max granularity.
|
|
93
|
+
|
|
94
|
+
For each kept section, prepare:
|
|
95
|
+
- `summary` — 1-3 sentences derived from `rawText` (mention formulas/values).
|
|
96
|
+
- `content` — enriched text with LaTeX inline where the source has math
|
|
97
|
+
(e.g. `$\sigma_d = f_{yd} \cdot \gamma$`). Open the PDF visually for pages
|
|
98
|
+
that contain formulas or multi-column tables — PyMuPDF mangles those.
|
|
99
|
+
- `rawContent` — use the pre-pass `rawText` as-is.
|
|
100
|
+
|
|
101
|
+
Keep the `nodeId` / `parentNodeId` / `path` / `pageStart` / `pageEnd` /
|
|
102
|
+
`orderIndex` / `depth` from the pre-pass — those are deterministic.
|
|
103
|
+
|
|
104
|
+
### 6. Insert sections (batched)
|
|
105
|
+
`ingest_create_sections { documentId, sections: [...] }` in batches of 30-50.
|
|
106
|
+
Parent links resolve via `parentNodeId` within the batch and across prior
|
|
107
|
+
batches. The call returns a `nodeId → sectionId` map — **use those `sectionId`s**
|
|
108
|
+
for every enrichment call below.
|
|
109
|
+
|
|
110
|
+
### 7. Enrich
|
|
111
|
+
- `ingest_attach_formula { sectionId, latex, description, formulaNumber }`
|
|
112
|
+
- `ingest_attach_table { sectionId, data: { headers, rows }, caption, tableNumber }`
|
|
113
|
+
- `ingest_attach_cross_ref { sourceSectionId, targetDocumentCode, targetSectionPath?, refText, refType }`
|
|
114
|
+
|
|
115
|
+
### 8. Upload figures
|
|
116
|
+
For each figure in `prepass.json#figures`:
|
|
117
|
+
- Read `.stratta-ingest/<code>/<fileName>` and base64-encode the bytes.
|
|
118
|
+
- Find the owning section: the kept section whose `pageStart..pageEnd`
|
|
119
|
+
range includes the figure's `page`.
|
|
120
|
+
- `ingest_upload_figure { sectionId, base64, mimeType: "image/png", caption, figureNumber }`
|
|
121
|
+
(≤ 8 MB per image).
|
|
122
|
+
|
|
123
|
+
### 9. Auto cross-references (optional but recommended)
|
|
124
|
+
`ingest_normalize_cross_refs { documentId }` scans every section's text for
|
|
125
|
+
references to other norms (SIA / SN EN / EN / ISO / DIN …) and rebuilds the
|
|
126
|
+
cross-ref index. Idempotent.
|
|
127
|
+
|
|
128
|
+
### 10. Publish
|
|
129
|
+
`ingest_publish { documentId }`. The norm is now queryable in your workspace
|
|
130
|
+
via `list_norms`, `get_toc`, `get_section`, `search_in_norm`, `get_figure`, etc.
|
|
131
|
+
|
|
132
|
+
## Quality bar
|
|
133
|
+
- Trust the pre-pass for `path` + `pageStart`/`pageEnd` — it's deterministic
|
|
134
|
+
and verified against the PDF bookmarks.
|
|
135
|
+
- Summaries drive navigation — be specific (mention key formulas/values).
|
|
136
|
+
- Keep `content` faithful to the source; don't invent values.
|
|
137
|
+
- If PyMuPDF mangled a formula (Greek letters, fractions, exponents broken),
|
|
138
|
+
re-read the relevant PDF page visually and write proper LaTeX.
|
|
139
|
+
- Never skip annexes — they hold key numeric values (zones, coefficients,
|
|
140
|
+
characteristic loads).
|
|
@@ -1,24 +0,0 @@
|
|
|
1
|
-
import type { ConvexHttpClient } from 'convex/browser';
|
|
2
|
-
export declare const getCrossRefsTool: {
|
|
3
|
-
readonly name: "get_cross_refs";
|
|
4
|
-
readonly description: "List outgoing cross-references from a section to other norms (e.g. SIA 261 §4.2 → SIA 263). For compound questions you MUST follow these refs: fetch each target via get_section before concluding. Skipping cross-refs is the #1 cause of incomplete answers — SIA norms intentionally distribute the rule, the coefficient, and the action across separate norms (260 / 261 / domain).";
|
|
5
|
-
readonly inputSchema: {
|
|
6
|
-
readonly type: "object";
|
|
7
|
-
readonly properties: {
|
|
8
|
-
readonly norm: {
|
|
9
|
-
readonly type: "string";
|
|
10
|
-
readonly description: "Norm code, e.g. \"SIA 261\".";
|
|
11
|
-
};
|
|
12
|
-
readonly path: {
|
|
13
|
-
readonly type: "string";
|
|
14
|
-
readonly description: "Section path, e.g. \"4.2.1\".";
|
|
15
|
-
};
|
|
16
|
-
};
|
|
17
|
-
readonly required: readonly ["norm", "path"];
|
|
18
|
-
readonly additionalProperties: false;
|
|
19
|
-
};
|
|
20
|
-
};
|
|
21
|
-
export declare function handleGetCrossRefs(client: ConvexHttpClient, args: {
|
|
22
|
-
norm: string;
|
|
23
|
-
path: string;
|
|
24
|
-
}): Promise<unknown>;
|
|
@@ -1,23 +0,0 @@
|
|
|
1
|
-
import { api } from '../client.js';
|
|
2
|
-
import { requireApiKey } from '../auth.js';
|
|
3
|
-
export const getCrossRefsTool = {
|
|
4
|
-
name: 'get_cross_refs',
|
|
5
|
-
description: "List outgoing cross-references from a section to other norms (e.g. SIA 261 §4.2 → SIA 263). For compound questions you MUST follow these refs: fetch each target via get_section before concluding. Skipping cross-refs is the #1 cause of incomplete answers — SIA norms intentionally distribute the rule, the coefficient, and the action across separate norms (260 / 261 / domain).",
|
|
6
|
-
inputSchema: {
|
|
7
|
-
type: 'object',
|
|
8
|
-
properties: {
|
|
9
|
-
norm: { type: 'string', description: 'Norm code, e.g. "SIA 261".' },
|
|
10
|
-
path: { type: 'string', description: 'Section path, e.g. "4.2.1".' },
|
|
11
|
-
},
|
|
12
|
-
required: ['norm', 'path'],
|
|
13
|
-
additionalProperties: false,
|
|
14
|
-
},
|
|
15
|
-
};
|
|
16
|
-
export async function handleGetCrossRefs(client, args) {
|
|
17
|
-
const result = await client.action(api._mcp.getCrossRefs, {
|
|
18
|
-
apiKey: requireApiKey(),
|
|
19
|
-
code: args.norm,
|
|
20
|
-
path: args.path,
|
|
21
|
-
});
|
|
22
|
-
return { norm: args.norm, path: args.path, crossRefs: result };
|
|
23
|
-
}
|
|
@@ -1,24 +0,0 @@
|
|
|
1
|
-
import type { ConvexHttpClient } from 'convex/browser';
|
|
2
|
-
export declare const getFigureTool: {
|
|
3
|
-
readonly name: "get_figure";
|
|
4
|
-
readonly description: "Retrieve a figure (image) referenced in a section. Returns the image inline (base64) so you can see and reason about it, plus a public `url` for full-size viewing. Use the `id` returned by get_section in its `figures` array.";
|
|
5
|
-
readonly inputSchema: {
|
|
6
|
-
readonly type: "object";
|
|
7
|
-
readonly properties: {
|
|
8
|
-
readonly norm: {
|
|
9
|
-
readonly type: "string";
|
|
10
|
-
readonly description: "Norm code, e.g. \"SIA 261\".";
|
|
11
|
-
};
|
|
12
|
-
readonly figureId: {
|
|
13
|
-
readonly type: "string";
|
|
14
|
-
readonly description: "Figure ID from get_section response (figures[].id).";
|
|
15
|
-
};
|
|
16
|
-
};
|
|
17
|
-
readonly required: readonly ["norm", "figureId"];
|
|
18
|
-
readonly additionalProperties: false;
|
|
19
|
-
};
|
|
20
|
-
};
|
|
21
|
-
export declare function handleGetFigure(client: ConvexHttpClient, args: {
|
|
22
|
-
norm: string;
|
|
23
|
-
figureId: string;
|
|
24
|
-
}): Promise<unknown>;
|