@stratta/mcp 0.9.7 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -20,6 +20,25 @@ MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients
20
20
  > a shell environment variable. If a key leaks, revoke it immediately at
21
21
  > https://stratta.ch/api-keys.
22
22
 
23
+ ## Do you need this package?
24
+
25
+ Often not. Stratta also runs as a **remote connector** — one address, a browser
26
+ sign-in, no key and no Node:
27
+
28
+ ```
29
+ https://stratta.ch/mcp
30
+ ```
31
+
32
+ That is the shorter path, and the only one that works in an agent running in the
33
+ cloud (claude.ai). See https://stratta.ch/docs/en/guides/connect-remote.
34
+
35
+ This package is what you want when:
36
+
37
+ - you are **ingesting a norm** — it reads a PDF from your disk and runs a Python
38
+ pre-pass, neither of which a remote connector can reach;
39
+ - your agent **cannot open a browser** — CI, a scheduled task, a server;
40
+ - you would simply rather run the server yourself.
41
+
23
42
  ## Install
24
43
 
25
44
  ### Claude Code (recommended)
@@ -148,15 +167,17 @@ settings come from environment variables — see [`.env.example`](./.env.example
148
167
  | `get_figure` | Retrieve a figure inline (base64 ImageContent) + public URL. |
149
168
  | `get_cross_refs` | Outgoing cross-refs from a section to other norms. |
150
169
 
151
- **Dossier** (5 tools — keep what was decided on a project):
152
-
153
- | Tool | Purpose |
154
- | ------------------ | ------------------------------------------------------------------------------------------------- |
155
- | `list_dossiers` | Your organisation's dossiers, most recently touched first, with open counts. |
156
- | `open_dossier` | Open a project's dossier, creating it if needed. Idempotent on the name. |
157
- | `save_finding` | Record one decision: a cited article, a retained value and why, an observation, an open question. |
158
- | `load_dossier` | Reload everything, unresolved questions first. Accepts the id or the name. |
159
- | `resolve_question` | Mark a question settled. The entry stays; it stops surfacing at the top. |
170
+ **Dossier** (7 tools — keep what was decided on a project):
171
+
172
+ | Tool | Purpose |
173
+ | ------------------ | ---------------------------------------------------------------------------------------- |
174
+ | `list_dossiers` | Your organisation's dossiers, most recently touched first, with open-question counts. |
175
+ | `open_dossier` | Open a project's dossier, creating it if needed. Idempotent on the name. |
176
+ | `open_question` | Open one question to settle, with optional named options. Idempotent on the title. |
177
+ | `save_finding` | Record one piece of evidence: a cited article, a retained value and why, an observation. |
178
+ | `record_decision` | Settle a question with a decision the engineer has confirmed, and the retained option. |
179
+ | `load_dossier` | Reload everything: questions with their evidence and decisions, open ones first. |
180
+ | `resolve_question` | Close a question without a decision, or reopen one. The evidence stays. |
160
181
 
161
182
  A dossier is read, annotated, reviewed and exported from
162
183
  [stratta.ch/dossiers](https://stratta.ch/dossiers).
@@ -4,14 +4,34 @@ export declare const openDossier: import("./define.js").ToolDef<{
4
4
  name: z.ZodString;
5
5
  reference: z.ZodOptional<z.ZodString>;
6
6
  }>;
7
+ export declare const openQuestion: import("./define.js").ToolDef<{
8
+ dossierId: z.ZodString;
9
+ title: z.ZodString;
10
+ body: z.ZodOptional<z.ZodString>;
11
+ options: z.ZodOptional<z.ZodArray<z.ZodString>>;
12
+ }>;
7
13
  export declare const saveFinding: import("./define.js").ToolDef<{
14
+ title: z.ZodString;
15
+ detail: z.ZodOptional<z.ZodString>;
16
+ value: z.ZodOptional<z.ZodString>;
17
+ confidence: z.ZodOptional<z.ZodEnum<{
18
+ established: "established";
19
+ judgement: "judgement";
20
+ to_confirm: "to_confirm";
21
+ }>>;
22
+ normCode: z.ZodOptional<z.ZodString>;
23
+ sectionPath: z.ZodOptional<z.ZodString>;
24
+ page: z.ZodOptional<z.ZodNumber>;
25
+ normEdition: z.ZodOptional<z.ZodString>;
8
26
  dossierId: z.ZodString;
27
+ questionId: z.ZodOptional<z.ZodString>;
9
28
  kind: z.ZodEnum<{
10
29
  reference: "reference";
11
30
  hypothesis: "hypothesis";
12
31
  observation: "observation";
13
- question: "question";
14
32
  }>;
33
+ }>;
34
+ export declare const recordDecision: import("./define.js").ToolDef<{
15
35
  title: z.ZodString;
16
36
  detail: z.ZodOptional<z.ZodString>;
17
37
  value: z.ZodOptional<z.ZodString>;
@@ -23,13 +43,16 @@ export declare const saveFinding: import("./define.js").ToolDef<{
23
43
  normCode: z.ZodOptional<z.ZodString>;
24
44
  sectionPath: z.ZodOptional<z.ZodString>;
25
45
  page: z.ZodOptional<z.ZodNumber>;
46
+ normEdition: z.ZodOptional<z.ZodString>;
47
+ questionId: z.ZodString;
48
+ retainedOption: z.ZodOptional<z.ZodString>;
26
49
  }>;
27
50
  export declare const loadDossier: import("./define.js").ToolDef<{
28
51
  dossierId: z.ZodOptional<z.ZodString>;
29
52
  name: z.ZodOptional<z.ZodString>;
30
53
  }>;
31
54
  export declare const resolveQuestion: import("./define.js").ToolDef<{
32
- entryId: z.ZodString;
55
+ questionId: z.ZodString;
33
56
  resolved: z.ZodOptional<z.ZodBoolean>;
34
57
  }>;
35
58
  export declare const dossierTools: (import("./define.js").ToolDef<{}> | import("./define.js").ToolDef<{
@@ -37,12 +60,30 @@ export declare const dossierTools: (import("./define.js").ToolDef<{}> | import("
37
60
  reference: z.ZodOptional<z.ZodString>;
38
61
  }> | import("./define.js").ToolDef<{
39
62
  dossierId: z.ZodString;
63
+ title: z.ZodString;
64
+ body: z.ZodOptional<z.ZodString>;
65
+ options: z.ZodOptional<z.ZodArray<z.ZodString>>;
66
+ }> | import("./define.js").ToolDef<{
67
+ title: z.ZodString;
68
+ detail: z.ZodOptional<z.ZodString>;
69
+ value: z.ZodOptional<z.ZodString>;
70
+ confidence: z.ZodOptional<z.ZodEnum<{
71
+ established: "established";
72
+ judgement: "judgement";
73
+ to_confirm: "to_confirm";
74
+ }>>;
75
+ normCode: z.ZodOptional<z.ZodString>;
76
+ sectionPath: z.ZodOptional<z.ZodString>;
77
+ page: z.ZodOptional<z.ZodNumber>;
78
+ normEdition: z.ZodOptional<z.ZodString>;
79
+ dossierId: z.ZodString;
80
+ questionId: z.ZodOptional<z.ZodString>;
40
81
  kind: z.ZodEnum<{
41
82
  reference: "reference";
42
83
  hypothesis: "hypothesis";
43
84
  observation: "observation";
44
- question: "question";
45
85
  }>;
86
+ }> | import("./define.js").ToolDef<{
46
87
  title: z.ZodString;
47
88
  detail: z.ZodOptional<z.ZodString>;
48
89
  value: z.ZodOptional<z.ZodString>;
@@ -54,10 +95,13 @@ export declare const dossierTools: (import("./define.js").ToolDef<{}> | import("
54
95
  normCode: z.ZodOptional<z.ZodString>;
55
96
  sectionPath: z.ZodOptional<z.ZodString>;
56
97
  page: z.ZodOptional<z.ZodNumber>;
98
+ normEdition: z.ZodOptional<z.ZodString>;
99
+ questionId: z.ZodString;
100
+ retainedOption: z.ZodOptional<z.ZodString>;
57
101
  }> | import("./define.js").ToolDef<{
58
102
  dossierId: z.ZodOptional<z.ZodString>;
59
103
  name: z.ZodOptional<z.ZodString>;
60
104
  }> | import("./define.js").ToolDef<{
61
- entryId: z.ZodString;
105
+ questionId: z.ZodString;
62
106
  resolved: z.ZodOptional<z.ZodBoolean>;
63
107
  }>)[];
@@ -3,17 +3,24 @@ import { api } from '../client.js';
3
3
  import { requireApiKey } from '../auth.js';
4
4
  import { defineTool } from './define.js';
5
5
  /**
6
- * The five dossier tools: how the work survives the conversation.
6
+ * The seven dossier tools: how the work survives the conversation.
7
7
  *
8
8
  * Everything else here reads a norm. These write down what was decided with
9
9
  * it, so the next conversation starts where the last one stopped and a human
10
10
  * can review it in the app.
11
11
  *
12
+ * A dossier is a set of QUESTIONS. Each question holds the evidence gathered
13
+ * for it, the options on the table and, once the engineer has confirmed it, a
14
+ * decision. The question is the unit of work: it is what gets assigned,
15
+ * reviewed and closed. The first model was a flat journal, and the one real
16
+ * dossier it produced held a single "question" with eight sub-questions in its
17
+ * body — unassignable, unanswerable one at a time, unreadable.
18
+ *
12
19
  * The descriptions carry more instruction than the read tools do, on purpose.
13
20
  * An agent left to guess will either record nothing — and the feature is dead
14
21
  * — or record every sentence it produces, which buries the three decisions
15
22
  * that mattered. `get_methodology` tells it when to open a dossier; these tell
16
- * it what is worth putting in one.
23
+ * it what is worth putting in one, and in what shape.
17
24
  */
18
25
  const write = { readOnlyHint: false, openWorldHint: false };
19
26
  const readOnly = { readOnlyHint: true, openWorldHint: false };
@@ -21,10 +28,57 @@ const dossierId = z
21
28
  .string()
22
29
  .min(1)
23
30
  .describe('Dossier id returned by open_dossier or list_dossiers.');
31
+ const questionId = z
32
+ .string()
33
+ .min(1)
34
+ .describe('Question id returned by open_question or load_dossier.');
35
+ /** The fields a piece of evidence or a decision carries. */
36
+ const entryFields = {
37
+ title: z
38
+ .string()
39
+ .min(1)
40
+ .max(200)
41
+ .describe('The fact in one line, as an engineer would write it in a report. One clause, one value or one observation; not a theme.'),
42
+ detail: z
43
+ .string()
44
+ .max(4000)
45
+ .optional()
46
+ .describe('The reasoning, in one or two sentences: why this value, what it rests on, what disagreed. Required in practice for a hypothesis.'),
47
+ value: z
48
+ .string()
49
+ .max(120)
50
+ .optional()
51
+ .describe('The retained value WITH its unit, e.g. "30°", "1,25 kN/m²".'),
52
+ confidence: z
53
+ .enum(['established', 'judgement', 'to_confirm'])
54
+ .optional()
55
+ .describe('How firm the value is. See the tool description.'),
56
+ normCode: z
57
+ .string()
58
+ .max(40)
59
+ .optional()
60
+ .describe('Norm this comes from, e.g. "SIA 267".'),
61
+ sectionPath: z
62
+ .string()
63
+ .max(60)
64
+ .optional()
65
+ .describe('Exact section path, e.g. "9.5.2.1".'),
66
+ page: z
67
+ .number()
68
+ .int()
69
+ .positive()
70
+ .optional()
71
+ .describe('Page in the norm, so a human can verify it in their PDF.'),
72
+ normEdition: z
73
+ .string()
74
+ .max(40)
75
+ .optional()
76
+ .describe('The edition cited, e.g. "2013" or "2020 + rect. 2022", when list_norms gives it.'),
77
+ };
24
78
  export const listDossiers = defineTool({
25
79
  name: 'list_dossiers',
26
80
  title: 'List dossiers',
27
- description: 'List the dossiers of YOUR organisation, most recently touched first. A dossier is a project record: what was decided, what it rests on, and what is still open. Call this when the user mentions a project by name, or asks what they were working on. Each row carries the count of entries and of UNRESOLVED questions — a dossier with open questions is the one to resume.',
81
+ description: 'List the dossiers of YOUR organisation, most recently touched first. A dossier is a project record: the questions to settle, what was decided, what it rests on. Call this when the user mentions a project by name, or asks what they were working on. Each row carries the count of entries and of OPEN questions — a dossier with open questions is the one to resume.',
28
82
  inputSchema: {},
29
83
  annotations: readOnly,
30
84
  run: async (client) => ({
@@ -36,7 +90,7 @@ export const listDossiers = defineTool({
36
90
  export const openDossier = defineTool({
37
91
  name: 'open_dossier',
38
92
  title: 'Open a dossier',
39
- description: "Open the dossier for a project, creating it if it does not exist yet. IDEMPOTENT on the name: calling it twice with the same name returns the same dossier rather than splitting a project's history in two. Call this when the user starts working on a named project and there is something worth keeping — a retained value, an assumption, a site observation, an unanswered question. Do not open one for a passing lookup. Returns { dossierId, created }.",
93
+ description: "Open the dossier for a project, creating it if it does not exist yet. IDEMPOTENT on the name: calling it twice with the same name returns the same dossier rather than splitting a project's history in two. Call this when the user starts working on a named project and there is something worth keeping — a question to settle, a retained value, an assumption, a site observation. Do not open one for a passing lookup. Returns { dossierId, created }.",
40
94
  inputSchema: {
41
95
  name: z
42
96
  .string()
@@ -56,63 +110,89 @@ export const openDossier = defineTool({
56
110
  reference: args.reference,
57
111
  }),
58
112
  });
113
+ export const openQuestion = defineTool({
114
+ name: 'open_question',
115
+ title: 'Open a question',
116
+ description: `Open ONE question to settle in a dossier, BEFORE gathering evidence for it. A question is the unit of work: "Can pile P38 be kept?", "Which φ'k do we retain for the fill?", "Is kinematic interaction to be checked?". Evidence saved with save_finding and the eventual record_decision hang off it.
117
+
118
+ ONE QUESTION PER THING TO SETTLE. Never put a list of sub-questions in the body: a question that bundles eight cannot be assigned, answered or closed one at a time. Open eight questions.
119
+
120
+ IDEMPOTENT on the title: reopening the same wording returns the question already open, with its evidence. Name the options on the table in \`options\` when the user or the norm puts several forward ("abandon the pile", "recover it by re-concreting plus integrity tests", "redistribute onto P37/P39"). Returns { questionId, created }.`,
121
+ inputSchema: {
122
+ dossierId,
123
+ title: z
124
+ .string()
125
+ .min(1)
126
+ .max(200)
127
+ .describe('The question in one line, phrased as a question.'),
128
+ body: z
129
+ .string()
130
+ .max(4000)
131
+ .optional()
132
+ .describe('What is at stake and why it is not settled, in two or three sentences. Not a list of sub-questions.'),
133
+ options: z
134
+ .array(z.string().min(1).max(120))
135
+ .max(10)
136
+ .optional()
137
+ .describe('The ways this could be settled, one short name each.'),
138
+ },
139
+ annotations: { ...write, idempotentHint: true },
140
+ run: (client, args) => client.action(api.dossiersApi.openQuestion, {
141
+ apiKey: requireApiKey(),
142
+ dossierId: args.dossierId,
143
+ title: args.title,
144
+ body: args.body,
145
+ options: args.options,
146
+ }),
147
+ });
59
148
  export const saveFinding = defineTool({
60
149
  name: 'save_finding',
61
- title: 'Save a finding to a dossier',
62
- description: `Record ONE decision in a dossier. Call it as you work, not in a batch at the end — a finding saved when it is made carries the reasoning that produced it.
150
+ title: 'Record evidence in a dossier',
151
+ description: `Record ONE piece of evidence in a dossier. Call it as you work, not in a batch at the end — a finding saved when it is made carries the reasoning that produced it.
63
152
 
64
153
  WHAT TO RECORD, by kind:
65
154
  - reference: what a norm says, that the project relies on. ALWAYS fill normCode + sectionPath (+ page). If you cannot cite it, it is not a reference.
66
155
  - hypothesis: a value the engineer RETAINS, and why. This is the one that matters most in geotechnics, where a retained value is a judgement between disagreeing measurements rather than the output of a formula. Put the number in \`value\` and the reasoning in \`detail\`.
67
156
  - observation: what the site, a borehole, or a survey showed. Include the date in \`detail\` when known.
68
- - question: something not settled. These surface FIRST when the dossier is reloaded, so record them even when you cannot answer — especially then.
69
157
 
70
- WHAT NOT TO RECORD: your own prose, intermediate steps, anything the user did not treat as a decision. A dossier of forty entries where three mattered is worse than a dossier of three.
158
+ KEEP IT SHORT: one clause, one value or one fact per call, in one or two sentences. Three short entries beat one paragraph that mixes the citation, the reasoning and the consequence. A question is NOT recorded here: call open_question. A decision is NOT recorded here: call record_decision.
159
+
160
+ FILE IT: pass \`questionId\` so the entry lands under the question it serves. If you do not yet know which question it serves, save it anyway without one — the engineer files it later.
71
161
 
72
162
  Set \`confidence\` whenever the entry is a value: established (computed or read directly), judgement (the engineer chose it), to_confirm (provisional, needs a test or a check). Confusing those three is the professional fault this field exists to prevent.`,
73
163
  inputSchema: {
74
164
  dossierId,
165
+ questionId: questionId
166
+ .optional()
167
+ .describe('The question this evidence serves, from open_question or load_dossier. Omit only when you do not know yet.'),
75
168
  kind: z
76
- .enum(['reference', 'hypothesis', 'observation', 'question'])
169
+ .enum(['reference', 'hypothesis', 'observation'])
77
170
  .describe('See the tool description: pick by what the entry IS.'),
78
- title: z
79
- .string()
80
- .min(1)
81
- .max(200)
82
- .describe('The decision in one line, as an engineer would write it in a report.'),
83
- detail: z
84
- .string()
85
- .max(4000)
86
- .optional()
87
- .describe('The reasoning: why this value, what it rests on, what disagreed. Required in practice for a hypothesis — without it the entry cannot be reviewed.'),
88
- value: z
171
+ ...entryFields,
172
+ },
173
+ annotations: write,
174
+ run: (client, args) => client.action(api.dossiersApi.saveFinding, {
175
+ apiKey: requireApiKey(),
176
+ ...args,
177
+ }),
178
+ });
179
+ export const recordDecision = defineTool({
180
+ name: 'record_decision',
181
+ title: 'Record a decision',
182
+ description: `Settle a question with a decision the engineer has CONFIRMED: what was decided, why, and the clause it rests on. The question leaves the open items and keeps its evidence; the decision becomes the entry a reviewer reads first.
183
+
184
+ Only call this once the user has confirmed the decision, never on your own reasoning. Name the retained option in \`retainedOption\` when the question listed some (the name as it was given; an option not listed yet is created). Put the number in \`value\` when the decision is a value, and set \`confidence\`. Returns { entryId }.`,
185
+ inputSchema: {
186
+ questionId,
187
+ retainedOption: z
89
188
  .string()
90
189
  .max(120)
91
190
  .optional()
92
- .describe('The retained value WITH its unit, e.g. "30°", "1,25 kN/m²".'),
93
- confidence: z
94
- .enum(['established', 'judgement', 'to_confirm'])
95
- .optional()
96
- .describe('How firm the value is. See the tool description.'),
97
- normCode: z
98
- .string()
99
- .max(40)
100
- .optional()
101
- .describe('Norm this comes from, e.g. "SIA 261".'),
102
- sectionPath: z
103
- .string()
104
- .max(60)
105
- .optional()
106
- .describe('Exact section path, e.g. "14.2".'),
107
- page: z
108
- .number()
109
- .int()
110
- .positive()
111
- .optional()
112
- .describe('Page in the norm, so a human can verify it in their PDF.'),
191
+ .describe('The option that was retained, by name.'),
192
+ ...entryFields,
113
193
  },
114
194
  annotations: write,
115
- run: (client, args) => client.action(api.dossiersApi.saveFinding, {
195
+ run: (client, args) => client.action(api.dossiersApi.recordDecision, {
116
196
  apiKey: requireApiKey(),
117
197
  ...args,
118
198
  }),
@@ -120,7 +200,7 @@ Set \`confidence\` whenever the entry is a value: established (computed or read
120
200
  export const loadDossier = defineTool({
121
201
  name: 'load_dossier',
122
202
  title: 'Load a dossier',
123
- description: 'Reload everything a dossier holds: retained values, what they rest on, site observations, open questions and the comments a colleague left. UNRESOLVED QUESTIONS COME FIRST — read them before anything else and tell the user what is still open. Call this whenever the user returns to a project ("reprends le dossier X", "on en était où sur Y") BEFORE answering anything about it, so you build on what was decided instead of deciding it again. Accepts either the id or the name the user uses.',
203
+ description: 'Reload everything a dossier holds: its questions with their evidence, options and decisions, then the evidence not yet filed under a question, then the comments a colleague left. OPEN QUESTIONS COME FIRST — read them before anything else and tell the user what is still open. Call this whenever the user returns to a project ("reprends le dossier X", "on en était où sur Y") BEFORE answering anything about it, so you build on what was decided instead of deciding it again. Accepts either the id or the name the user uses.',
124
204
  inputSchema: {
125
205
  dossierId: z
126
206
  .string()
@@ -143,23 +223,20 @@ export const loadDossier = defineTool({
143
223
  });
144
224
  export const resolveQuestion = defineTool({
145
225
  name: 'resolve_question',
146
- title: 'Close an open question',
147
- description: 'Mark a question entry as settled, once it actually is. The entry stays in the dossier — the trail of what was once uncertain is part of the record — but it stops surfacing at the top on reload. Only call this when the user has confirmed the answer, never on your own reasoning.',
226
+ title: 'Close a question',
227
+ description: 'Close a question that stopped mattering WITHOUT a decision: the variant was abandoned, the client withdrew the request, the point became moot. The question stays in the dossier, marked closed, with its evidence. To settle a question with an answer, call record_decision instead. Pass resolved: false to reopen a question. Only call this when the user says so, never on your own reasoning.',
148
228
  inputSchema: {
149
- entryId: z
150
- .string()
151
- .min(1)
152
- .describe('Entry id of the question, from load_dossier.'),
229
+ questionId,
153
230
  resolved: z
154
231
  .boolean()
155
232
  .optional()
156
- .describe('Defaults to true. Pass false to reopen a question.'),
233
+ .describe('Defaults to true (close). Pass false to reopen.'),
157
234
  },
158
235
  annotations: { ...write, idempotentHint: true },
159
236
  run: async (client, args) => {
160
237
  await client.action(api.dossiersApi.resolveQuestion, {
161
238
  apiKey: requireApiKey(),
162
- entryId: args.entryId,
239
+ questionId: args.questionId,
163
240
  resolved: args.resolved,
164
241
  });
165
242
  return { ok: true };
@@ -168,7 +245,9 @@ export const resolveQuestion = defineTool({
168
245
  export const dossierTools = [
169
246
  listDossiers,
170
247
  openDossier,
248
+ openQuestion,
171
249
  saveFinding,
250
+ recordDecision,
172
251
  loadDossier,
173
252
  resolveQuestion,
174
253
  ];
@@ -44,7 +44,7 @@ export const listNorms = defineTool({
44
44
  export const getToc = defineTool({
45
45
  name: 'get_toc',
46
46
  title: 'Get table of contents',
47
- description: "Get the high-level table of contents for a norm. By default returns only top-level chapters (depth=1) to stay light. Call get_subtree on a specific chapter's path to drill into sections + subsections. Increase maxDepth if you need a wider overview (cost: response size grows fast). Each node has nodeId, path, title, summary, pageStart, pageEnd, depth, and children (empty at the maxDepth boundary).",
47
+ description: "Get the high-level table of contents for a norm. By default returns only top-level chapters (depth=1) to stay light. Call get_subtree on a specific chapter's path to drill into sections + subsections. Increase maxDepth if you need a wider overview (cost: response size grows fast). Each node has path, title, summary, pageStart, pageEnd, depth, and children (empty at the maxDepth boundary).",
48
48
  inputSchema: {
49
49
  norm,
50
50
  maxDepth: z
@@ -76,7 +76,7 @@ export const getToc = defineTool({
76
76
  export const getSubtree = defineTool({
77
77
  name: 'get_subtree',
78
78
  title: 'Get subtree',
79
- description: 'Drill down into a specific chapter or section. Returns the subtree rooted at `path` with optional depth limit (relative to the root). Use this after get_toc to explore one chapter in detail without fetching the entire TOC. Each node has nodeId, path, title, summary, depth, pageStart, pageEnd, and recursive children.',
79
+ description: 'Drill down into a specific chapter or section. Returns the subtree rooted at `path` with optional depth limit (relative to the root). Use this after get_toc to explore one chapter in detail without fetching the entire TOC. Each node has path, title, summary, depth, pageStart, pageEnd, and recursive children.',
80
80
  inputSchema: {
81
81
  norm,
82
82
  path: sectionPath.describe('Section path to root the subtree at, e.g. "14" or "14.2".'),
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@stratta/mcp",
3
3
  "mcpName": "ch.stratta/mcp",
4
- "version": "0.9.7",
4
+ "version": "0.10.0",
5
5
  "description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
6
6
  "license": "UNLICENSED",
7
7
  "author": "SmartFlow <hello@stratta.ch>",
@@ -77,10 +77,163 @@ def detect_running_text(doc: "fitz.Document") -> set[str]:
77
77
  CHAP_UPPER_RE = re.compile(
78
78
  r"^(\d+)\s+([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{3,120})\s*$", re.MULTILINE
79
79
  )
80
+ # Norms adopted from CEN (SIA 262.6xx, SIA 267.1xx) set their headings in
81
+ # sentence case, which CHAP_UPPER_RE rejects by design. Used only as a second
82
+ # pass, when the uppercase form found almost nothing — on a native SIA norm it
83
+ # would promote body sentences to chapters.
84
+ CHAP_MIXED_RE = re.compile(
85
+ r"^(\d{1,2})\s+([A-ZÀ-Þ][A-Za-zÀ-ÿ][^\n]{2,90})\s*$", re.MULTILINE
86
+ )
80
87
  ANNEX_BODY_RE = re.compile(
81
88
  r"^(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)$", re.MULTILINE
82
89
  )
83
90
 
91
+ # A heading split across two lines: the number alone, the title underneath.
92
+ # Comes from the numbering column of CEN-style layouts, where PyMuPDF reads the
93
+ # column before the text. Left as-is, every regex below misses the heading.
94
+ SPLIT_NUM_RE = re.compile(r"^\s*(\d{1,2}(?:\.\d{1,2}){0,3})\s*$")
95
+ SPLIT_TITLE_RE = re.compile(r"^\s*([A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-][^\n]{2,90})\s*$")
96
+ # Words that open a continuing sentence, never a heading.
97
+ SPLIT_STOP_RE = re.compile(
98
+ r"^(Le |La |Les |Il |Elle |Dans |Pour |Selon |Cette |Ce |Ces |Si |En |Au |Aux |De |Des |Du |Un |Une )",
99
+ )
100
+
101
+
102
+ def join_split_headings(text: str) -> str:
103
+ """Rewrite `12\\nTitre` as `12 Titre` so the heading regexes can see it.
104
+
105
+ Conservative on purpose: the title line must look like a title (starts
106
+ uppercase, no sentence-ending punctuation, not a sentence opener). A false
107
+ join invents a chapter, which is worse than missing one.
108
+ """
109
+ lines = text.splitlines()
110
+ out: list[str] = []
111
+ i = 0
112
+ while i < len(lines):
113
+ m = SPLIT_NUM_RE.match(lines[i])
114
+ if m and i + 1 < len(lines):
115
+ nxt = lines[i + 1]
116
+ t = SPLIT_TITLE_RE.match(nxt)
117
+ if t and not nxt.rstrip().endswith((".", ",", ";", ":")) and not SPLIT_STOP_RE.match(t.group(1)):
118
+ out.append(f"{m.group(1)} {t.group(1).strip()}")
119
+ i += 2
120
+ continue
121
+ out.append(lines[i])
122
+ i += 1
123
+ return "\n".join(out)
124
+
125
+
126
+ def heading_text(doc: "fitz.Document", pageno: int) -> str:
127
+ """Page text prepared for heading detection (never for section content)."""
128
+ return join_split_headings(doc[pageno].get_text("text"))
129
+
130
+
131
+ NOT_A_TITLE_RE = re.compile(
132
+ r"^(?:EN|SN|ISO|DIN|SIA|NOTE|Tableau|Figure|Table|Bild)\b|^\W|\.$", re.IGNORECASE
133
+ )
134
+ # A heading cut mid-phrase by the column break: "Résistance à la flexion au".
135
+ TRUNCATED_RE = re.compile(
136
+ r"\b(?:ou|et|au|aux|de|des|du|le|la|les|un|une|dans|pour|par|sur|avec|sans|selon|entre)$",
137
+ re.IGNORECASE,
138
+ )
139
+
140
+
141
+ def plausible_title(title: str) -> bool:
142
+ """Reject what a numbered line can be other than a heading.
143
+
144
+ A stray table cell ("I ~ ~"), a normative reference ("EN 12063:2024 (F)") or
145
+ a wrapped sentence all match the heading shape. Promoting one invents a
146
+ chapter, and an invented chapter swallows the page range of a real one.
147
+ """
148
+ t = title.strip()
149
+ if not 3 <= len(t) <= 90:
150
+ return False
151
+ # A normative heading is capitalised. A lowercase one is a table row that
152
+ # happens to sit behind a number ("2.3 retrait des obstacles").
153
+ if not (t[0].isupper() or t[0].isdigit()):
154
+ return False
155
+ if NOT_A_TITLE_RE.search(t):
156
+ return False
157
+ if SPLIT_STOP_RE.match(t) or TRUNCATED_RE.search(t):
158
+ return False
159
+ letters = sum(1 for c in t if c.isalpha() or c.isspace())
160
+ return letters / len(t) >= 0.7
161
+
162
+
163
+ def dense_heading_pages(doc: "fitz.Document", toc_pages: set[int]) -> set[int]:
164
+ """Pages listing many numbered headings: a contents page, whatever its
165
+ typography. Detecting it by leader dots alone misses the ones set without
166
+ them, and every heading read there points at the contents page instead of
167
+ the section it names."""
168
+ dense: set[int] = set()
169
+ for i in range(doc.page_count):
170
+ if i in toc_pages:
171
+ continue
172
+ found = {m.group(1) for m in SUB_RE.finditer(heading_text(doc, i))}
173
+ # A page of the body drills into one chapter; a contents page walks
174
+ # across several. Counting headings alone would drop a dense page of
175
+ # definitions (3.1 … 3.8), which is real content.
176
+ if len(found) >= 5 and len({n.split(".", 1)[0] for n in found}) >= 3:
177
+ dense.add(i)
178
+ return dense
179
+
180
+
181
+ def prune_chapters(chapters: dict[str, dict]) -> dict[str, dict]:
182
+ """Keep the run of chapter numbers that reads like a table of contents.
183
+
184
+ Real chapters are consecutive and move forward through the document. A lone
185
+ "25" between chapters 1 and 2, or a chapter that starts thirty pages before
186
+ the one preceding it, is a detection artefact.
187
+ """
188
+ numbered = sorted(((int(k), k) for k in chapters if k.isdigit()))
189
+ if not numbered:
190
+ return chapters
191
+
192
+ # A norm with N detected chapters does not have a chapter 40. Missing a few
193
+ # headings is normal; a number far past the count is a table cell.
194
+ ceiling = 2 * len(numbered) + 3
195
+ numbered = [(n, k) for n, k in numbered if n <= ceiling]
196
+ if not numbered:
197
+ return chapters
198
+
199
+ # Longest run whose pages move forward, so one bad page does not discard
200
+ # every chapter after it.
201
+ best = [1] * len(numbered)
202
+ prev = [-1] * len(numbered)
203
+ for i in range(len(numbered)):
204
+ page_i = chapters[numbered[i][1]]["pageStart"]
205
+ for j in range(i):
206
+ if chapters[numbered[j][1]]["pageStart"] <= page_i and best[j] + 1 > best[i]:
207
+ best[i], prev[i] = best[j] + 1, j
208
+ idx = best.index(max(best))
209
+ chain = []
210
+ while idx != -1:
211
+ chain.append(numbered[idx][1])
212
+ idx = prev[idx]
213
+
214
+ # Missing a heading or two leaves a small gap; doubling (10 → 20 → 30) means
215
+ # the numbers stopped being chapters and started being table rows.
216
+ kept: dict[str, dict] = {}
217
+ previous = None
218
+ for key in reversed(chain):
219
+ n = int(key)
220
+ if previous is not None and n - previous > 3 and n >= 2 * previous:
221
+ break
222
+ kept[key] = chapters[key]
223
+ previous = n
224
+
225
+ # Annexes close a norm, in order. One that lands before the last chapter was
226
+ # read off the contents page, and its page range would swallow the document.
227
+ last_page = max((m["pageStart"] for m in kept.values()), default=0)
228
+ for key, meta in sorted(
229
+ ((k, m) for k, m in chapters.items() if not k.isdigit()), key=lambda kv: kv[0]
230
+ ):
231
+ if meta["pageStart"] < last_page:
232
+ continue
233
+ kept[key] = meta
234
+ last_page = meta["pageStart"]
235
+ return kept
236
+
84
237
 
85
238
  def extract_chapters(
86
239
  doc: "fitz.Document", toc_pages: set[int] | None = None
@@ -105,32 +258,64 @@ def extract_chapters(
105
258
  if chapters:
106
259
  return chapters
107
260
 
108
- # Fallback: no bookmarks → scan body for uppercase chapter headings.
109
- skip = toc_pages or set()
110
- for pageno in range(doc.page_count):
111
- if pageno in skip:
112
- continue
113
- text = doc[pageno].get_text("text")
114
- for m in CHAP_UPPER_RE.finditer(text):
115
- num = m.group(1)
116
- title = clean(m.group(2))
117
- if int(num) > 50:
118
- continue
119
- if num in chapters:
120
- continue
121
- chapters[num] = {"title": title, "pageStart": pageno + 1}
122
- for m in ANNEX_BODY_RE.finditer(text):
123
- letter = m.group(1)
124
- kind = m.group(2) or ""
125
- rest = clean(m.group(3) or "")
126
- path = f"Annexe {letter}"
127
- if path in chapters:
261
+ # Fallback: no bookmarks → scan body for chapter headings.
262
+ skip = set(toc_pages or set())
263
+
264
+ def scan(pattern: "re.Pattern[str]", skip_pages: set[int]) -> dict[str, dict]:
265
+ found: dict[str, dict] = {}
266
+ for pageno in range(doc.page_count):
267
+ if pageno in skip_pages:
128
268
  continue
129
- t = path + (f" ({kind})" if kind else "")
130
- if rest and rest.lower() != path.lower():
131
- t += f" — {rest}"
132
- chapters[path] = {"title": t, "pageStart": pageno + 1}
133
- return chapters
269
+ text = heading_text(doc, pageno)
270
+ for m in pattern.finditer(text):
271
+ num, title = m.group(1), clean(m.group(2))
272
+ if int(num) > 50 or num in found or not plausible_title(title):
273
+ continue
274
+ found[num] = {"title": title, "pageStart": pageno + 1}
275
+ for m in ANNEX_BODY_RE.finditer(text):
276
+ letter, kind = m.group(1), m.group(2) or ""
277
+ rest = clean(m.group(3) or "")
278
+ path = f"Annexe {letter}"
279
+ if path in found:
280
+ continue
281
+ t = path + (f" ({kind})" if kind else "")
282
+ if rest and rest.lower() != path.lower():
283
+ t += f" — {rest}"
284
+ found[path] = {"title": t, "pageStart": pageno + 1}
285
+ return found
286
+
287
+ def toc_like_pages(found: dict[str, dict]) -> set[int]:
288
+ """Pages holding three or more chapter headings are the table of
289
+ contents, whatever the typography. Without this every chapter of such a
290
+ norm starts on the contents page, and every section body is wrong."""
291
+ per_page: dict[int, int] = defaultdict(int)
292
+ for meta in found.values():
293
+ per_page[meta["pageStart"]] += 1
294
+ return {p - 1 for p, n in per_page.items() if n >= 3}
295
+
296
+ for pattern in (CHAP_UPPER_RE, CHAP_MIXED_RE):
297
+ chapters = scan(pattern, skip)
298
+ extra = toc_like_pages(chapters)
299
+ if extra:
300
+ chapters = scan(pattern, skip | extra)
301
+ if len([k for k in chapters if k.isdigit()]) >= 3:
302
+ break
303
+
304
+ return prune_chapters(chapters)
305
+
306
+
307
+ def chapter_page_spans(chapters: dict[str, dict], n_pages: int) -> dict[str, tuple[int, int]]:
308
+ """First and last page of each numbered chapter, from where the next starts."""
309
+ ordered = sorted(
310
+ ((k, v["pageStart"]) for k, v in chapters.items()),
311
+ key=lambda kv: kv[1],
312
+ )
313
+ spans: dict[str, tuple[int, int]] = {}
314
+ for idx, (key, start) in enumerate(ordered):
315
+ end = ordered[idx + 1][1] - 1 if idx + 1 < len(ordered) else n_pages
316
+ if key.isdigit():
317
+ spans[key] = (start, max(start, end))
318
+ return spans
134
319
 
135
320
 
136
321
  def extract_subsections(
@@ -138,13 +323,15 @@ def extract_subsections(
138
323
  chap_prefixes: set[str],
139
324
  toc_pages: set[int],
140
325
  running: set[str],
326
+ chapter_spans: dict[str, tuple[int, int]] | None = None,
141
327
  ) -> dict[str, tuple[str, int]]:
142
328
  """Numeric sub-sections (depths 2-4) detected in body pages."""
143
329
  out: dict[str, tuple[str, int]] = {}
330
+ skip = set(toc_pages) | dense_heading_pages(doc, toc_pages)
144
331
  for i in range(doc.page_count):
145
- if i in toc_pages:
332
+ if i in skip:
146
333
  continue
147
- text = doc[i].get_text("text")
334
+ text = heading_text(doc, i)
148
335
  for m in SUB_RE.finditer(text):
149
336
  num = m.group(1)
150
337
  title = clean(m.group(2))
@@ -153,7 +340,13 @@ def extract_subsections(
153
340
  top = num.split(".", 1)[0]
154
341
  if top not in chap_prefixes:
155
342
  continue
156
- if len(title) < 3 or re.fullmatch(r"[^A-Za-zÀ-ÿ]+", title):
343
+ # A sub-section sits inside its chapter. "1.1" found sixty pages
344
+ # after chapter 1 ended is a numbered line in an annexe, and keeping
345
+ # it hangs an unrelated page under the wrong parent.
346
+ span = chapter_spans.get(top) if chapter_spans else None
347
+ if span and not span[0] <= i + 1 <= span[1]:
348
+ continue
349
+ if not plausible_title(title):
157
350
  continue
158
351
  if num in out:
159
352
  continue
@@ -218,6 +411,44 @@ def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
218
411
  chunks = page_texts[ps - 1 : pe]
219
412
  node["rawText"] = "\n".join(chunks).strip()
220
413
 
414
+ trim_shared_pages(nodes, page_texts)
415
+
416
+
417
+ def trim_shared_pages(nodes: list[dict], page_texts: list[str]) -> None:
418
+ """Cut the opening page where several sections start on it.
419
+
420
+ Page granularity is the unit everywhere else, but two sections beginning on
421
+ the same page would otherwise carry byte-identical text — so a summary reads
422
+ like its neighbour's, and the table of contents misleads before anything is
423
+ even opened.
424
+ """
425
+ by_page: dict[int, list[dict]] = defaultdict(list)
426
+ for node in nodes:
427
+ by_page[node["pageStart"]].append(node)
428
+
429
+ for page, group in by_page.items():
430
+ if len(group) < 2:
431
+ continue
432
+ text = page_texts[page - 1]
433
+ # Where each section's heading sits on that page, in reading order.
434
+ marks: list[tuple[int, dict]] = []
435
+ for node in group:
436
+ title = node["title"].split(" — ")[-1].strip()
437
+ at = text.find(title) if len(title) >= 4 else -1
438
+ if at == -1:
439
+ at = text.find(f"{node['path']} ")
440
+ marks.append((at, node))
441
+ if any(at == -1 for at, _ in marks):
442
+ continue # a heading we cannot locate: leave the whole page in place
443
+ marks.sort(key=lambda m: m[0])
444
+ for i, (at, node) in enumerate(marks):
445
+ end = marks[i + 1][0] if i + 1 < len(marks) else len(text)
446
+ head = text[at:end].strip()
447
+ if not head:
448
+ continue
449
+ rest = "\n".join(page_texts[node["pageStart"] : node["pageEnd"]]).strip()
450
+ node["rawText"] = f"{head}\n{rest}".strip() if rest else head
451
+
221
452
 
222
453
  def extract_figures(
223
454
  doc: "fitz.Document",
@@ -295,7 +526,8 @@ def main() -> None:
295
526
  running = detect_running_text(doc)
296
527
  chapters = extract_chapters(doc, toc_pages)
297
528
  chap_prefixes = {k for k in chapters if k.isdigit()}
298
- subsections = extract_subsections(doc, chap_prefixes, toc_pages, running)
529
+ spans = chapter_page_spans(chapters, doc.page_count)
530
+ subsections = extract_subsections(doc, chap_prefixes, toc_pages, running, spans)
299
531
  nodes = build_tree(chapters, subsections, doc.page_count)
300
532
  extract_section_text(doc, nodes)
301
533
  figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)