@stratta/mcp 0.9.7 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -9
- package/dist/tools/dossier.d.ts +48 -4
- package/dist/tools/dossier.js +130 -51
- package/dist/tools/read.js +2 -2
- package/package.json +1 -1
- package/scripts/ingest-prepass.py +261 -29
package/README.md
CHANGED
|
@@ -20,6 +20,25 @@ MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients
|
|
|
20
20
|
> a shell environment variable. If a key leaks, revoke it immediately at
|
|
21
21
|
> https://stratta.ch/api-keys.
|
|
22
22
|
|
|
23
|
+
## Do you need this package?
|
|
24
|
+
|
|
25
|
+
Often not. Stratta also runs as a **remote connector** — one address, a browser
|
|
26
|
+
sign-in, no key and no Node:
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
https://stratta.ch/mcp
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
That is the shorter path, and the only one that works in an agent running in the
|
|
33
|
+
cloud (claude.ai). See https://stratta.ch/docs/en/guides/connect-remote.
|
|
34
|
+
|
|
35
|
+
This package is what you want when:
|
|
36
|
+
|
|
37
|
+
- you are **ingesting a norm** — it reads a PDF from your disk and runs a Python
|
|
38
|
+
pre-pass, neither of which a remote connector can reach;
|
|
39
|
+
- your agent **cannot open a browser** — CI, a scheduled task, a server;
|
|
40
|
+
- you would simply rather run the server yourself.
|
|
41
|
+
|
|
23
42
|
## Install
|
|
24
43
|
|
|
25
44
|
### Claude Code (recommended)
|
|
@@ -148,15 +167,17 @@ settings come from environment variables — see [`.env.example`](./.env.example
|
|
|
148
167
|
| `get_figure` | Retrieve a figure inline (base64 ImageContent) + public URL. |
|
|
149
168
|
| `get_cross_refs` | Outgoing cross-refs from a section to other norms. |
|
|
150
169
|
|
|
151
|
-
**Dossier** (
|
|
152
|
-
|
|
153
|
-
| Tool | Purpose
|
|
154
|
-
| ------------------ |
|
|
155
|
-
| `list_dossiers` | Your organisation's dossiers, most recently touched first, with open counts.
|
|
156
|
-
| `open_dossier` | Open a project's dossier, creating it if needed. Idempotent on the name.
|
|
157
|
-
| `
|
|
158
|
-
| `
|
|
159
|
-
| `
|
|
170
|
+
**Dossier** (7 tools — keep what was decided on a project):
|
|
171
|
+
|
|
172
|
+
| Tool | Purpose |
|
|
173
|
+
| ------------------ | ---------------------------------------------------------------------------------------- |
|
|
174
|
+
| `list_dossiers` | Your organisation's dossiers, most recently touched first, with open-question counts. |
|
|
175
|
+
| `open_dossier` | Open a project's dossier, creating it if needed. Idempotent on the name. |
|
|
176
|
+
| `open_question` | Open one question to settle, with optional named options. Idempotent on the title. |
|
|
177
|
+
| `save_finding` | Record one piece of evidence: a cited article, a retained value and why, an observation. |
|
|
178
|
+
| `record_decision` | Settle a question with a decision the engineer has confirmed, and the retained option. |
|
|
179
|
+
| `load_dossier` | Reload everything: questions with their evidence and decisions, open ones first. |
|
|
180
|
+
| `resolve_question` | Close a question without a decision, or reopen one. The evidence stays. |
|
|
160
181
|
|
|
161
182
|
A dossier is read, annotated, reviewed and exported from
|
|
162
183
|
[stratta.ch/dossiers](https://stratta.ch/dossiers).
|
package/dist/tools/dossier.d.ts
CHANGED
|
@@ -4,14 +4,34 @@ export declare const openDossier: import("./define.js").ToolDef<{
|
|
|
4
4
|
name: z.ZodString;
|
|
5
5
|
reference: z.ZodOptional<z.ZodString>;
|
|
6
6
|
}>;
|
|
7
|
+
export declare const openQuestion: import("./define.js").ToolDef<{
|
|
8
|
+
dossierId: z.ZodString;
|
|
9
|
+
title: z.ZodString;
|
|
10
|
+
body: z.ZodOptional<z.ZodString>;
|
|
11
|
+
options: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
12
|
+
}>;
|
|
7
13
|
export declare const saveFinding: import("./define.js").ToolDef<{
|
|
14
|
+
title: z.ZodString;
|
|
15
|
+
detail: z.ZodOptional<z.ZodString>;
|
|
16
|
+
value: z.ZodOptional<z.ZodString>;
|
|
17
|
+
confidence: z.ZodOptional<z.ZodEnum<{
|
|
18
|
+
established: "established";
|
|
19
|
+
judgement: "judgement";
|
|
20
|
+
to_confirm: "to_confirm";
|
|
21
|
+
}>>;
|
|
22
|
+
normCode: z.ZodOptional<z.ZodString>;
|
|
23
|
+
sectionPath: z.ZodOptional<z.ZodString>;
|
|
24
|
+
page: z.ZodOptional<z.ZodNumber>;
|
|
25
|
+
normEdition: z.ZodOptional<z.ZodString>;
|
|
8
26
|
dossierId: z.ZodString;
|
|
27
|
+
questionId: z.ZodOptional<z.ZodString>;
|
|
9
28
|
kind: z.ZodEnum<{
|
|
10
29
|
reference: "reference";
|
|
11
30
|
hypothesis: "hypothesis";
|
|
12
31
|
observation: "observation";
|
|
13
|
-
question: "question";
|
|
14
32
|
}>;
|
|
33
|
+
}>;
|
|
34
|
+
export declare const recordDecision: import("./define.js").ToolDef<{
|
|
15
35
|
title: z.ZodString;
|
|
16
36
|
detail: z.ZodOptional<z.ZodString>;
|
|
17
37
|
value: z.ZodOptional<z.ZodString>;
|
|
@@ -23,13 +43,16 @@ export declare const saveFinding: import("./define.js").ToolDef<{
|
|
|
23
43
|
normCode: z.ZodOptional<z.ZodString>;
|
|
24
44
|
sectionPath: z.ZodOptional<z.ZodString>;
|
|
25
45
|
page: z.ZodOptional<z.ZodNumber>;
|
|
46
|
+
normEdition: z.ZodOptional<z.ZodString>;
|
|
47
|
+
questionId: z.ZodString;
|
|
48
|
+
retainedOption: z.ZodOptional<z.ZodString>;
|
|
26
49
|
}>;
|
|
27
50
|
export declare const loadDossier: import("./define.js").ToolDef<{
|
|
28
51
|
dossierId: z.ZodOptional<z.ZodString>;
|
|
29
52
|
name: z.ZodOptional<z.ZodString>;
|
|
30
53
|
}>;
|
|
31
54
|
export declare const resolveQuestion: import("./define.js").ToolDef<{
|
|
32
|
-
|
|
55
|
+
questionId: z.ZodString;
|
|
33
56
|
resolved: z.ZodOptional<z.ZodBoolean>;
|
|
34
57
|
}>;
|
|
35
58
|
export declare const dossierTools: (import("./define.js").ToolDef<{}> | import("./define.js").ToolDef<{
|
|
@@ -37,12 +60,30 @@ export declare const dossierTools: (import("./define.js").ToolDef<{}> | import("
|
|
|
37
60
|
reference: z.ZodOptional<z.ZodString>;
|
|
38
61
|
}> | import("./define.js").ToolDef<{
|
|
39
62
|
dossierId: z.ZodString;
|
|
63
|
+
title: z.ZodString;
|
|
64
|
+
body: z.ZodOptional<z.ZodString>;
|
|
65
|
+
options: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
66
|
+
}> | import("./define.js").ToolDef<{
|
|
67
|
+
title: z.ZodString;
|
|
68
|
+
detail: z.ZodOptional<z.ZodString>;
|
|
69
|
+
value: z.ZodOptional<z.ZodString>;
|
|
70
|
+
confidence: z.ZodOptional<z.ZodEnum<{
|
|
71
|
+
established: "established";
|
|
72
|
+
judgement: "judgement";
|
|
73
|
+
to_confirm: "to_confirm";
|
|
74
|
+
}>>;
|
|
75
|
+
normCode: z.ZodOptional<z.ZodString>;
|
|
76
|
+
sectionPath: z.ZodOptional<z.ZodString>;
|
|
77
|
+
page: z.ZodOptional<z.ZodNumber>;
|
|
78
|
+
normEdition: z.ZodOptional<z.ZodString>;
|
|
79
|
+
dossierId: z.ZodString;
|
|
80
|
+
questionId: z.ZodOptional<z.ZodString>;
|
|
40
81
|
kind: z.ZodEnum<{
|
|
41
82
|
reference: "reference";
|
|
42
83
|
hypothesis: "hypothesis";
|
|
43
84
|
observation: "observation";
|
|
44
|
-
question: "question";
|
|
45
85
|
}>;
|
|
86
|
+
}> | import("./define.js").ToolDef<{
|
|
46
87
|
title: z.ZodString;
|
|
47
88
|
detail: z.ZodOptional<z.ZodString>;
|
|
48
89
|
value: z.ZodOptional<z.ZodString>;
|
|
@@ -54,10 +95,13 @@ export declare const dossierTools: (import("./define.js").ToolDef<{}> | import("
|
|
|
54
95
|
normCode: z.ZodOptional<z.ZodString>;
|
|
55
96
|
sectionPath: z.ZodOptional<z.ZodString>;
|
|
56
97
|
page: z.ZodOptional<z.ZodNumber>;
|
|
98
|
+
normEdition: z.ZodOptional<z.ZodString>;
|
|
99
|
+
questionId: z.ZodString;
|
|
100
|
+
retainedOption: z.ZodOptional<z.ZodString>;
|
|
57
101
|
}> | import("./define.js").ToolDef<{
|
|
58
102
|
dossierId: z.ZodOptional<z.ZodString>;
|
|
59
103
|
name: z.ZodOptional<z.ZodString>;
|
|
60
104
|
}> | import("./define.js").ToolDef<{
|
|
61
|
-
|
|
105
|
+
questionId: z.ZodString;
|
|
62
106
|
resolved: z.ZodOptional<z.ZodBoolean>;
|
|
63
107
|
}>)[];
|
package/dist/tools/dossier.js
CHANGED
|
@@ -3,17 +3,24 @@ import { api } from '../client.js';
|
|
|
3
3
|
import { requireApiKey } from '../auth.js';
|
|
4
4
|
import { defineTool } from './define.js';
|
|
5
5
|
/**
|
|
6
|
-
* The
|
|
6
|
+
* The seven dossier tools: how the work survives the conversation.
|
|
7
7
|
*
|
|
8
8
|
* Everything else here reads a norm. These write down what was decided with
|
|
9
9
|
* it, so the next conversation starts where the last one stopped and a human
|
|
10
10
|
* can review it in the app.
|
|
11
11
|
*
|
|
12
|
+
* A dossier is a set of QUESTIONS. Each question holds the evidence gathered
|
|
13
|
+
* for it, the options on the table and, once the engineer has confirmed it, a
|
|
14
|
+
* decision. The question is the unit of work: it is what gets assigned,
|
|
15
|
+
* reviewed and closed. The first model was a flat journal, and the one real
|
|
16
|
+
* dossier it produced held a single "question" with eight sub-questions in its
|
|
17
|
+
* body — unassignable, unanswerable one at a time, unreadable.
|
|
18
|
+
*
|
|
12
19
|
* The descriptions carry more instruction than the read tools do, on purpose.
|
|
13
20
|
* An agent left to guess will either record nothing — and the feature is dead
|
|
14
21
|
* — or record every sentence it produces, which buries the three decisions
|
|
15
22
|
* that mattered. `get_methodology` tells it when to open a dossier; these tell
|
|
16
|
-
* it what is worth putting in one.
|
|
23
|
+
* it what is worth putting in one, and in what shape.
|
|
17
24
|
*/
|
|
18
25
|
const write = { readOnlyHint: false, openWorldHint: false };
|
|
19
26
|
const readOnly = { readOnlyHint: true, openWorldHint: false };
|
|
@@ -21,10 +28,57 @@ const dossierId = z
|
|
|
21
28
|
.string()
|
|
22
29
|
.min(1)
|
|
23
30
|
.describe('Dossier id returned by open_dossier or list_dossiers.');
|
|
31
|
+
const questionId = z
|
|
32
|
+
.string()
|
|
33
|
+
.min(1)
|
|
34
|
+
.describe('Question id returned by open_question or load_dossier.');
|
|
35
|
+
/** The fields a piece of evidence or a decision carries. */
|
|
36
|
+
const entryFields = {
|
|
37
|
+
title: z
|
|
38
|
+
.string()
|
|
39
|
+
.min(1)
|
|
40
|
+
.max(200)
|
|
41
|
+
.describe('The fact in one line, as an engineer would write it in a report. One clause, one value or one observation; not a theme.'),
|
|
42
|
+
detail: z
|
|
43
|
+
.string()
|
|
44
|
+
.max(4000)
|
|
45
|
+
.optional()
|
|
46
|
+
.describe('The reasoning, in one or two sentences: why this value, what it rests on, what disagreed. Required in practice for a hypothesis.'),
|
|
47
|
+
value: z
|
|
48
|
+
.string()
|
|
49
|
+
.max(120)
|
|
50
|
+
.optional()
|
|
51
|
+
.describe('The retained value WITH its unit, e.g. "30°", "1,25 kN/m²".'),
|
|
52
|
+
confidence: z
|
|
53
|
+
.enum(['established', 'judgement', 'to_confirm'])
|
|
54
|
+
.optional()
|
|
55
|
+
.describe('How firm the value is. See the tool description.'),
|
|
56
|
+
normCode: z
|
|
57
|
+
.string()
|
|
58
|
+
.max(40)
|
|
59
|
+
.optional()
|
|
60
|
+
.describe('Norm this comes from, e.g. "SIA 267".'),
|
|
61
|
+
sectionPath: z
|
|
62
|
+
.string()
|
|
63
|
+
.max(60)
|
|
64
|
+
.optional()
|
|
65
|
+
.describe('Exact section path, e.g. "9.5.2.1".'),
|
|
66
|
+
page: z
|
|
67
|
+
.number()
|
|
68
|
+
.int()
|
|
69
|
+
.positive()
|
|
70
|
+
.optional()
|
|
71
|
+
.describe('Page in the norm, so a human can verify it in their PDF.'),
|
|
72
|
+
normEdition: z
|
|
73
|
+
.string()
|
|
74
|
+
.max(40)
|
|
75
|
+
.optional()
|
|
76
|
+
.describe('The edition cited, e.g. "2013" or "2020 + rect. 2022", when list_norms gives it.'),
|
|
77
|
+
};
|
|
24
78
|
export const listDossiers = defineTool({
|
|
25
79
|
name: 'list_dossiers',
|
|
26
80
|
title: 'List dossiers',
|
|
27
|
-
description: 'List the dossiers of YOUR organisation, most recently touched first. A dossier is a project record: what was decided, what it rests on
|
|
81
|
+
description: 'List the dossiers of YOUR organisation, most recently touched first. A dossier is a project record: the questions to settle, what was decided, what it rests on. Call this when the user mentions a project by name, or asks what they were working on. Each row carries the count of entries and of OPEN questions — a dossier with open questions is the one to resume.',
|
|
28
82
|
inputSchema: {},
|
|
29
83
|
annotations: readOnly,
|
|
30
84
|
run: async (client) => ({
|
|
@@ -36,7 +90,7 @@ export const listDossiers = defineTool({
|
|
|
36
90
|
export const openDossier = defineTool({
|
|
37
91
|
name: 'open_dossier',
|
|
38
92
|
title: 'Open a dossier',
|
|
39
|
-
description: "Open the dossier for a project, creating it if it does not exist yet. IDEMPOTENT on the name: calling it twice with the same name returns the same dossier rather than splitting a project's history in two. Call this when the user starts working on a named project and there is something worth keeping — a retained value, an assumption, a site observation
|
|
93
|
+
description: "Open the dossier for a project, creating it if it does not exist yet. IDEMPOTENT on the name: calling it twice with the same name returns the same dossier rather than splitting a project's history in two. Call this when the user starts working on a named project and there is something worth keeping — a question to settle, a retained value, an assumption, a site observation. Do not open one for a passing lookup. Returns { dossierId, created }.",
|
|
40
94
|
inputSchema: {
|
|
41
95
|
name: z
|
|
42
96
|
.string()
|
|
@@ -56,63 +110,89 @@ export const openDossier = defineTool({
|
|
|
56
110
|
reference: args.reference,
|
|
57
111
|
}),
|
|
58
112
|
});
|
|
113
|
+
export const openQuestion = defineTool({
|
|
114
|
+
name: 'open_question',
|
|
115
|
+
title: 'Open a question',
|
|
116
|
+
description: `Open ONE question to settle in a dossier, BEFORE gathering evidence for it. A question is the unit of work: "Can pile P38 be kept?", "Which φ'k do we retain for the fill?", "Is kinematic interaction to be checked?". Evidence saved with save_finding and the eventual record_decision hang off it.
|
|
117
|
+
|
|
118
|
+
ONE QUESTION PER THING TO SETTLE. Never put a list of sub-questions in the body: a question that bundles eight cannot be assigned, answered or closed one at a time. Open eight questions.
|
|
119
|
+
|
|
120
|
+
IDEMPOTENT on the title: reopening the same wording returns the question already open, with its evidence. Name the options on the table in \`options\` when the user or the norm puts several forward ("abandon the pile", "recover it by re-concreting plus integrity tests", "redistribute onto P37/P39"). Returns { questionId, created }.`,
|
|
121
|
+
inputSchema: {
|
|
122
|
+
dossierId,
|
|
123
|
+
title: z
|
|
124
|
+
.string()
|
|
125
|
+
.min(1)
|
|
126
|
+
.max(200)
|
|
127
|
+
.describe('The question in one line, phrased as a question.'),
|
|
128
|
+
body: z
|
|
129
|
+
.string()
|
|
130
|
+
.max(4000)
|
|
131
|
+
.optional()
|
|
132
|
+
.describe('What is at stake and why it is not settled, in two or three sentences. Not a list of sub-questions.'),
|
|
133
|
+
options: z
|
|
134
|
+
.array(z.string().min(1).max(120))
|
|
135
|
+
.max(10)
|
|
136
|
+
.optional()
|
|
137
|
+
.describe('The ways this could be settled, one short name each.'),
|
|
138
|
+
},
|
|
139
|
+
annotations: { ...write, idempotentHint: true },
|
|
140
|
+
run: (client, args) => client.action(api.dossiersApi.openQuestion, {
|
|
141
|
+
apiKey: requireApiKey(),
|
|
142
|
+
dossierId: args.dossierId,
|
|
143
|
+
title: args.title,
|
|
144
|
+
body: args.body,
|
|
145
|
+
options: args.options,
|
|
146
|
+
}),
|
|
147
|
+
});
|
|
59
148
|
export const saveFinding = defineTool({
|
|
60
149
|
name: 'save_finding',
|
|
61
|
-
title: '
|
|
62
|
-
description: `Record ONE
|
|
150
|
+
title: 'Record evidence in a dossier',
|
|
151
|
+
description: `Record ONE piece of evidence in a dossier. Call it as you work, not in a batch at the end — a finding saved when it is made carries the reasoning that produced it.
|
|
63
152
|
|
|
64
153
|
WHAT TO RECORD, by kind:
|
|
65
154
|
- reference: what a norm says, that the project relies on. ALWAYS fill normCode + sectionPath (+ page). If you cannot cite it, it is not a reference.
|
|
66
155
|
- hypothesis: a value the engineer RETAINS, and why. This is the one that matters most in geotechnics, where a retained value is a judgement between disagreeing measurements rather than the output of a formula. Put the number in \`value\` and the reasoning in \`detail\`.
|
|
67
156
|
- observation: what the site, a borehole, or a survey showed. Include the date in \`detail\` when known.
|
|
68
|
-
- question: something not settled. These surface FIRST when the dossier is reloaded, so record them even when you cannot answer — especially then.
|
|
69
157
|
|
|
70
|
-
|
|
158
|
+
KEEP IT SHORT: one clause, one value or one fact per call, in one or two sentences. Three short entries beat one paragraph that mixes the citation, the reasoning and the consequence. A question is NOT recorded here: call open_question. A decision is NOT recorded here: call record_decision.
|
|
159
|
+
|
|
160
|
+
FILE IT: pass \`questionId\` so the entry lands under the question it serves. If you do not yet know which question it serves, save it anyway without one — the engineer files it later.
|
|
71
161
|
|
|
72
162
|
Set \`confidence\` whenever the entry is a value: established (computed or read directly), judgement (the engineer chose it), to_confirm (provisional, needs a test or a check). Confusing those three is the professional fault this field exists to prevent.`,
|
|
73
163
|
inputSchema: {
|
|
74
164
|
dossierId,
|
|
165
|
+
questionId: questionId
|
|
166
|
+
.optional()
|
|
167
|
+
.describe('The question this evidence serves, from open_question or load_dossier. Omit only when you do not know yet.'),
|
|
75
168
|
kind: z
|
|
76
|
-
.enum(['reference', 'hypothesis', 'observation'
|
|
169
|
+
.enum(['reference', 'hypothesis', 'observation'])
|
|
77
170
|
.describe('See the tool description: pick by what the entry IS.'),
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
171
|
+
...entryFields,
|
|
172
|
+
},
|
|
173
|
+
annotations: write,
|
|
174
|
+
run: (client, args) => client.action(api.dossiersApi.saveFinding, {
|
|
175
|
+
apiKey: requireApiKey(),
|
|
176
|
+
...args,
|
|
177
|
+
}),
|
|
178
|
+
});
|
|
179
|
+
export const recordDecision = defineTool({
|
|
180
|
+
name: 'record_decision',
|
|
181
|
+
title: 'Record a decision',
|
|
182
|
+
description: `Settle a question with a decision the engineer has CONFIRMED: what was decided, why, and the clause it rests on. The question leaves the open items and keeps its evidence; the decision becomes the entry a reviewer reads first.
|
|
183
|
+
|
|
184
|
+
Only call this once the user has confirmed the decision, never on your own reasoning. Name the retained option in \`retainedOption\` when the question listed some (the name as it was given; an option not listed yet is created). Put the number in \`value\` when the decision is a value, and set \`confidence\`. Returns { entryId }.`,
|
|
185
|
+
inputSchema: {
|
|
186
|
+
questionId,
|
|
187
|
+
retainedOption: z
|
|
89
188
|
.string()
|
|
90
189
|
.max(120)
|
|
91
190
|
.optional()
|
|
92
|
-
.describe('The
|
|
93
|
-
|
|
94
|
-
.enum(['established', 'judgement', 'to_confirm'])
|
|
95
|
-
.optional()
|
|
96
|
-
.describe('How firm the value is. See the tool description.'),
|
|
97
|
-
normCode: z
|
|
98
|
-
.string()
|
|
99
|
-
.max(40)
|
|
100
|
-
.optional()
|
|
101
|
-
.describe('Norm this comes from, e.g. "SIA 261".'),
|
|
102
|
-
sectionPath: z
|
|
103
|
-
.string()
|
|
104
|
-
.max(60)
|
|
105
|
-
.optional()
|
|
106
|
-
.describe('Exact section path, e.g. "14.2".'),
|
|
107
|
-
page: z
|
|
108
|
-
.number()
|
|
109
|
-
.int()
|
|
110
|
-
.positive()
|
|
111
|
-
.optional()
|
|
112
|
-
.describe('Page in the norm, so a human can verify it in their PDF.'),
|
|
191
|
+
.describe('The option that was retained, by name.'),
|
|
192
|
+
...entryFields,
|
|
113
193
|
},
|
|
114
194
|
annotations: write,
|
|
115
|
-
run: (client, args) => client.action(api.dossiersApi.
|
|
195
|
+
run: (client, args) => client.action(api.dossiersApi.recordDecision, {
|
|
116
196
|
apiKey: requireApiKey(),
|
|
117
197
|
...args,
|
|
118
198
|
}),
|
|
@@ -120,7 +200,7 @@ Set \`confidence\` whenever the entry is a value: established (computed or read
|
|
|
120
200
|
export const loadDossier = defineTool({
|
|
121
201
|
name: 'load_dossier',
|
|
122
202
|
title: 'Load a dossier',
|
|
123
|
-
description: 'Reload everything a dossier holds:
|
|
203
|
+
description: 'Reload everything a dossier holds: its questions with their evidence, options and decisions, then the evidence not yet filed under a question, then the comments a colleague left. OPEN QUESTIONS COME FIRST — read them before anything else and tell the user what is still open. Call this whenever the user returns to a project ("reprends le dossier X", "on en était où sur Y") BEFORE answering anything about it, so you build on what was decided instead of deciding it again. Accepts either the id or the name the user uses.',
|
|
124
204
|
inputSchema: {
|
|
125
205
|
dossierId: z
|
|
126
206
|
.string()
|
|
@@ -143,23 +223,20 @@ export const loadDossier = defineTool({
|
|
|
143
223
|
});
|
|
144
224
|
export const resolveQuestion = defineTool({
|
|
145
225
|
name: 'resolve_question',
|
|
146
|
-
title: 'Close
|
|
147
|
-
description: '
|
|
226
|
+
title: 'Close a question',
|
|
227
|
+
description: 'Close a question that stopped mattering WITHOUT a decision: the variant was abandoned, the client withdrew the request, the point became moot. The question stays in the dossier, marked closed, with its evidence. To settle a question with an answer, call record_decision instead. Pass resolved: false to reopen a question. Only call this when the user says so, never on your own reasoning.',
|
|
148
228
|
inputSchema: {
|
|
149
|
-
|
|
150
|
-
.string()
|
|
151
|
-
.min(1)
|
|
152
|
-
.describe('Entry id of the question, from load_dossier.'),
|
|
229
|
+
questionId,
|
|
153
230
|
resolved: z
|
|
154
231
|
.boolean()
|
|
155
232
|
.optional()
|
|
156
|
-
.describe('Defaults to true. Pass false to reopen
|
|
233
|
+
.describe('Defaults to true (close). Pass false to reopen.'),
|
|
157
234
|
},
|
|
158
235
|
annotations: { ...write, idempotentHint: true },
|
|
159
236
|
run: async (client, args) => {
|
|
160
237
|
await client.action(api.dossiersApi.resolveQuestion, {
|
|
161
238
|
apiKey: requireApiKey(),
|
|
162
|
-
|
|
239
|
+
questionId: args.questionId,
|
|
163
240
|
resolved: args.resolved,
|
|
164
241
|
});
|
|
165
242
|
return { ok: true };
|
|
@@ -168,7 +245,9 @@ export const resolveQuestion = defineTool({
|
|
|
168
245
|
export const dossierTools = [
|
|
169
246
|
listDossiers,
|
|
170
247
|
openDossier,
|
|
248
|
+
openQuestion,
|
|
171
249
|
saveFinding,
|
|
250
|
+
recordDecision,
|
|
172
251
|
loadDossier,
|
|
173
252
|
resolveQuestion,
|
|
174
253
|
];
|
package/dist/tools/read.js
CHANGED
|
@@ -44,7 +44,7 @@ export const listNorms = defineTool({
|
|
|
44
44
|
export const getToc = defineTool({
|
|
45
45
|
name: 'get_toc',
|
|
46
46
|
title: 'Get table of contents',
|
|
47
|
-
description: "Get the high-level table of contents for a norm. By default returns only top-level chapters (depth=1) to stay light. Call get_subtree on a specific chapter's path to drill into sections + subsections. Increase maxDepth if you need a wider overview (cost: response size grows fast). Each node has
|
|
47
|
+
description: "Get the high-level table of contents for a norm. By default returns only top-level chapters (depth=1) to stay light. Call get_subtree on a specific chapter's path to drill into sections + subsections. Increase maxDepth if you need a wider overview (cost: response size grows fast). Each node has path, title, summary, pageStart, pageEnd, depth, and children (empty at the maxDepth boundary).",
|
|
48
48
|
inputSchema: {
|
|
49
49
|
norm,
|
|
50
50
|
maxDepth: z
|
|
@@ -76,7 +76,7 @@ export const getToc = defineTool({
|
|
|
76
76
|
export const getSubtree = defineTool({
|
|
77
77
|
name: 'get_subtree',
|
|
78
78
|
title: 'Get subtree',
|
|
79
|
-
description: 'Drill down into a specific chapter or section. Returns the subtree rooted at `path` with optional depth limit (relative to the root). Use this after get_toc to explore one chapter in detail without fetching the entire TOC. Each node has
|
|
79
|
+
description: 'Drill down into a specific chapter or section. Returns the subtree rooted at `path` with optional depth limit (relative to the root). Use this after get_toc to explore one chapter in detail without fetching the entire TOC. Each node has path, title, summary, depth, pageStart, pageEnd, and recursive children.',
|
|
80
80
|
inputSchema: {
|
|
81
81
|
norm,
|
|
82
82
|
path: sectionPath.describe('Section path to root the subtree at, e.g. "14" or "14.2".'),
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stratta/mcp",
|
|
3
3
|
"mcpName": "ch.stratta/mcp",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.10.0",
|
|
5
5
|
"description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
|
|
6
6
|
"license": "UNLICENSED",
|
|
7
7
|
"author": "SmartFlow <hello@stratta.ch>",
|
|
@@ -77,10 +77,163 @@ def detect_running_text(doc: "fitz.Document") -> set[str]:
|
|
|
77
77
|
CHAP_UPPER_RE = re.compile(
|
|
78
78
|
r"^(\d+)\s+([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{3,120})\s*$", re.MULTILINE
|
|
79
79
|
)
|
|
80
|
+
# Norms adopted from CEN (SIA 262.6xx, SIA 267.1xx) set their headings in
|
|
81
|
+
# sentence case, which CHAP_UPPER_RE rejects by design. Used only as a second
|
|
82
|
+
# pass, when the uppercase form found almost nothing — on a native SIA norm it
|
|
83
|
+
# would promote body sentences to chapters.
|
|
84
|
+
CHAP_MIXED_RE = re.compile(
|
|
85
|
+
r"^(\d{1,2})\s+([A-ZÀ-Þ][A-Za-zÀ-ÿ][^\n]{2,90})\s*$", re.MULTILINE
|
|
86
|
+
)
|
|
80
87
|
ANNEX_BODY_RE = re.compile(
|
|
81
88
|
r"^(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)$", re.MULTILINE
|
|
82
89
|
)
|
|
83
90
|
|
|
91
|
+
# A heading split across two lines: the number alone, the title underneath.
|
|
92
|
+
# Comes from the numbering column of CEN-style layouts, where PyMuPDF reads the
|
|
93
|
+
# column before the text. Left as-is, every regex below misses the heading.
|
|
94
|
+
SPLIT_NUM_RE = re.compile(r"^\s*(\d{1,2}(?:\.\d{1,2}){0,3})\s*$")
|
|
95
|
+
SPLIT_TITLE_RE = re.compile(r"^\s*([A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-][^\n]{2,90})\s*$")
|
|
96
|
+
# Words that open a continuing sentence, never a heading.
|
|
97
|
+
SPLIT_STOP_RE = re.compile(
|
|
98
|
+
r"^(Le |La |Les |Il |Elle |Dans |Pour |Selon |Cette |Ce |Ces |Si |En |Au |Aux |De |Des |Du |Un |Une )",
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def join_split_headings(text: str) -> str:
|
|
103
|
+
"""Rewrite `12\\nTitre` as `12 Titre` so the heading regexes can see it.
|
|
104
|
+
|
|
105
|
+
Conservative on purpose: the title line must look like a title (starts
|
|
106
|
+
uppercase, no sentence-ending punctuation, not a sentence opener). A false
|
|
107
|
+
join invents a chapter, which is worse than missing one.
|
|
108
|
+
"""
|
|
109
|
+
lines = text.splitlines()
|
|
110
|
+
out: list[str] = []
|
|
111
|
+
i = 0
|
|
112
|
+
while i < len(lines):
|
|
113
|
+
m = SPLIT_NUM_RE.match(lines[i])
|
|
114
|
+
if m and i + 1 < len(lines):
|
|
115
|
+
nxt = lines[i + 1]
|
|
116
|
+
t = SPLIT_TITLE_RE.match(nxt)
|
|
117
|
+
if t and not nxt.rstrip().endswith((".", ",", ";", ":")) and not SPLIT_STOP_RE.match(t.group(1)):
|
|
118
|
+
out.append(f"{m.group(1)} {t.group(1).strip()}")
|
|
119
|
+
i += 2
|
|
120
|
+
continue
|
|
121
|
+
out.append(lines[i])
|
|
122
|
+
i += 1
|
|
123
|
+
return "\n".join(out)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def heading_text(doc: "fitz.Document", pageno: int) -> str:
|
|
127
|
+
"""Page text prepared for heading detection (never for section content)."""
|
|
128
|
+
return join_split_headings(doc[pageno].get_text("text"))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
NOT_A_TITLE_RE = re.compile(
|
|
132
|
+
r"^(?:EN|SN|ISO|DIN|SIA|NOTE|Tableau|Figure|Table|Bild)\b|^\W|\.$", re.IGNORECASE
|
|
133
|
+
)
|
|
134
|
+
# A heading cut mid-phrase by the column break: "Résistance à la flexion au".
|
|
135
|
+
TRUNCATED_RE = re.compile(
|
|
136
|
+
r"\b(?:ou|et|au|aux|de|des|du|le|la|les|un|une|dans|pour|par|sur|avec|sans|selon|entre)$",
|
|
137
|
+
re.IGNORECASE,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def plausible_title(title: str) -> bool:
|
|
142
|
+
"""Reject what a numbered line can be other than a heading.
|
|
143
|
+
|
|
144
|
+
A stray table cell ("I ~ ~"), a normative reference ("EN 12063:2024 (F)") or
|
|
145
|
+
a wrapped sentence all match the heading shape. Promoting one invents a
|
|
146
|
+
chapter, and an invented chapter swallows the page range of a real one.
|
|
147
|
+
"""
|
|
148
|
+
t = title.strip()
|
|
149
|
+
if not 3 <= len(t) <= 90:
|
|
150
|
+
return False
|
|
151
|
+
# A normative heading is capitalised. A lowercase one is a table row that
|
|
152
|
+
# happens to sit behind a number ("2.3 retrait des obstacles").
|
|
153
|
+
if not (t[0].isupper() or t[0].isdigit()):
|
|
154
|
+
return False
|
|
155
|
+
if NOT_A_TITLE_RE.search(t):
|
|
156
|
+
return False
|
|
157
|
+
if SPLIT_STOP_RE.match(t) or TRUNCATED_RE.search(t):
|
|
158
|
+
return False
|
|
159
|
+
letters = sum(1 for c in t if c.isalpha() or c.isspace())
|
|
160
|
+
return letters / len(t) >= 0.7
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def dense_heading_pages(doc: "fitz.Document", toc_pages: set[int]) -> set[int]:
|
|
164
|
+
"""Pages listing many numbered headings: a contents page, whatever its
|
|
165
|
+
typography. Detecting it by leader dots alone misses the ones set without
|
|
166
|
+
them, and every heading read there points at the contents page instead of
|
|
167
|
+
the section it names."""
|
|
168
|
+
dense: set[int] = set()
|
|
169
|
+
for i in range(doc.page_count):
|
|
170
|
+
if i in toc_pages:
|
|
171
|
+
continue
|
|
172
|
+
found = {m.group(1) for m in SUB_RE.finditer(heading_text(doc, i))}
|
|
173
|
+
# A page of the body drills into one chapter; a contents page walks
|
|
174
|
+
# across several. Counting headings alone would drop a dense page of
|
|
175
|
+
# definitions (3.1 … 3.8), which is real content.
|
|
176
|
+
if len(found) >= 5 and len({n.split(".", 1)[0] for n in found}) >= 3:
|
|
177
|
+
dense.add(i)
|
|
178
|
+
return dense
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def prune_chapters(chapters: dict[str, dict]) -> dict[str, dict]:
|
|
182
|
+
"""Keep the run of chapter numbers that reads like a table of contents.
|
|
183
|
+
|
|
184
|
+
Real chapters are consecutive and move forward through the document. A lone
|
|
185
|
+
"25" between chapters 1 and 2, or a chapter that starts thirty pages before
|
|
186
|
+
the one preceding it, is a detection artefact.
|
|
187
|
+
"""
|
|
188
|
+
numbered = sorted(((int(k), k) for k in chapters if k.isdigit()))
|
|
189
|
+
if not numbered:
|
|
190
|
+
return chapters
|
|
191
|
+
|
|
192
|
+
# A norm with N detected chapters does not have a chapter 40. Missing a few
|
|
193
|
+
# headings is normal; a number far past the count is a table cell.
|
|
194
|
+
ceiling = 2 * len(numbered) + 3
|
|
195
|
+
numbered = [(n, k) for n, k in numbered if n <= ceiling]
|
|
196
|
+
if not numbered:
|
|
197
|
+
return chapters
|
|
198
|
+
|
|
199
|
+
# Longest run whose pages move forward, so one bad page does not discard
|
|
200
|
+
# every chapter after it.
|
|
201
|
+
best = [1] * len(numbered)
|
|
202
|
+
prev = [-1] * len(numbered)
|
|
203
|
+
for i in range(len(numbered)):
|
|
204
|
+
page_i = chapters[numbered[i][1]]["pageStart"]
|
|
205
|
+
for j in range(i):
|
|
206
|
+
if chapters[numbered[j][1]]["pageStart"] <= page_i and best[j] + 1 > best[i]:
|
|
207
|
+
best[i], prev[i] = best[j] + 1, j
|
|
208
|
+
idx = best.index(max(best))
|
|
209
|
+
chain = []
|
|
210
|
+
while idx != -1:
|
|
211
|
+
chain.append(numbered[idx][1])
|
|
212
|
+
idx = prev[idx]
|
|
213
|
+
|
|
214
|
+
# Missing a heading or two leaves a small gap; doubling (10 → 20 → 30) means
|
|
215
|
+
# the numbers stopped being chapters and started being table rows.
|
|
216
|
+
kept: dict[str, dict] = {}
|
|
217
|
+
previous = None
|
|
218
|
+
for key in reversed(chain):
|
|
219
|
+
n = int(key)
|
|
220
|
+
if previous is not None and n - previous > 3 and n >= 2 * previous:
|
|
221
|
+
break
|
|
222
|
+
kept[key] = chapters[key]
|
|
223
|
+
previous = n
|
|
224
|
+
|
|
225
|
+
# Annexes close a norm, in order. One that lands before the last chapter was
|
|
226
|
+
# read off the contents page, and its page range would swallow the document.
|
|
227
|
+
last_page = max((m["pageStart"] for m in kept.values()), default=0)
|
|
228
|
+
for key, meta in sorted(
|
|
229
|
+
((k, m) for k, m in chapters.items() if not k.isdigit()), key=lambda kv: kv[0]
|
|
230
|
+
):
|
|
231
|
+
if meta["pageStart"] < last_page:
|
|
232
|
+
continue
|
|
233
|
+
kept[key] = meta
|
|
234
|
+
last_page = meta["pageStart"]
|
|
235
|
+
return kept
|
|
236
|
+
|
|
84
237
|
|
|
85
238
|
def extract_chapters(
|
|
86
239
|
doc: "fitz.Document", toc_pages: set[int] | None = None
|
|
@@ -105,32 +258,64 @@ def extract_chapters(
|
|
|
105
258
|
if chapters:
|
|
106
259
|
return chapters
|
|
107
260
|
|
|
108
|
-
# Fallback: no bookmarks → scan body for
|
|
109
|
-
skip = toc_pages or set()
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
num = m.group(1)
|
|
116
|
-
title = clean(m.group(2))
|
|
117
|
-
if int(num) > 50:
|
|
118
|
-
continue
|
|
119
|
-
if num in chapters:
|
|
120
|
-
continue
|
|
121
|
-
chapters[num] = {"title": title, "pageStart": pageno + 1}
|
|
122
|
-
for m in ANNEX_BODY_RE.finditer(text):
|
|
123
|
-
letter = m.group(1)
|
|
124
|
-
kind = m.group(2) or ""
|
|
125
|
-
rest = clean(m.group(3) or "")
|
|
126
|
-
path = f"Annexe {letter}"
|
|
127
|
-
if path in chapters:
|
|
261
|
+
# Fallback: no bookmarks → scan body for chapter headings.
|
|
262
|
+
skip = set(toc_pages or set())
|
|
263
|
+
|
|
264
|
+
def scan(pattern: "re.Pattern[str]", skip_pages: set[int]) -> dict[str, dict]:
|
|
265
|
+
found: dict[str, dict] = {}
|
|
266
|
+
for pageno in range(doc.page_count):
|
|
267
|
+
if pageno in skip_pages:
|
|
128
268
|
continue
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
269
|
+
text = heading_text(doc, pageno)
|
|
270
|
+
for m in pattern.finditer(text):
|
|
271
|
+
num, title = m.group(1), clean(m.group(2))
|
|
272
|
+
if int(num) > 50 or num in found or not plausible_title(title):
|
|
273
|
+
continue
|
|
274
|
+
found[num] = {"title": title, "pageStart": pageno + 1}
|
|
275
|
+
for m in ANNEX_BODY_RE.finditer(text):
|
|
276
|
+
letter, kind = m.group(1), m.group(2) or ""
|
|
277
|
+
rest = clean(m.group(3) or "")
|
|
278
|
+
path = f"Annexe {letter}"
|
|
279
|
+
if path in found:
|
|
280
|
+
continue
|
|
281
|
+
t = path + (f" ({kind})" if kind else "")
|
|
282
|
+
if rest and rest.lower() != path.lower():
|
|
283
|
+
t += f" — {rest}"
|
|
284
|
+
found[path] = {"title": t, "pageStart": pageno + 1}
|
|
285
|
+
return found
|
|
286
|
+
|
|
287
|
+
def toc_like_pages(found: dict[str, dict]) -> set[int]:
|
|
288
|
+
"""Pages holding three or more chapter headings are the table of
|
|
289
|
+
contents, whatever the typography. Without this every chapter of such a
|
|
290
|
+
norm starts on the contents page, and every section body is wrong."""
|
|
291
|
+
per_page: dict[int, int] = defaultdict(int)
|
|
292
|
+
for meta in found.values():
|
|
293
|
+
per_page[meta["pageStart"]] += 1
|
|
294
|
+
return {p - 1 for p, n in per_page.items() if n >= 3}
|
|
295
|
+
|
|
296
|
+
for pattern in (CHAP_UPPER_RE, CHAP_MIXED_RE):
|
|
297
|
+
chapters = scan(pattern, skip)
|
|
298
|
+
extra = toc_like_pages(chapters)
|
|
299
|
+
if extra:
|
|
300
|
+
chapters = scan(pattern, skip | extra)
|
|
301
|
+
if len([k for k in chapters if k.isdigit()]) >= 3:
|
|
302
|
+
break
|
|
303
|
+
|
|
304
|
+
return prune_chapters(chapters)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def chapter_page_spans(chapters: dict[str, dict], n_pages: int) -> dict[str, tuple[int, int]]:
|
|
308
|
+
"""First and last page of each numbered chapter, from where the next starts."""
|
|
309
|
+
ordered = sorted(
|
|
310
|
+
((k, v["pageStart"]) for k, v in chapters.items()),
|
|
311
|
+
key=lambda kv: kv[1],
|
|
312
|
+
)
|
|
313
|
+
spans: dict[str, tuple[int, int]] = {}
|
|
314
|
+
for idx, (key, start) in enumerate(ordered):
|
|
315
|
+
end = ordered[idx + 1][1] - 1 if idx + 1 < len(ordered) else n_pages
|
|
316
|
+
if key.isdigit():
|
|
317
|
+
spans[key] = (start, max(start, end))
|
|
318
|
+
return spans
|
|
134
319
|
|
|
135
320
|
|
|
136
321
|
def extract_subsections(
|
|
@@ -138,13 +323,15 @@ def extract_subsections(
|
|
|
138
323
|
chap_prefixes: set[str],
|
|
139
324
|
toc_pages: set[int],
|
|
140
325
|
running: set[str],
|
|
326
|
+
chapter_spans: dict[str, tuple[int, int]] | None = None,
|
|
141
327
|
) -> dict[str, tuple[str, int]]:
|
|
142
328
|
"""Numeric sub-sections (depths 2-4) detected in body pages."""
|
|
143
329
|
out: dict[str, tuple[str, int]] = {}
|
|
330
|
+
skip = set(toc_pages) | dense_heading_pages(doc, toc_pages)
|
|
144
331
|
for i in range(doc.page_count):
|
|
145
|
-
if i in
|
|
332
|
+
if i in skip:
|
|
146
333
|
continue
|
|
147
|
-
text = doc
|
|
334
|
+
text = heading_text(doc, i)
|
|
148
335
|
for m in SUB_RE.finditer(text):
|
|
149
336
|
num = m.group(1)
|
|
150
337
|
title = clean(m.group(2))
|
|
@@ -153,7 +340,13 @@ def extract_subsections(
|
|
|
153
340
|
top = num.split(".", 1)[0]
|
|
154
341
|
if top not in chap_prefixes:
|
|
155
342
|
continue
|
|
156
|
-
|
|
343
|
+
# A sub-section sits inside its chapter. "1.1" found sixty pages
|
|
344
|
+
# after chapter 1 ended is a numbered line in an annexe, and keeping
|
|
345
|
+
# it hangs an unrelated page under the wrong parent.
|
|
346
|
+
span = chapter_spans.get(top) if chapter_spans else None
|
|
347
|
+
if span and not span[0] <= i + 1 <= span[1]:
|
|
348
|
+
continue
|
|
349
|
+
if not plausible_title(title):
|
|
157
350
|
continue
|
|
158
351
|
if num in out:
|
|
159
352
|
continue
|
|
@@ -218,6 +411,44 @@ def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
|
|
|
218
411
|
chunks = page_texts[ps - 1 : pe]
|
|
219
412
|
node["rawText"] = "\n".join(chunks).strip()
|
|
220
413
|
|
|
414
|
+
trim_shared_pages(nodes, page_texts)
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def trim_shared_pages(nodes: list[dict], page_texts: list[str]) -> None:
|
|
418
|
+
"""Cut the opening page where several sections start on it.
|
|
419
|
+
|
|
420
|
+
Page granularity is the unit everywhere else, but two sections beginning on
|
|
421
|
+
the same page would otherwise carry byte-identical text — so a summary reads
|
|
422
|
+
like its neighbour's, and the table of contents misleads before anything is
|
|
423
|
+
even opened.
|
|
424
|
+
"""
|
|
425
|
+
by_page: dict[int, list[dict]] = defaultdict(list)
|
|
426
|
+
for node in nodes:
|
|
427
|
+
by_page[node["pageStart"]].append(node)
|
|
428
|
+
|
|
429
|
+
for page, group in by_page.items():
|
|
430
|
+
if len(group) < 2:
|
|
431
|
+
continue
|
|
432
|
+
text = page_texts[page - 1]
|
|
433
|
+
# Where each section's heading sits on that page, in reading order.
|
|
434
|
+
marks: list[tuple[int, dict]] = []
|
|
435
|
+
for node in group:
|
|
436
|
+
title = node["title"].split(" — ")[-1].strip()
|
|
437
|
+
at = text.find(title) if len(title) >= 4 else -1
|
|
438
|
+
if at == -1:
|
|
439
|
+
at = text.find(f"{node['path']} ")
|
|
440
|
+
marks.append((at, node))
|
|
441
|
+
if any(at == -1 for at, _ in marks):
|
|
442
|
+
continue # a heading we cannot locate: leave the whole page in place
|
|
443
|
+
marks.sort(key=lambda m: m[0])
|
|
444
|
+
for i, (at, node) in enumerate(marks):
|
|
445
|
+
end = marks[i + 1][0] if i + 1 < len(marks) else len(text)
|
|
446
|
+
head = text[at:end].strip()
|
|
447
|
+
if not head:
|
|
448
|
+
continue
|
|
449
|
+
rest = "\n".join(page_texts[node["pageStart"] : node["pageEnd"]]).strip()
|
|
450
|
+
node["rawText"] = f"{head}\n{rest}".strip() if rest else head
|
|
451
|
+
|
|
221
452
|
|
|
222
453
|
def extract_figures(
|
|
223
454
|
doc: "fitz.Document",
|
|
@@ -295,7 +526,8 @@ def main() -> None:
|
|
|
295
526
|
running = detect_running_text(doc)
|
|
296
527
|
chapters = extract_chapters(doc, toc_pages)
|
|
297
528
|
chap_prefixes = {k for k in chapters if k.isdigit()}
|
|
298
|
-
|
|
529
|
+
spans = chapter_page_spans(chapters, doc.page_count)
|
|
530
|
+
subsections = extract_subsections(doc, chap_prefixes, toc_pages, running, spans)
|
|
299
531
|
nodes = build_tree(chapters, subsections, doc.page_count)
|
|
300
532
|
extract_section_text(doc, nodes)
|
|
301
533
|
figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)
|