@stratta/mcp 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -3
- package/dist/confirm.d.ts +15 -0
- package/dist/confirm.js +54 -0
- package/dist/errors.js +27 -0
- package/dist/index.js +27 -2
- package/dist/prompts.d.ts +4 -0
- package/dist/prompts.js +40 -0
- package/dist/resources.d.ts +3 -0
- package/dist/resources.js +35 -0
- package/dist/tools/catalog.gen.d.ts +34 -0
- package/dist/tools/catalog.gen.js +724 -0
- package/dist/tools/dossier.d.ts +48 -0
- package/dist/tools/dossier.js +249 -127
- package/dist/tools/ingest.d.ts +8 -0
- package/dist/tools/ingest.js +116 -98
- package/dist/tools/read.d.ts +42 -17
- package/dist/tools/read.js +160 -84
- package/dist/usage.d.ts +3 -0
- package/dist/usage.js +50 -0
- package/package.json +1 -1
- package/scripts/ingest-prepass.py +499 -54
- package/scripts/tests/test_prepass.py +211 -0
- package/skills/consult-stratta/SKILL.md +73 -0
- package/skills/ingest-norm/SKILL.md +99 -44
- package/skills/verification-note/SKILL.md +54 -0
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
"""Tests of the ingest pre-pass on PDFs built here, with PyMuPDF.
|
|
2
|
+
|
|
3
|
+
No licensed norm can sit in the repository, so each test writes the PDF it
|
|
4
|
+
needs: a native SIA layout (bookmarks, uppercase chapters, X.Y and X.Y.Z
|
|
5
|
+
headings, numbered clauses, a running header, page numbers, a word hyphenated
|
|
6
|
+
at a line end), a CEN layout (no bookmarks, sentence-case chapters, the number
|
|
7
|
+
on its own line above the title), and a table whose rows look like headings.
|
|
8
|
+
|
|
9
|
+
Run: python -m pytest packages/mcp/scripts/tests
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import importlib.util
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
import fitz
|
|
19
|
+
import pytest
|
|
20
|
+
|
|
21
|
+
HERE = Path(__file__).resolve().parent
|
|
22
|
+
SCRIPT = HERE.parent / "ingest-prepass.py"
|
|
23
|
+
|
|
24
|
+
spec = importlib.util.spec_from_file_location("prepass", SCRIPT)
|
|
25
|
+
prepass = importlib.util.module_from_spec(spec)
|
|
26
|
+
sys.modules["prepass"] = prepass
|
|
27
|
+
spec.loader.exec_module(prepass) # type: ignore[union-attr]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def write_pdf(path: Path, pages: list[list[str]], toc: list[list] | None = None) -> None:
|
|
31
|
+
"""One text line per list entry, 16 pt apart, top to bottom."""
|
|
32
|
+
doc = fitz.open()
|
|
33
|
+
for lines in pages:
|
|
34
|
+
page = doc.new_page(width=595, height=842)
|
|
35
|
+
y = 60
|
|
36
|
+
for line in lines:
|
|
37
|
+
page.insert_text((56, y), line, fontsize=10, fontname="helv")
|
|
38
|
+
y += 16
|
|
39
|
+
if toc:
|
|
40
|
+
doc.set_toc(toc)
|
|
41
|
+
doc.save(str(path))
|
|
42
|
+
doc.close()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def run(pdf: Path, out: Path) -> dict:
|
|
46
|
+
"""The pre-pass as `main()` runs it, without the CLI."""
|
|
47
|
+
doc = fitz.open(str(pdf))
|
|
48
|
+
toc_pages = prepass.detect_toc_pages(doc)
|
|
49
|
+
running = prepass.detect_running_text(doc)
|
|
50
|
+
chapters = prepass.extract_chapters(doc, toc_pages)
|
|
51
|
+
prefixes = {k for k in chapters if k.isdigit()}
|
|
52
|
+
spans = prepass.chapter_page_spans(chapters, doc.page_count)
|
|
53
|
+
subs = prepass.extract_subsections(doc, prefixes, toc_pages, running, spans)
|
|
54
|
+
subs, pruned = prepass.prune_subsections(subs, chapters)
|
|
55
|
+
nodes = prepass.build_tree(chapters, subs, doc.page_count)
|
|
56
|
+
warnings = [f"pruned: {p}" for p in pruned]
|
|
57
|
+
warnings += prepass.extract_section_text(doc, nodes, running)
|
|
58
|
+
return {"nodes": nodes, "warnings": warnings, "chapters": chapters}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
HEADER = "SIA 267, Copyright 2013 by SIA Zurich"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def sia_like(tmp_path: Path) -> Path:
|
|
65
|
+
pdf = tmp_path / "sia.pdf"
|
|
66
|
+
body = "Le terrain de fondation est caracterise par des essais."
|
|
67
|
+
pages = [
|
|
68
|
+
[HEADER, "1 DOMAINE D'APPLICATION", "1.1 Objet", "1.1.1 La presente norme traite des fondations.", "1.1.2 Elle s'applique aux ouvrages neufs.", "3"],
|
|
69
|
+
[HEADER, "2 TERMINOLOGIE", "2.1 Termes techniques", "2.1.1 Pieu : element de fondation elance.", "4"],
|
|
70
|
+
[HEADER, "9 FONDATIONS SUR PIEUX", "9.1 Delimitation", "9.1.1 Ce chapitre traite des pieux fores et battus.", "9.5 Dimensionnement", "9.5.1 Generalites", "9.5.1.1 " + body, "5"],
|
|
71
|
+
[HEADER, "9.5.2 Resistance ultime du pieu isole", "9.5.2.1 La resistance ultime se compose de la resistance de pointe et du frot-", "tement lateral, voir chiffre 9.5.3.", "9.5.3 Frottement lateral", "9.5.3.1 Le frottement lateral se determine par essais.", "6"],
|
|
72
|
+
[HEADER, "9.6 Dispositions d'execution", "9.6.1 Les pieux sont executes selon la norme SIA 267/1.", "7"],
|
|
73
|
+
[HEADER, "Annexe A (normative)", "A.1 Valeurs indicatives", "Le tableau donne des valeurs indicatives.", "8"],
|
|
74
|
+
]
|
|
75
|
+
toc = [
|
|
76
|
+
[1, "1 DOMAINE D'APPLICATION", 1],
|
|
77
|
+
[1, "2 TERMINOLOGIE", 2],
|
|
78
|
+
[1, "9 FONDATIONS SUR PIEUX", 3],
|
|
79
|
+
[1, "Annexe A (normative)", 6],
|
|
80
|
+
]
|
|
81
|
+
write_pdf(pdf, pages, toc)
|
|
82
|
+
return pdf
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def test_tree_descends_to_titled_clauses(tmp_path):
|
|
86
|
+
result = run(sia_like(tmp_path), tmp_path)
|
|
87
|
+
paths = [n["path"] for n in result["nodes"]]
|
|
88
|
+
assert paths[:3] == ["1", "1.1", "2"]
|
|
89
|
+
assert "9.5" in paths and "9.5.1" in paths and "9.5.2" in paths and "9.5.3" in paths
|
|
90
|
+
assert "9.6" in paths and "Annexe A" in paths
|
|
91
|
+
# Numbered paragraphs are clauses inside their heading, not nodes.
|
|
92
|
+
assert "9.5.2.1" not in paths and "1.1.1" not in paths
|
|
93
|
+
by_path = {n["path"]: n for n in result["nodes"]}
|
|
94
|
+
assert by_path["9.5.2"]["depth"] == 2
|
|
95
|
+
assert by_path["9.5.2"]["parentNodeId"] == by_path["9.5"]["nodeId"]
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_each_node_carries_only_its_own_text(tmp_path):
|
|
99
|
+
result = run(sia_like(tmp_path), tmp_path)
|
|
100
|
+
by_path = {n["path"]: n["rawText"] for n in result["nodes"]}
|
|
101
|
+
# The chapter holds its intro, not the text of every section under it.
|
|
102
|
+
assert "9.5.2.1" not in by_path["9"]
|
|
103
|
+
assert "Delimitation" not in by_path["9"] or by_path["9"].count("\n") < 3
|
|
104
|
+
# A section holds its clauses up to the next heading, whatever its depth.
|
|
105
|
+
assert "9.5.2.1 La resistance ultime" in by_path["9.5.2"]
|
|
106
|
+
assert "9.5.3.1" not in by_path["9.5.2"]
|
|
107
|
+
assert "9.5.3.1 Le frottement lateral" in by_path["9.5.3"]
|
|
108
|
+
# 9.5.1 gets its clause, 9.5 only what is between its heading and 9.5.1.
|
|
109
|
+
assert "9.5.1.1" in by_path["9.5.1"]
|
|
110
|
+
assert "9.5.1.1" not in by_path["9.5"]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_pages_are_cleaned(tmp_path):
|
|
114
|
+
result = run(sia_like(tmp_path), tmp_path)
|
|
115
|
+
for node in result["nodes"]:
|
|
116
|
+
assert HEADER not in node["rawText"], node["path"]
|
|
117
|
+
# Bare page numbers are gone; the clause numbers are joined to their text.
|
|
118
|
+
assert not any(line.strip().isdigit() for line in node["rawText"].splitlines()), node["path"]
|
|
119
|
+
text = {n["path"]: n["rawText"] for n in result["nodes"]}["9.5.2"]
|
|
120
|
+
# Hyphenation at the line end is resolved by PyMuPDF.
|
|
121
|
+
assert "frottement lateral" in text
|
|
122
|
+
assert "frot-" not in text
|
|
123
|
+
assert result["warnings"] == []
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def cen_like(tmp_path: Path) -> Path:
|
|
127
|
+
"""No bookmarks; the number sits on its own line above a sentence-case title."""
|
|
128
|
+
pdf = tmp_path / "cen.pdf"
|
|
129
|
+
pages = [
|
|
130
|
+
["1", "Domaine d'application", "Le present document specifie les exigences.", "2", "References normatives", "EN 1997-1, Calcul geotechnique."],
|
|
131
|
+
["3", "Termes et definitions", "3.1", "Ancrage", "Element transmettant une force de traction au terrain.", "3.2", "Tirant", "Partie libre de l'ancrage."],
|
|
132
|
+
["4", "Execution", "4.1", "Forage", "Le forage est realise sans deblaiement."],
|
|
133
|
+
]
|
|
134
|
+
write_pdf(pdf, pages)
|
|
135
|
+
return pdf
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_cen_layout_without_bookmarks(tmp_path):
|
|
139
|
+
result = run(cen_like(tmp_path), tmp_path)
|
|
140
|
+
paths = [n["path"] for n in result["nodes"]]
|
|
141
|
+
assert paths[:2] == ["1", "2"]
|
|
142
|
+
assert "3" in paths and "3.1" in paths and "3.2" in paths and "4.1" in paths
|
|
143
|
+
by_path = {n["path"]: n["rawText"] for n in result["nodes"]}
|
|
144
|
+
assert "Element transmettant" in by_path["3.1"]
|
|
145
|
+
assert "Partie libre" not in by_path["3.1"]
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def table_like(tmp_path: Path) -> Path:
|
|
149
|
+
"""A chapter whose table rows look like numbered headings."""
|
|
150
|
+
pdf = tmp_path / "table.pdf"
|
|
151
|
+
pages = [
|
|
152
|
+
["1 GENERALITES", "1.1 Objet", "La presente norme fixe les prestations.", "2 PRESTATIONS", "2.1 Etudes preliminaires", "Tableau 1 : phases", "2.3 retrait des obstacles", "2.2 Avant-projet", "Description de la phase."],
|
|
153
|
+
["3 HONORAIRES", "3.1 Calcul", "Le calcul se fait au temps employe."],
|
|
154
|
+
]
|
|
155
|
+
write_pdf(pdf, pages)
|
|
156
|
+
return pdf
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def test_table_rows_are_not_headings(tmp_path):
|
|
160
|
+
result = run(table_like(tmp_path), tmp_path)
|
|
161
|
+
paths = [n["path"] for n in result["nodes"]]
|
|
162
|
+
# A lowercase numbered line is a table row.
|
|
163
|
+
assert "2.3" not in paths
|
|
164
|
+
assert "2.1" in paths and "2.2" in paths and "3.1" in paths
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def test_prune_subsections_drops_out_of_sequence_child():
|
|
168
|
+
chapters = {"4": {"title": "X", "pageStart": 10}, "9": {"title": "Y", "pageStart": 60}}
|
|
169
|
+
subs = {
|
|
170
|
+
"4.1": ("Premiere", 11),
|
|
171
|
+
"4.2": ("Deuxieme", 63), # found inside chapter 9: a table row
|
|
172
|
+
"4.3": ("Troisieme", 14),
|
|
173
|
+
"9.1": ("Une", 61),
|
|
174
|
+
}
|
|
175
|
+
kept, dropped = prepass.prune_subsections(subs, chapters)
|
|
176
|
+
assert set(kept) == {"4.1", "4.3", "9.1"}
|
|
177
|
+
assert len(dropped) == 1 and dropped[0].startswith("4.2")
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def test_join_clause_numbers_and_page_cleaning():
|
|
181
|
+
joined = prepass.join_clause_numbers("9.5.1.1\nLes pieux sont fores.\n9.5.1.2\nIls sont battus.")
|
|
182
|
+
assert joined == "9.5.1.1 Les pieux sont fores.\n9.5.1.2 Ils sont battus."
|
|
183
|
+
cleaned = prepass.clean_page("SIA 267, Copyright\nTexte\n63", {"SIA 267, Copyright"}, 110)
|
|
184
|
+
assert cleaned == "Texte"
|
|
185
|
+
# A number in the middle of a long page is content, not a page number.
|
|
186
|
+
middle = "\n".join(["a", "b", "c", "d", "42", "e", "f", "g", "h"])
|
|
187
|
+
assert "42" in prepass.clean_page(middle, set(), 110)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def test_recovers_chapter_numbers_by_position(tmp_path):
|
|
191
|
+
"""An OCR'd copy loses the numbering column on a run of pages: the chapter
|
|
192
|
+
titles survive without their number, and so do their sub-headings."""
|
|
193
|
+
pdf = tmp_path / "ocr.pdf"
|
|
194
|
+
body = "Le terrain de fondation est caracterise par des essais."
|
|
195
|
+
pages = [
|
|
196
|
+
["DOMAINE D'APPLICATION", "Delimitation", "La norme traite des fondations.", "1"],
|
|
197
|
+
["1 TERMINOLOGIE", "1.1 Termes techniques", "Pieu : element de fondation elance.", "2"],
|
|
198
|
+
["2 PRINCIPES", "2.1 Generalites", body, "3"],
|
|
199
|
+
["TERRAIN DE FONDATION", "Generalites", body, "4"],
|
|
200
|
+
["ANALYSE STRUCTURALE", "Generalites", body, "5"],
|
|
201
|
+
["5 DIMENSIONNEMENT", "5.1 Generalites", body, "6"],
|
|
202
|
+
]
|
|
203
|
+
write_pdf(pdf, pages)
|
|
204
|
+
chapters = run(pdf, tmp_path)["chapters"]
|
|
205
|
+
assert chapters["3"]["title"] == "TERRAIN DE FONDATION"
|
|
206
|
+
assert chapters["3"]["pageStart"] == 4
|
|
207
|
+
assert chapters["4"]["title"] == "ANALYSE STRUCTURALE"
|
|
208
|
+
assert chapters["4"]["pageStart"] == 5
|
|
209
|
+
assert chapters["0"]["pageStart"] == 1
|
|
210
|
+
assert prepass.missing_chapter_numbers(chapters) == []
|
|
211
|
+
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: consult-stratta
|
|
3
|
+
description: Use Stratta to answer civil engineering questions from licensed SIA, Eurocode, or office norms with exact citations, cross-reference checks, figures, and optional project dossiers. Trigger phrases: "selon la norme", "d'apres la SIA", "que dit l'Eurocode", "verifie dans les normes", "reprends le dossier".
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Consult Stratta
|
|
7
|
+
|
|
8
|
+
Use this skill when the user asks a technical civil engineering question that should be answered from the norms available in Stratta, or when the user asks to resume, create or update a Stratta project dossier.
|
|
9
|
+
|
|
10
|
+
You answer practising engineers from the norms their organization holds in Stratta, and only from them. The corpus below is the whole of what you can read: a norm absent from it cannot be consulted, and a question outside it is answered by saying so, not by general knowledge. Reply in the language the user writes in; quote the norm in the language it is written in and translate only when asked.
|
|
11
|
+
|
|
12
|
+
## First call
|
|
13
|
+
|
|
14
|
+
Call `get_methodology` before the first technical answer of a conversation: it returns this method together with the table of the norms the connected organization can read, their scope and coverage, and the dependencies among them. Follow it for the rest of the conversation. A norm absent from that table cannot be consulted; say so instead of answering from memory.
|
|
15
|
+
|
|
16
|
+
## Method
|
|
17
|
+
|
|
18
|
+
1. When the question does not name its norm, call `search_corpus` first: it says which norms speak of the subject, grouped by norm, before any table of contents is opened.
|
|
19
|
+
2. `get_toc(code, maxDepth: 1)` on the norm that governs, then `get_subtree(code, chapter, maxDepth: 2)` on the chapter that fits: read summaries, choose the section.
|
|
20
|
+
3. `get_section(code, path)` on the section retained. It comes as a document with its formulas and tables in place, its page range, the sections below it and the references it makes. Descend by `children` rather than loading a chapter whole; when `truncated` is true, call again with `offset`.
|
|
21
|
+
4. Follow the references marked resolvable: the rule, the coefficient and the action are often spread over separate clauses or norms, and the answer is in the one you have not read yet.
|
|
22
|
+
5. `get_figure` when a section points at a figure the reasoning depends on: a load diagram, a design chart, a standard cross-section.
|
|
23
|
+
6. `search_in_norm` inside one norm when its table of contents suggests no obvious section. Search in the language of the norm; accents and case do not matter; a clause number as keyword returns that clause.
|
|
24
|
+
|
|
25
|
+
When the question names a norm absent from the corpus but a norm you hold refers to it, read that clause first: the pointer (code and clause number) usually lives there, and it is an answer.
|
|
26
|
+
|
|
27
|
+
## Answer contract
|
|
28
|
+
|
|
29
|
+
- Every technical statement rests on a section you read in this conversation and cites it as `[CODE § path, p. N]`, for example `[SIA 267 § 9.5.2.1, p. 63]`. A value with no clause is not an answer.
|
|
30
|
+
- A calculation shows its inputs, each with its citation, its formula in LaTeX (`$$...$$`) with every symbol defined and its unit (kN, kPa, MPa, m).
|
|
31
|
+
- Before answering, check: each value has its clause; each clause was read, not remembered from a summary; the resolvable references were followed; what the corpus does not cover is named, with the norm or clause where it is likely to be found.
|
|
32
|
+
- When several norms are cited, group by norm under headings: principle, then method, then conditions of use.
|
|
33
|
+
- Once per conversation, say that the answer is indicative and that the full text of the norm must be consulted before use on a project.
|
|
34
|
+
|
|
35
|
+
## Budget
|
|
36
|
+
|
|
37
|
+
No `get_toc` without `maxDepth`. No section loaded whole when its children would do. Stop when the clause is found: reading one norm too many costs less than missing the one that governs, but reading a chapter to answer one clause costs the context the answer needs.
|
|
38
|
+
|
|
39
|
+
## Demonstration document
|
|
40
|
+
|
|
41
|
+
DEMO 001 is a fictional document shipped with Stratta so that navigation and citation can be exercised before a licensed norm is ingested. Answer a question about it exactly as you would for a norm: read the section, cite it as [DEMO 001 §path, p. N], give the value. Then say once, plainly, that the document is fictional and that none of its values may be used in a project: the office must ingest the norm it holds a licence for. Never refuse to read it, and never present its values as those of a real norm.
|
|
42
|
+
|
|
43
|
+
## Dossier
|
|
44
|
+
|
|
45
|
+
A conversation disappears, a project lasts months. The dossier is what remains: what was retained, what it rests on, and what is still open. It is the trail an engineer hands to a reviewer or an insurer.
|
|
46
|
+
|
|
47
|
+
**Open one** (`open_dossier`) when the user names a project, a structure, a site or a mandate, when a value is retained for a design rather than looked up, when a hypothesis is set, when a site observation is reported, or when a question is left open. Do not open a dossier for a one-off lookup with no project behind it.
|
|
48
|
+
|
|
49
|
+
**Reload it first** (`list_dossiers`, `load_dossier`) BEFORE answering, as soon as the user returns to a project already named ("pick up dossier X", "where were we on Y"): reloading is what stops a decision already taken from being taken again. Read its entries and its open questions before answering anything about that project.
|
|
50
|
+
|
|
51
|
+
**Read its attachments.** A dossier carries the project's own documents (site reports, borehole logs, meeting minutes): `list_attachments`, then `read_attachment` on what bears on the question, or `search_in_dossier` when the dossier holds many. A fact from a piece is evidence like a clause: cite it as [name, p. N] and file it with `attachmentId` and `attachmentPage`.
|
|
52
|
+
|
|
53
|
+
**Start from a template when one fits.** The office may have written checklists for its types of structure (`list_templates`); on a new project that matches one, `apply_template` opens the questions with the clauses they rest on, and says which clauses the corpus lacks.
|
|
54
|
+
|
|
55
|
+
**Shape.** A dossier is a set of questions, each with its evidence, its options and eventually its decision. The question is the unit of work ("can pile P38 be kept?", "which friction angle do we retain?"). Open one with `open_question` BEFORE gathering evidence for it, one question per thing to settle, and name the options on the table when there are several.
|
|
56
|
+
|
|
57
|
+
**What to write** (`save_finding`, one clause, one value or one fact per call, in one or two sentences, filed under its question with `questionId`):
|
|
58
|
+
- reference: what the norm says, with `normCode`, `sectionPath`, `page`, and `normEdition` taken from the header of the section read;
|
|
59
|
+
- hypothesis: the retained value AND the reasoning; in geotechnics a retained value is a judgement, not the output of a calculation;
|
|
60
|
+
- observation: what the site showed, with the date when known;
|
|
61
|
+
- decision (`record_decision`): what was decided, why, the clause it rests on and the retained option, ONLY after the user confirmed it.
|
|
62
|
+
|
|
63
|
+
**What not to write:** your own prose, the intermediate steps, anything the user did not treat as a decision. Set `confidence` whenever a value is at stake: established, judgement or to_confirm. The dossier documents the engineer's reasoning; it does not replace it: never conclude that a structure complies, never write a hypothesis the user has not confirmed.
|
|
64
|
+
|
|
65
|
+
## Boundaries
|
|
66
|
+
|
|
67
|
+
- Do not present Stratta output as final engineering approval, and never state that a structure complies.
|
|
68
|
+
- Do not offer, suggest or link to a plan change. When a limit is reached (`QUOTA_EXCEEDED`), report the limit and stop.
|
|
69
|
+
- Do not answer from norms the organization has not licensed and ingested.
|
|
70
|
+
- Do not expose internal identifiers unless the next Stratta tool call requires them.
|
|
71
|
+
- Ingesting a new norm starts from a PDF on the user's own machine: it goes through the local `@stratta/mcp` server and its `ingest-norm` skill, never through the remote connector.
|
|
72
|
+
|
|
73
|
+
methodology: v2
|
|
@@ -78,56 +78,92 @@ Output under `.stratta-ingest/<code-slug>/`:
|
|
|
78
78
|
`prepass.json` structure:
|
|
79
79
|
|
|
80
80
|
- `doc` — `pageCount`, detected `language`, `tocSource`, raw `metadata`.
|
|
81
|
-
- `stats` — `sectionCount`, `byDepth`, `figureCount
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
81
|
+
- `stats` — `sectionCount`, `byDepth`, `figureCount`, `charsPerPage`,
|
|
82
|
+
`emptyLeaves`. `charsPerPage` under 300 means the text layer is missing
|
|
83
|
+
(a scan without OCR): stop and tell the user.
|
|
84
|
+
- `warnings[]` — what the pre-pass could not decide on its own: chapters
|
|
85
|
+
missing from the sequence (an OCR'd copy often loses the big chapter
|
|
86
|
+
number; read them off the contents page and add them), sub-sections dropped
|
|
87
|
+
as out of sequence, headings it could not locate, leaves with no text. An
|
|
88
|
+
empty list is the normal case on a native SIA PDF. **Read it before step 5**
|
|
89
|
+
and look at the pages it names.
|
|
90
|
+
- `sections[]` — full hierarchical tree (depth **0 = chapter**, 1 = X.Y,
|
|
91
|
+
2 = X.Y.Z, 3 = X.Y.Z.W), each with `nodeId`, `parentNodeId`, `path`
|
|
92
|
+
(`"4.2.1"` or `"Annexe B"`), `title`, `depth`, `pageStart`, `pageEnd`,
|
|
93
|
+
`orderIndex`, `rawText` and `textSource`. `rawText` is **the text between
|
|
94
|
+
this heading and the next one**, on cleaned pages: no running header, no
|
|
95
|
+
page number, hyphenation resolved, each numbered clause (`9.5.1.1 Les
|
|
96
|
+
pieux…`) on its own line. A chapter whose text lives in its sections has an
|
|
97
|
+
empty `rawText`; that is normal.
|
|
86
98
|
- `figures[]` — one entry per `Figure N` caption: `figureNumber`, `caption`,
|
|
87
99
|
`page`, `fileName`, `mimeType`.
|
|
88
100
|
|
|
89
|
-
|
|
90
|
-
|
|
101
|
+
Every titled heading is a node (typically 250 to 350 on a SIA norm); the
|
|
102
|
+
numbered clauses stay inside their heading's text.
|
|
91
103
|
|
|
92
104
|
### 4. Create the document
|
|
93
105
|
|
|
94
106
|
Read `prepass.json`, then:
|
|
95
|
-
`ingest_create_document { code, year, title, language: <doc.language>, totalPages: <doc.pageCount
|
|
107
|
+
`ingest_create_document { code, year, title, language: <doc.language>, totalPages: <doc.pageCount>, scope }`
|
|
108
|
+
where `scope` is one sentence on what the norm covers (from its first chapter);
|
|
109
|
+
it is shown to every agent in the served methodology, so write it as a routing
|
|
110
|
+
hint: "Geotechnical design: foundations, piles, anchors, retaining structures".
|
|
96
111
|
→ returns `documentId`. Keep it for every subsequent call.
|
|
97
112
|
|
|
98
|
-
### 5.
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
113
|
+
### 5. Validate the tree, write the summaries
|
|
114
|
+
|
|
115
|
+
**Keep every node of the pre-pass.** A clause (`9.5.2.1`) is an address an
|
|
116
|
+
engineer cites and an agent is sent to; folding it into `9.5.2` made the
|
|
117
|
+
citation impossible and the section too long to read. Do not aggregate,
|
|
118
|
+
do not drop leaves. A typical SIA norm gives 250 to 800 nodes; the quota
|
|
119
|
+
allows 1 000 per norm.
|
|
120
|
+
|
|
121
|
+
What the agentic pass is for:
|
|
122
|
+
|
|
123
|
+
- **Validate the tree**, not rewrite it. Read `warnings[]` and the pages it
|
|
124
|
+
names; fix a wrong title, a missing chapter number, a wrong page range.
|
|
125
|
+
Keep `nodeId` / `parentNodeId` / `path` / `depth` / `pageStart` /
|
|
126
|
+
`pageEnd` / `orderIndex` from the pre-pass otherwise.
|
|
127
|
+
- **`content` = the pre-pass `rawText`, verbatim.** It is the text of the
|
|
128
|
+
norm between this heading and the next, already cleaned (no running
|
|
129
|
+
header, no page number, hyphenation resolved, one clause per line). Do
|
|
130
|
+
not paraphrase it, do not summarise it, do not put LaTeX inline: the
|
|
131
|
+
agent that reads it later must read the norm, not a rewrite. Formulas and
|
|
132
|
+
tables are attached separately (step 7) and rendered at their anchor.
|
|
133
|
+
- **`rawContent`** = the same `rawText`.
|
|
134
|
+
- **`summary`**: 1 to 3 sentences on **chapters and X.Y nodes only**, stating
|
|
135
|
+
what the section decides and the values it holds (this is what
|
|
136
|
+
`get_toc` shows, so it drives navigation). Leave `summary` empty (`""`)
|
|
137
|
+
on deeper nodes: they inherit their nearest ancestor's at read time.
|
|
138
|
+
- Note the norm's **scope** in one sentence (from chapter 1) for the
|
|
139
|
+
document record; it will feed the methodology.
|
|
118
140
|
|
|
119
141
|
### 6. Insert sections (batched)
|
|
120
142
|
|
|
121
|
-
`ingest_create_sections { documentId, sections: [...] }` in batches of
|
|
122
|
-
Parent links resolve via `parentNodeId` within
|
|
123
|
-
|
|
124
|
-
|
|
143
|
+
`ingest_create_sections { documentId, sections: [...] }` in batches of **150**
|
|
144
|
+
(the tool accepts up to 200). Parent links resolve via `parentNodeId` within
|
|
145
|
+
the batch and across prior batches, so send the nodes in document order. The
|
|
146
|
+
call returns a `nodeId → sectionId` map; **use those `sectionId`s** for every
|
|
147
|
+
enrichment call below. A batch is idempotent on `nodeId`: retrying it after
|
|
148
|
+
a network error does not duplicate anything.
|
|
125
149
|
|
|
126
150
|
### 7. Enrich
|
|
127
151
|
|
|
152
|
+
Open the PDF visually for the pages the pre-pass flagged with a formula or
|
|
153
|
+
a table (PyMuPDF mangles both), and attach what you read by its number, so
|
|
154
|
+
that the read tools can place it at its anchor in the text:
|
|
155
|
+
|
|
128
156
|
- `ingest_attach_formula { sectionId, latex, description, formulaNumber }`
|
|
157
|
+
where `formulaNumber` is the number printed in the margin, e.g. `"(9.12)"`.
|
|
129
158
|
- `ingest_attach_table { sectionId, data: { headers, rows }, caption, tableNumber }`
|
|
159
|
+
where `tableNumber` is the printed label, e.g. `"Tableau 3"`.
|
|
130
160
|
- `ingest_attach_cross_ref { sourceSectionId, targetDocumentCode, targetSectionPath?, refText, refType }`
|
|
161
|
+
only for references the automatic pass (step 9) cannot read, such as a
|
|
162
|
+
reference written in prose without a clause number. Manual rows survive a
|
|
163
|
+
later `ingest_normalize_cross_refs`.
|
|
164
|
+
|
|
165
|
+
Attach a formula or a table to the **deepest** node whose text names it.
|
|
166
|
+
Both calls are idempotent on their number: re-attaching updates the row.
|
|
131
167
|
|
|
132
168
|
### 8. Upload figures
|
|
133
169
|
|
|
@@ -139,24 +175,43 @@ For each figure in `prepass.json#figures`:
|
|
|
139
175
|
- `ingest_upload_figure { sectionId, base64, mimeType: "image/png", caption, figureNumber }`
|
|
140
176
|
(≤ 8 MB per image).
|
|
141
177
|
|
|
142
|
-
### 9. Auto cross-references (
|
|
178
|
+
### 9. Auto cross-references (always)
|
|
143
179
|
|
|
144
|
-
`ingest_normalize_cross_refs { documentId }`
|
|
145
|
-
|
|
146
|
-
|
|
180
|
+
`ingest_normalize_cross_refs { documentId }` reads every node's text for what
|
|
181
|
+
it points at: other clauses, annexes, figures and tables of the same norm
|
|
182
|
+
(FR, DE, IT, EN wordings) and other norms (SIA / SN EN / EN / ISO / DIN …),
|
|
183
|
+
and resolves each internal target against the tree. Idempotent; manual
|
|
184
|
+
attachments are kept. Its result tells how many internal references
|
|
185
|
+
resolved: on a SIA norm expect most of them to.
|
|
147
186
|
|
|
148
187
|
### 10. Publish
|
|
149
188
|
|
|
150
|
-
`ingest_publish { documentId }`. The
|
|
151
|
-
|
|
189
|
+
`ingest_publish { documentId }`. The server scores the document first (text per
|
|
190
|
+
page, pages under a chapter, empty leaves, attached formulas and tables) and
|
|
191
|
+
**refuses below 30/100** with the reasons. A refusal means the ingestion is
|
|
192
|
+
wrong, almost always because `content` holds summaries instead of the text of
|
|
193
|
+
the norm: go back to step 5, do not retry and do not pass `force`. `force: true`
|
|
194
|
+
is for a document that really is that short, and only after the user said so.
|
|
195
|
+
|
|
196
|
+
Show the returned `quality` (score and flags) to the user. The norm is now
|
|
197
|
+
queryable in your workspace via `list_norms`, `get_toc`, `get_section`,
|
|
198
|
+
`search_in_norm`, `get_figure`, etc.
|
|
152
199
|
|
|
153
200
|
## Quality bar
|
|
154
201
|
|
|
155
|
-
- Trust the pre-pass for `path
|
|
156
|
-
and
|
|
157
|
-
|
|
158
|
-
-
|
|
202
|
+
- Trust the pre-pass for the tree and the text: `path`, `pageStart`/`pageEnd`
|
|
203
|
+
and `rawText` are deterministic. Correct what `warnings[]` points at, and
|
|
204
|
+
nothing else.
|
|
205
|
+
- Every node is kept. The read tools address clauses; a folded clause cannot
|
|
206
|
+
be cited.
|
|
207
|
+
- `content` is the norm's text, never a rewrite. Values live in the text and
|
|
208
|
+
in the attached tables and formulas, never in a summary.
|
|
209
|
+
- Summaries on chapters and X.Y drive navigation: name what the section
|
|
210
|
+
decides and the values it carries.
|
|
159
211
|
- If PyMuPDF mangled a formula (Greek letters, fractions, exponents broken),
|
|
160
|
-
re-read the relevant PDF page visually and
|
|
161
|
-
|
|
212
|
+
re-read the relevant PDF page visually and attach proper LaTeX under its
|
|
213
|
+
printed number.
|
|
214
|
+
- Never skip annexes: they hold key numeric values (zones, coefficients,
|
|
162
215
|
characteristic loads).
|
|
216
|
+
- Read the score `ingest_publish` returns and show it to the user, flags
|
|
217
|
+
included.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: verification-note
|
|
3
|
+
description: Draft a verification note from a Stratta project dossier, one section per question, every paragraph resting on a dossier entry and every gap written as a gap. Trigger phrases: "redige la note de verification", "note de verification", "prepare le rapport du dossier", "verification note".
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Verification note
|
|
7
|
+
|
|
8
|
+
Use this skill when the user asks for the verification note, the review report or the
|
|
9
|
+
summary of a Stratta project dossier. The note is what an engineer hands to a reviewer or
|
|
10
|
+
an insurer: it says what was retained, what it rests on, and what is still open. It never
|
|
11
|
+
says more than the dossier does.
|
|
12
|
+
|
|
13
|
+
## What to produce
|
|
14
|
+
|
|
15
|
+
A Markdown document, given back to the user, never written to the dossier. Its shape is
|
|
16
|
+
the one the server prints as PDF (the "Note de vérification" button on the dossier page),
|
|
17
|
+
so the two can be compared line by line:
|
|
18
|
+
|
|
19
|
+
1. A title with the dossier name and reference, the organization, the date and the
|
|
20
|
+
dossier status.
|
|
21
|
+
2. One section per question, open questions first, then decided, then closed. Each
|
|
22
|
+
section has these parts, in this order, in the language of the user:
|
|
23
|
+
- **Contexte**: the question body and the options considered, the retained one marked.
|
|
24
|
+
- **Exigence**: what the norm requires, from the `reference` entries, each with its
|
|
25
|
+
citation `[CODE § path, p. N]` or `[attachment name, p. N]`.
|
|
26
|
+
- **Preuves**: observations and hypotheses, with their confidence when set.
|
|
27
|
+
- **Décision**: the decision entry, with the clause it rests on; or "close sans
|
|
28
|
+
décision"; or a gap.
|
|
29
|
+
- **Reste ouvert**: entries marked `to_confirm`, and what is missing to settle.
|
|
30
|
+
3. A section for the evidence not filed under any question.
|
|
31
|
+
4. An annex listing every clause cited, with the text read through `get_section` at the
|
|
32
|
+
time of drafting, so the reviewer reads the norm and not a paraphrase.
|
|
33
|
+
|
|
34
|
+
## Rules
|
|
35
|
+
|
|
36
|
+
- Load the dossier first: `list_dossiers`, then `load_dossier`. Read its attachments
|
|
37
|
+
(`list_attachments`, `read_attachment`) when a question rests on one.
|
|
38
|
+
- Every paragraph ends with the identifier of the entry it rests on, in the form
|
|
39
|
+
`(#xxxxxx)` using the last six characters of the entry id. A paragraph with no entry
|
|
40
|
+
behind it does not exist in this note.
|
|
41
|
+
- A missing requirement, decision or proof is written as `[À établir : …]` (or the
|
|
42
|
+
equivalent in the user's language), never as a plausible sentence. The note is
|
|
43
|
+
valuable because its gaps are visible.
|
|
44
|
+
- Quote the norm from a section you read in this conversation. Do not complete a clause
|
|
45
|
+
from memory; if the clause is not in the corpus, say so in the gap.
|
|
46
|
+
- Do not conclude that a structure complies, and do not write a decision the dossier
|
|
47
|
+
does not carry.
|
|
48
|
+
- End with the line: "Document préparé par un agent à partir du dossier Stratta, à
|
|
49
|
+
valider et signer par l'ingénieur responsable."
|
|
50
|
+
|
|
51
|
+
## After drafting
|
|
52
|
+
|
|
53
|
+
Tell the user that the server can produce the same note as a PDF, hashed and timestamped
|
|
54
|
+
by a public authority, from the dossier page, and that the two should agree.
|