pyannotators-entityfishing 1.6.97__tar.gz → 1.6.99__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/PKG-INFO +1 -1
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/Taskfile.yml +1 -1
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/src/pyannotators_entityfishing/__init__.py +1 -1
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/.claude-plugin/plugin.json +1 -1
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/resources/help/help_examples.py +52 -4
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/resources/help/plugin-help.py +184 -53
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/plugin-help/SKILL.md +12 -4
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/.claude/skills/ef-mapping/SKILL.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/.gitignore +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/.gitmodules +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/.python-version +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/README.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/pyproject.toml +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/src/pyannotators_entityfishing/candidates.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/src/pyannotators_entityfishing/ef_client.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/src/pyannotators_entityfishing/entityfishing.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/.claude-plugin/marketplace.json +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/HOWTO.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/README.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/resources/Stages.yml +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/resources/Taskfile.yml +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/resources/help/template.en.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/resources/help/template.fr.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/SKILL.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/annotator.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/converter.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/formatter.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/importer.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/processor.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/segmenter.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/skills/new-plugin/references/types.md +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/tools/check-marketplace.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/submodules/python-archetype/vars/pythonPipeline.groovy +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/build_rerank_data.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/check_rerank_endpoint.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/el_jev.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/eval_linking.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/eval_mappings.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/eval_rerank.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/finetune_rerank.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/merge_rerank.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/recalibrate_linking.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/review_linking.html +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/review_linking.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/sherpa_plan_project.py +0 -0
- {pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/wdfilter.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pyannotators-entityfishing
|
|
3
|
-
Version: 1.6.
|
|
3
|
+
Version: 1.6.99
|
|
4
4
|
Summary: Annotator based on entity-fishing
|
|
5
5
|
Project-URL: Homepage, https://github.com/oterrier/pyannotators_entityfishing/
|
|
6
6
|
Author-email: Olivier Terrier <olivier.terrier@kairntech.com>
|
|
@@ -27,7 +27,7 @@ includes:
|
|
|
27
27
|
# les siens qui n'ont de sens que dans Jenkins (init, version, py:publish), et son
|
|
28
28
|
# checkStageDrift() vérifie que cette liste en est un sous-ensemble ORDONNÉ.
|
|
29
29
|
STAGES: >-
|
|
30
|
-
py:sync py:lint py:test
|
|
30
|
+
py:sync py:lint py:test py:help-check
|
|
31
31
|
py:sbom py:check-vulnerabilities
|
|
32
32
|
py:build py:publish
|
|
33
33
|
TEST_STAGE: py:test
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
|
|
3
3
|
"name": "python-archetype",
|
|
4
4
|
"displayName": "Kairntech python-archetype",
|
|
5
|
-
"version": "1.3.
|
|
5
|
+
"version": "1.3.4",
|
|
6
6
|
"description": "Cree et entretient les depots Python de Kairntech qui consomment python-archetype : creation d'un plugin a partir de sa graine, page d'aide Sherpa d'un plugin, et mesure de la derive d'un depot par rapport a elle.",
|
|
7
7
|
"author": {
|
|
8
8
|
"name": "Kairntech",
|
|
@@ -24,6 +24,7 @@ Le dépôt du plugin rend la fixture visible depuis 'tests/conftest.py' :
|
|
|
24
24
|
from help_examples import help_example # noqa: E402, F401
|
|
25
25
|
"""
|
|
26
26
|
|
|
27
|
+
import io
|
|
27
28
|
import json
|
|
28
29
|
from pathlib import Path
|
|
29
30
|
|
|
@@ -36,14 +37,61 @@ EXAMPLES_DIR = Path("build") / "help-examples"
|
|
|
36
37
|
MAX_CONTENT = 100_000
|
|
37
38
|
|
|
38
39
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
40
|
+
# Un classeur se montre par sa première feuille : au-delà, la page n'en citerait qu'un extrait.
|
|
41
|
+
MAX_SHEET_ROWS = 50
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _sheet(data: bytes) -> dict | None:
|
|
45
|
+
"""La première feuille d'un classeur Excel, ou None si ce n'en est pas un (ou sans openpyxl)."""
|
|
46
|
+
if not data.startswith(b"PK"):
|
|
47
|
+
return None
|
|
42
48
|
try:
|
|
43
|
-
|
|
49
|
+
import openpyxl
|
|
50
|
+
|
|
51
|
+
workbook = openpyxl.load_workbook(io.BytesIO(data), read_only=True, data_only=True)
|
|
52
|
+
except Exception: # noqa: BLE001 -- une archive zip qui n'est pas un classeur, ou pas d'openpyxl
|
|
53
|
+
return None
|
|
54
|
+
sheet = workbook.worksheets[0]
|
|
55
|
+
rows = []
|
|
56
|
+
for row in sheet.iter_rows(values_only=True):
|
|
57
|
+
if len(rows) == MAX_SHEET_ROWS:
|
|
58
|
+
break
|
|
59
|
+
rows.append(["" if value is None else str(value) for value in row])
|
|
60
|
+
# Sans les colonnes vides de droite, qu'openpyxl rend quand une ligne est plus longue.
|
|
61
|
+
width = max((max((i + 1 for i, v in enumerate(row) if v), default=0) for row in rows), default=0)
|
|
62
|
+
return {"name": sheet.title, "sheets": len(workbook.worksheets), "rows": [row[:width] for row in rows]}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _text(data: bytes) -> tuple[str, str] | None:
|
|
66
|
+
"""Le texte et son encodage : UTF-8, sinon Latin-1 (cp1252) pour un texte sans octet de contrôle."""
|
|
67
|
+
try:
|
|
68
|
+
return data.decode("utf-8"), "utf-8"
|
|
44
69
|
except UnicodeDecodeError:
|
|
70
|
+
pass
|
|
71
|
+
# Latin-1 décode n'importe quels octets : un octet de contrôle (hors tabulation et fins
|
|
72
|
+
# de ligne) trahit un contenu binaire.
|
|
73
|
+
if any(byte < 0x20 and byte not in b"\t\n\r\f" for byte in data):
|
|
74
|
+
return None
|
|
75
|
+
try:
|
|
76
|
+
return data.decode("cp1252"), "cp1252"
|
|
77
|
+
except UnicodeDecodeError:
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _content(data: bytes) -> dict:
|
|
82
|
+
"""Le contenu d'un fichier ou d'une réponse : son texte, la première feuille d'un classeur, ou sa taille."""
|
|
83
|
+
content = {"size": len(data)}
|
|
84
|
+
sheet = _sheet(data)
|
|
85
|
+
if sheet is not None:
|
|
86
|
+
content["sheet"] = sheet
|
|
87
|
+
return content
|
|
88
|
+
decoded = _text(data)
|
|
89
|
+
if decoded is None:
|
|
45
90
|
return content
|
|
91
|
+
text, encoding = decoded
|
|
46
92
|
content["text"] = text[:MAX_CONTENT]
|
|
93
|
+
if encoding != "utf-8":
|
|
94
|
+
content["encoding"] = encoding
|
|
47
95
|
if len(text) > MAX_CONTENT:
|
|
48
96
|
content["truncated"] = True
|
|
49
97
|
return content
|
|
@@ -142,11 +142,10 @@ TEXTS = {
|
|
|
142
142
|
"sentence": "Phrase",
|
|
143
143
|
"categories": "Catégories",
|
|
144
144
|
"produced_text": "Texte produit :",
|
|
145
|
-
"sentences_count": "{n} phrase(s).",
|
|
146
145
|
"no_sentences": "Aucune phrase.",
|
|
147
146
|
"no_documents": "Aucun document.",
|
|
148
|
-
"no_concepts": "Aucun
|
|
149
|
-
"concepts_count": "{n}
|
|
147
|
+
"no_concepts": "Aucun terme.",
|
|
148
|
+
"concepts_count": "{n} terme(s).",
|
|
150
149
|
"response": "Réponse",
|
|
151
150
|
"file_named": "fichier",
|
|
152
151
|
"binary": "Contenu binaire, {n} octets.",
|
|
@@ -158,6 +157,21 @@ TEXTS = {
|
|
|
158
157
|
"score": "Score",
|
|
159
158
|
"identifier": "Identifiant",
|
|
160
159
|
"alt_forms": "Formes alternatives",
|
|
160
|
+
"no_parameters": "Ce plugin n'a aucun paramètre.",
|
|
161
|
+
"annotations_removed": "Annotations en entrée retirées : toutes ({n}).",
|
|
162
|
+
"categories_removed": "Catégories en entrée retirées : toutes ({n}).",
|
|
163
|
+
"sentences_removed": "Phrases en entrée retirées : toutes ({n}).",
|
|
164
|
+
"metadata_removed": "Métadonnées retirées :",
|
|
165
|
+
"sentences": "Phrases ({n}) :",
|
|
166
|
+
"empty_sentence": "*(phrase vide)*",
|
|
167
|
+
"produced_document": "Document produit",
|
|
168
|
+
"identifier_named": "identifiant",
|
|
169
|
+
"title": "Titre",
|
|
170
|
+
"properties": "Propriétés :",
|
|
171
|
+
"sheet": "Feuille `{name}`",
|
|
172
|
+
"first_sheet": "Feuille `{name}`, la première des {n}",
|
|
173
|
+
"encoding": "encodé en {encoding}",
|
|
174
|
+
"decimal": ",",
|
|
161
175
|
},
|
|
162
176
|
"en": {
|
|
163
177
|
"languages": "**Supported languages:**",
|
|
@@ -189,11 +203,10 @@ TEXTS = {
|
|
|
189
203
|
"sentence": "Sentence",
|
|
190
204
|
"categories": "Categories",
|
|
191
205
|
"produced_text": "Produced text:",
|
|
192
|
-
"sentences_count": "{n} sentence(s).",
|
|
193
206
|
"no_sentences": "No sentences.",
|
|
194
207
|
"no_documents": "No documents.",
|
|
195
|
-
"no_concepts": "No
|
|
196
|
-
"concepts_count": "{n}
|
|
208
|
+
"no_concepts": "No terms.",
|
|
209
|
+
"concepts_count": "{n} term(s).",
|
|
197
210
|
"response": "Response",
|
|
198
211
|
"file_named": "file",
|
|
199
212
|
"binary": "Binary content, {n} bytes.",
|
|
@@ -205,6 +218,21 @@ TEXTS = {
|
|
|
205
218
|
"score": "Score",
|
|
206
219
|
"identifier": "Identifier",
|
|
207
220
|
"alt_forms": "Alternative forms",
|
|
221
|
+
"no_parameters": "This plugin has no parameters.",
|
|
222
|
+
"annotations_removed": "Input annotations removed: all of them ({n}).",
|
|
223
|
+
"categories_removed": "Input categories removed: all of them ({n}).",
|
|
224
|
+
"sentences_removed": "Input sentences removed: all of them ({n}).",
|
|
225
|
+
"metadata_removed": "Metadata removed:",
|
|
226
|
+
"sentences": "Sentences ({n}):",
|
|
227
|
+
"empty_sentence": "*(empty sentence)*",
|
|
228
|
+
"produced_document": "Produced document",
|
|
229
|
+
"identifier_named": "identifier",
|
|
230
|
+
"title": "Title",
|
|
231
|
+
"properties": "Properties:",
|
|
232
|
+
"sheet": "Sheet `{name}`",
|
|
233
|
+
"first_sheet": "Sheet `{name}`, the first of {n}",
|
|
234
|
+
"encoding": "encoded in {encoding}",
|
|
235
|
+
"decimal": ".",
|
|
208
236
|
},
|
|
209
237
|
}
|
|
210
238
|
|
|
@@ -287,6 +315,8 @@ def existing_rows(body: str) -> dict[str, list[str]]:
|
|
|
287
315
|
def parameters_block(schema: dict, old_body: str, lang: str) -> str:
|
|
288
316
|
texts = TEXTS[lang]
|
|
289
317
|
old = existing_rows(old_body)
|
|
318
|
+
if not any("internal" not in str(prop.get("extra", "")) for prop in schema.get("properties", {}).values()):
|
|
319
|
+
return texts["no_parameters"] + "\n"
|
|
290
320
|
lines = [texts["header"], "|---|---|---|"]
|
|
291
321
|
for name, prop in schema.get("properties", {}).items():
|
|
292
322
|
extra = {flag.strip() for flag in str(prop.get("extra", "")).split(",")}
|
|
@@ -358,17 +388,29 @@ def text_lead(doc: dict, lang: str) -> list[str]:
|
|
|
358
388
|
return [lead + texts["colon"], *excerpt(doc.get("text", ""), lang)]
|
|
359
389
|
|
|
360
390
|
|
|
391
|
+
# Une ligne indentée, un simple retour à la ligne entre deux lignes, ou des espaces alignés :
|
|
392
|
+
# une citation Markdown les écraserait.
|
|
393
|
+
LAYOUT = re.compile(r"^[ \t]+\S|\S[ \t]*\n[ \t]*\S|\S {2,}\S|\t", re.MULTILINE)
|
|
394
|
+
|
|
395
|
+
|
|
361
396
|
def excerpt(text: str, lang: str) -> list[str]:
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
397
|
+
"""Le texte, cité ; dans un bloc de code quand sa mise en page compte, pour qu'on la voie."""
|
|
398
|
+
shown = text if len(text) <= MAX_TEXT else text[:MAX_TEXT].rsplit(" ", 1)[0] + " …"
|
|
399
|
+
parts = code_block(shown, "text", lang, max_line=None) if LAYOUT.search(text.strip("\n")) else [quote(shown)]
|
|
400
|
+
if len(text) > MAX_TEXT:
|
|
401
|
+
parts.append(TEXTS[lang]["truncated_text"].format(n=len(text)))
|
|
402
|
+
return parts
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def number(value: float, lang: str) -> str:
|
|
406
|
+
return f"{value:.2f}".replace(".", TEXTS[lang]["decimal"])
|
|
365
407
|
|
|
366
408
|
|
|
367
|
-
def code_block(text: str, kind: str, lang: str) -> list[str]:
|
|
409
|
+
def code_block(text: str, kind: str, lang: str, max_line: int | None = MAX_LINE) -> list[str]:
|
|
368
410
|
"""Un extrait de contenu dans un bloc de code ; 'kind' est un type MIME ou un nom de fichier."""
|
|
369
411
|
fence_lang = next((name for name in ("json", "xml", "html", "csv", "markdown") if name in kind.lower()), "text")
|
|
370
|
-
lines = text.
|
|
371
|
-
shown = [line if len(line) <=
|
|
412
|
+
lines = text.strip("\n").splitlines()
|
|
413
|
+
shown = [line if max_line is None or len(line) <= max_line else line[:max_line] + " …" for line in lines[:MAX_LINES]]
|
|
372
414
|
fence = "~~~~" if "```" in text else "```"
|
|
373
415
|
parts = [f"{fence}{fence_lang}\n" + "\n".join(shown) + f"\n{fence}"]
|
|
374
416
|
if len(lines) > MAX_LINES:
|
|
@@ -376,16 +418,51 @@ def code_block(text: str, kind: str, lang: str) -> list[str]:
|
|
|
376
418
|
return parts
|
|
377
419
|
|
|
378
420
|
|
|
421
|
+
# L'encodage que la fixture a reconnu, sous le nom que connaît un utilisateur.
|
|
422
|
+
ENCODINGS = {"cp1252": "Latin-1 / Windows-1252"}
|
|
423
|
+
|
|
424
|
+
|
|
379
425
|
def file_lead(source: dict, lang: str) -> list[str]:
|
|
380
426
|
"""Le fichier d'entrée d'un convertisseur ou d'un importeur : son nom, puis son contenu."""
|
|
381
427
|
texts = TEXTS[lang]
|
|
382
428
|
media_type = source.get("content_type")
|
|
383
|
-
|
|
384
|
-
if "
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
429
|
+
details = [media_type] if media_type else []
|
|
430
|
+
if source.get("encoding"):
|
|
431
|
+
details.append(texts["encoding"].format(encoding=ENCODINGS.get(source["encoding"], source["encoding"])))
|
|
432
|
+
lead = f"{texts['file']}{texts['colon']} `{source.get('filename', '')}`" + (f" ({', '.join(details)})" if details else "")
|
|
433
|
+
return [lead, *content_view(source, media_type or source.get("filename", ""), lang)]
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def column_name(index: int) -> str:
|
|
437
|
+
"""A, B, …, Z, AA, AB… : le nom d'une colonne Excel."""
|
|
438
|
+
name = ""
|
|
439
|
+
index += 1
|
|
440
|
+
while index:
|
|
441
|
+
index, rest = divmod(index - 1, 26)
|
|
442
|
+
name = chr(ord("A") + rest) + name
|
|
443
|
+
return name
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def content_view(content: dict, kind: str, lang: str) -> list[str]:
|
|
447
|
+
"""Le contenu d'un fichier ou d'une réponse : son texte, la première feuille d'un classeur, ou sa taille."""
|
|
448
|
+
texts = TEXTS[lang]
|
|
449
|
+
if "text" in content:
|
|
450
|
+
return code_block(content["text"], kind, lang)
|
|
451
|
+
if "sheet" in content:
|
|
452
|
+
# La taille d'un classeur dépend de la version d'openpyxl : on montre ses cellules.
|
|
453
|
+
sheet = content["sheet"]
|
|
454
|
+
many = sheet.get("sheets", 1) > 1
|
|
455
|
+
lead = texts["first_sheet" if many else "sheet"].format(name=cell(sheet.get("name", "")), n=sheet.get("sheets", 1))
|
|
456
|
+
rows = [[cell(value) or " " for value in row] for row in sheet.get("rows", [])]
|
|
457
|
+
if not rows:
|
|
458
|
+
return [lead + texts["colon"], texts["empty"]]
|
|
459
|
+
width = max(len(row) for row in rows)
|
|
460
|
+
rows = [row + [" "] * (width - len(row)) for row in rows]
|
|
461
|
+
# Les colonnes nommées comme dans Excel : rien ne dit si la première ligne est un en-tête.
|
|
462
|
+
return [lead + texts["colon"], table([column_name(i) for i in range(width)], rows, lang)]
|
|
463
|
+
if "size" in content:
|
|
464
|
+
return [texts["binary"].format(n=content["size"])]
|
|
465
|
+
return []
|
|
389
466
|
|
|
390
467
|
|
|
391
468
|
def annotations_table(annotations: list[dict], lang: str, doc_text: str = "") -> str:
|
|
@@ -424,7 +501,7 @@ def categories_table(categories: list[dict], lang: str) -> str:
|
|
|
424
501
|
for c in categories:
|
|
425
502
|
row = [f"`{cell(c.get('labelName') or c.get('label') or '')}`"]
|
|
426
503
|
if with_score:
|
|
427
|
-
row.append(
|
|
504
|
+
row.append(number(c["score"], lang) if isinstance(c.get("score"), (int, float)) else "—")
|
|
428
505
|
rows.append(row)
|
|
429
506
|
return table(header, rows, lang)
|
|
430
507
|
|
|
@@ -446,34 +523,100 @@ def input_view(doc: dict, lang: str) -> list[str]:
|
|
|
446
523
|
return parts
|
|
447
524
|
|
|
448
525
|
|
|
526
|
+
MARKDOWN_INLINE = re.compile(r"([*_`])")
|
|
527
|
+
MARKDOWN_LEAD = re.compile(r"^([-+#>])")
|
|
528
|
+
MARKDOWN_NUMBER = re.compile(r"^(\d+)([.)])")
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def sentence_text(text: str, sent: dict, lang: str, limit: int | None = None) -> str:
|
|
532
|
+
# Échappée : une phrase « * Conseil » ou « 1. Bilan » deviendrait une liste dans la liste.
|
|
533
|
+
phrase = MARKDOWN_INLINE.sub(r"\\\1", cell(text[sent["start"] : sent["end"]]))
|
|
534
|
+
phrase = MARKDOWN_NUMBER.sub(r"\1\\\2", MARKDOWN_LEAD.sub(r"\\\1", phrase))
|
|
535
|
+
if not phrase:
|
|
536
|
+
return TEXTS[lang]["empty_sentence"]
|
|
537
|
+
return phrase if limit is None or len(phrase) <= limit else phrase[:limit].rsplit(" ", 1)[0] + " …"
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def sentences_list(doc: dict, lang: str) -> list[str]:
|
|
541
|
+
"""Les phrases, numérotées ; une phrase vide est dite, pas sautée."""
|
|
542
|
+
texts = TEXTS[lang]
|
|
543
|
+
sentences = doc.get("sentences") or []
|
|
544
|
+
if not sentences:
|
|
545
|
+
return [texts["no_sentences"]]
|
|
546
|
+
text = doc.get("text", "")
|
|
547
|
+
parts = ["\n".join(f"{n}. {sentence_text(text, s, lang)}" for n, s in enumerate(sentences[:MAX_ROWS], start=1))]
|
|
548
|
+
if len(sentences) > MAX_ROWS:
|
|
549
|
+
parts.append(texts["truncated_rows"].format(n=len(sentences)))
|
|
550
|
+
return parts
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def sentence_categories(doc: dict, lang: str) -> list[str]:
|
|
554
|
+
"""Un processeur peut catégoriser chaque phrase plutôt que le document : phrase, labels et scores."""
|
|
555
|
+
texts = TEXTS[lang]
|
|
556
|
+
text = doc.get("text", "")
|
|
557
|
+
rows = []
|
|
558
|
+
for sent in doc["sentences"]:
|
|
559
|
+
if not sent.get("categories"):
|
|
560
|
+
continue
|
|
561
|
+
labels = []
|
|
562
|
+
for c in sent["categories"]:
|
|
563
|
+
label = f"`{cell(c.get('labelName') or c.get('label') or '')}`"
|
|
564
|
+
labels.append(label + (f" ({number(c['score'], lang)})" if isinstance(c.get("score"), (int, float)) else ""))
|
|
565
|
+
rows.append([sentence_text(text, sent, lang, limit=80), ", ".join(labels)])
|
|
566
|
+
return [texts["sentence_categories"], table([texts["sentence"], texts["categories"]], rows, lang)] if rows else []
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
def document_lead(doc: dict, lang: str) -> list[str]:
|
|
570
|
+
"""Un document produit sans document en entrée (un convertisseur) : identifiant, titre, propriétés."""
|
|
571
|
+
texts = TEXTS[lang]
|
|
572
|
+
lead = texts["produced_document"]
|
|
573
|
+
if doc.get("identifier"):
|
|
574
|
+
lead += f", {texts['identifier_named']} `{cell(doc['identifier'])}`"
|
|
575
|
+
parts = [lead + "."]
|
|
576
|
+
if doc.get("title"):
|
|
577
|
+
parts.append(f"{texts['title']}{texts['colon']} {cell(doc['title'])}")
|
|
578
|
+
if doc.get("properties"):
|
|
579
|
+
parts += [texts["properties"], metadata_table(doc["properties"], lang)]
|
|
580
|
+
return parts
|
|
581
|
+
|
|
582
|
+
|
|
449
583
|
def output_view(doc: dict, before: dict | None, lang: str) -> list[str]:
|
|
450
|
-
"""Un document en sortie : ce qui a changé par rapport à l'entrée,
|
|
584
|
+
"""Un document en sortie : ce qui a changé par rapport à l'entrée, ce qui a été retiré compris."""
|
|
451
585
|
texts = TEXTS[lang]
|
|
586
|
+
parts = document_lead(doc, lang) if before is None else []
|
|
452
587
|
before = before or {}
|
|
453
|
-
parts = []
|
|
454
588
|
if doc.get("text", "") != before.get("text", ""):
|
|
455
589
|
parts += [texts["produced_text"], *excerpt(doc.get("text", ""), lang)]
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
590
|
+
# Une liste vide vaut une liste absente (exclude_none ne retire que les None).
|
|
591
|
+
if (doc.get("sentences") or []) != (before.get("sentences") or []):
|
|
592
|
+
if doc.get("sentences"):
|
|
593
|
+
categorized = sentence_categories(doc, lang)
|
|
594
|
+
if categorized:
|
|
595
|
+
parts += categorized
|
|
596
|
+
else:
|
|
597
|
+
parts += [texts["sentences"].format(n=len(doc["sentences"])), *sentences_list(doc, lang)]
|
|
598
|
+
else:
|
|
599
|
+
parts.append(texts["sentences_removed"].format(n=len(before["sentences"])))
|
|
600
|
+
# Une liste qui disparaît se dit : sans quoi un processeur qui vide le document paraîtrait inactif.
|
|
601
|
+
tables = {
|
|
602
|
+
"annotations": lambda: annotations_table(doc["annotations"], lang, doc.get("text", "")),
|
|
603
|
+
"categories": lambda: categories_table(doc["categories"], lang),
|
|
604
|
+
}
|
|
605
|
+
for key, table_of in tables.items():
|
|
606
|
+
if (doc.get(key) or []) == (before.get(key) or []):
|
|
607
|
+
continue
|
|
608
|
+
if doc.get(key):
|
|
609
|
+
parts += [texts[f"{key}_out"], table_of()]
|
|
610
|
+
else:
|
|
611
|
+
parts.append(texts[f"{key}_removed"].format(n=len(before[key])))
|
|
473
612
|
old_metadata = before.get("metadata") or {}
|
|
474
|
-
|
|
613
|
+
metadata = doc.get("metadata") or {}
|
|
614
|
+
changed = {key: value for key, value in metadata.items() if old_metadata.get(key) != value}
|
|
475
615
|
if changed:
|
|
476
616
|
parts += [texts["metadata_out"], metadata_table(changed, lang)]
|
|
617
|
+
removed = [key for key in old_metadata if key not in metadata]
|
|
618
|
+
if removed:
|
|
619
|
+
parts.append(f"{texts['metadata_removed']} " + ", ".join(f"`{cell(key)}`" for key in removed) + ".")
|
|
477
620
|
return parts or [texts["unchanged"]]
|
|
478
621
|
|
|
479
622
|
|
|
@@ -516,19 +659,10 @@ def processor_example(example: dict, schema: dict, lang: str) -> str:
|
|
|
516
659
|
|
|
517
660
|
def segmenter_example(example: dict, schema: dict, lang: str) -> str:
|
|
518
661
|
"""Le texte, les paramètres, puis les phrases, numérotées."""
|
|
519
|
-
texts = TEXTS[lang]
|
|
520
662
|
parts = [part for doc in as_list(example["inputs"]) for part in text_lead(doc, lang)]
|
|
521
663
|
parts.append(parameters_line(example["parameters"], schema, lang))
|
|
522
664
|
for doc in as_list(example["outputs"]):
|
|
523
|
-
|
|
524
|
-
if not sentences:
|
|
525
|
-
parts.append(texts["no_sentences"])
|
|
526
|
-
continue
|
|
527
|
-
text = doc.get("text", "")
|
|
528
|
-
items = [f"{n}. {cell(text[s['start'] : s['end']])}" for n, s in enumerate(sentences[:MAX_ROWS], start=1)]
|
|
529
|
-
parts.append("\n".join(items))
|
|
530
|
-
if len(sentences) > MAX_ROWS:
|
|
531
|
-
parts.append(texts["truncated_rows"].format(n=len(sentences)))
|
|
665
|
+
parts += sentences_list(doc, lang)
|
|
532
666
|
return "\n\n".join(parts) + "\n"
|
|
533
667
|
|
|
534
668
|
|
|
@@ -552,10 +686,7 @@ def formatter_example(example: dict, schema: dict, lang: str) -> str:
|
|
|
552
686
|
if filename:
|
|
553
687
|
lead += f", {texts['file_named']} `{filename.group(1)}`"
|
|
554
688
|
parts.append(lead)
|
|
555
|
-
|
|
556
|
-
parts += code_block(response["text"], media_type or "", lang)
|
|
557
|
-
elif "size" in response:
|
|
558
|
-
parts.append(texts["binary"].format(n=response["size"]))
|
|
689
|
+
parts += content_view(response, media_type or "", lang)
|
|
559
690
|
return "\n\n".join(parts) + "\n"
|
|
560
691
|
|
|
561
692
|
|
|
@@ -115,11 +115,11 @@ que le test passe à `help_example` aussi :
|
|
|
115
115
|
| type | `inputs` | `outputs` | ce que la page montre |
|
|
116
116
|
|------|----------|-----------|------------------------|
|
|
117
117
|
| annotateur | les documents | les documents | le texte, puis les annotations |
|
|
118
|
-
| processeur | les documents | les documents | le texte et ce qu'il porte déjà, puis **ce qui a changé** : texte, phrases, annotations, catégories (du document ou par phrase), métadonnées |
|
|
118
|
+
| processeur | les documents | les documents | le texte et ce qu'il porte déjà, puis **ce qui a changé** : texte, phrases, annotations, catégories (du document ou par phrase, avec leur score), métadonnées — **et ce qui a été retiré** |
|
|
119
119
|
| segmenteur | les documents | les documents | le texte, puis les phrases numérotées |
|
|
120
|
-
| convertisseur | l'`UploadFile` | les documents | le fichier (extrait), puis les
|
|
121
|
-
| formateur | le document | la `Response` | le document, puis le type, le nom de fichier et un extrait de la réponse —
|
|
122
|
-
| importeur | le chemin du fichier | **la liste** des concepts | le fichier (extrait), puis les
|
|
120
|
+
| convertisseur | l'`UploadFile` | les documents | le fichier (extrait), puis chaque document produit : identifiant, titre, propriétés, texte et phrases numérotées (les phrases vides comprises) |
|
|
121
|
+
| formateur | le document | la `Response` | le document, puis le type, le nom de fichier et un extrait de la réponse — la première feuille pour un classeur Excel, la taille pour un autre contenu binaire |
|
|
122
|
+
| importeur | le chemin du fichier | **la liste** des concepts | le fichier (extrait, ou la première feuille d'un classeur Excel), puis les termes : identifiant, forme préférée, formes alternatives |
|
|
123
123
|
|
|
124
124
|
Trois pièges, que la fixture signale ou ne peut pas rattraper :
|
|
125
125
|
|
|
@@ -136,6 +136,14 @@ Les textes, fichiers et réponses longs sont tronqués au rendu (1 000 caractèr
|
|
|
136
136
|
20 lignes de 160 caractères, 20 lignes de tableau) : choisir des fixtures courtes
|
|
137
137
|
plutôt que compter sur la troncature.
|
|
138
138
|
|
|
139
|
+
Un texte dont la mise en page compte (une ligne indentée, un simple retour à la
|
|
140
|
+
ligne, des espaces alignés) est montré dans un bloc de code plutôt qu'en citation,
|
|
141
|
+
qui l'écraserait. Un fichier qui n'est pas en UTF-8 est lu en Latin-1 et le rendu le
|
|
142
|
+
dit. Un classeur Excel est montré par sa première feuille, pas par sa taille : la
|
|
143
|
+
taille d'un `.xlsx` change avec la version d'openpyxl et ferait échouer
|
|
144
|
+
`py:help-check`. Il faut openpyxl dans l'environnement de test, ce qui est déjà le
|
|
145
|
+
cas d'un plugin qui lit ou écrit de l'Excel.
|
|
146
|
+
|
|
139
147
|
Un exemple dans une langue (un texte français dans la page anglaise) se garde tel
|
|
140
148
|
quel — la langue du document est indiquée par le rendu (« Text (French) ») : le
|
|
141
149
|
traduire, c'est montrer une sortie que personne n'a obtenue.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/build_rerank_data.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/eval_linking.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/eval_mappings.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/eval_rerank.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/finetune_rerank.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/merge_rerank.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/recalibrate_linking.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/review_linking.html
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/review_linking.py
RENAMED
|
File without changes
|
{pyannotators_entityfishing-1.6.97 → pyannotators_entityfishing-1.6.99}/tools/sherpa_plan_project.py
RENAMED
|
File without changes
|
|
File without changes
|