@piecemaker-legal/piecemaker 1.0.6 → 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{abnfDiagram-VCTEODGH-DF7LFtT-.js → abnfDiagram-VCTEODGH-CR3_ADBu.js} +1 -1
- package/dist/assets/{arc-xOmjWeVH.js → arc-b92imMmc.js} +1 -1
- package/dist/assets/{architectureDiagram-5GKGNRK7-CESzNPwa.js → architectureDiagram-5GKGNRK7-Dfm5Epud.js} +1 -1
- package/dist/assets/{blockDiagram-I7D4REHJ-BjdN78Ad.js → blockDiagram-I7D4REHJ-B8kLLyAv.js} +1 -1
- package/dist/assets/{c4Diagram-7LVT6UL2-BcK9EDWT.js → c4Diagram-7LVT6UL2-BWSyAAWX.js} +1 -1
- package/dist/assets/channel-BbcW_Tkh.js +1 -0
- package/dist/assets/{chunk-2Q5K7J3B-DGUN1JFP.js → chunk-2Q5K7J3B-Ci743mcC.js} +1 -1
- package/dist/assets/{chunk-5VM5RSS4-BG3BBcFr.js → chunk-5VM5RSS4-SdXpal6h.js} +1 -1
- package/dist/assets/{chunk-F27PBJKO-CLIOfrF5.js → chunk-F27PBJKO-Rof94ueX.js} +1 -1
- package/dist/assets/{chunk-IMKFNOWR-D6lGkqij.js → chunk-IMKFNOWR-B3st6ECc.js} +1 -1
- package/dist/assets/{chunk-JWPE2WC7-BmF9vLBT.js → chunk-JWPE2WC7-alXmu5xU.js} +1 -1
- package/dist/assets/{chunk-POPQ4Y6H-C_0SR3ak.js → chunk-POPQ4Y6H-dMs1OCce.js} +1 -1
- package/dist/assets/{chunk-SVP7TREG-rT20Wfpp.js → chunk-SVP7TREG-BmM8FsmF.js} +1 -1
- package/dist/assets/{chunk-TICWLB2K-Cn2hkmcf.js → chunk-TICWLB2K-CL1LytYG.js} +1 -1
- package/dist/assets/{chunk-XXDRQBXY-Dm_eP1_v.js → chunk-XXDRQBXY-7U9iRPT1.js} +1 -1
- package/dist/assets/classDiagram-ZZMXUADV-D9_YRZ7i.js +1 -0
- package/dist/assets/classDiagram-v2-VYDZK3BY-D9_YRZ7i.js +1 -0
- package/dist/assets/{cose-bilkent-JH36ORCC-cJeeJsep.js → cose-bilkent-JH36ORCC-DFOMlhj2.js} +1 -1
- package/dist/assets/{cynefin-OW5HDTMX-DX2ePIXe.js → cynefin-OW5HDTMX-CDdJFd9d.js} +1 -1
- package/dist/assets/{cynefinDiagram-5FMLGOSQ-BvWXN-RR.js → cynefinDiagram-5FMLGOSQ-DPyVNrMW.js} +1 -1
- package/dist/assets/{dagre-GXQ25YYZ-CkwPRhGR.js → dagre-GXQ25YYZ-B3bpHa9D.js} +1 -1
- package/dist/assets/{diagram-S7CK7UJ4-CBGNzYyv.js → diagram-S7CK7UJ4-DeFO2rXB.js} +1 -1
- package/dist/assets/{diagram-UQ7AKVKN-CiwCU0SJ.js → diagram-UQ7AKVKN-CbdGTOVf.js} +1 -1
- package/dist/assets/{diagram-VSXAHHWV-Aekg4nTk.js → diagram-VSXAHHWV-DFfNQyMd.js} +1 -1
- package/dist/assets/{diagram-VX7I27RA-DPFP5K7r.js → diagram-VX7I27RA-DbNZ6_X3.js} +1 -1
- package/dist/assets/{diagram-Z3DM3KII-BHJkxoey.js → diagram-Z3DM3KII-B8F-ZkNJ.js} +1 -1
- package/dist/assets/{ebnfDiagram-PWID7BFC-2rV1P32T.js → ebnfDiagram-PWID7BFC-BfCCPPAP.js} +1 -1
- package/dist/assets/{erDiagram-RLTQ6QDP-C4BjyS_H.js → erDiagram-RLTQ6QDP-DcV-3Kzv.js} +1 -1
- package/dist/assets/{flowDiagram-HODETNUW-DL1TBlAG.js → flowDiagram-HODETNUW-BwSkJXCo.js} +1 -1
- package/dist/assets/{ganttDiagram-EL5Y4UJY-MKjvzCWz.js → ganttDiagram-EL5Y4UJY-BffHk77T.js} +1 -1
- package/dist/assets/{gitGraphDiagram-WWUBYQGX-CS_oTGK5.js → gitGraphDiagram-WWUBYQGX-CQ1EWa74.js} +1 -1
- package/dist/assets/{index-XOGQDqU3.js → index-BqK33gaZ.js} +120 -120
- package/dist/assets/index-D_Dc_nQv.css +1 -0
- package/dist/assets/{infoDiagram-27XIBGKW-ihUr2GjN.js → infoDiagram-27XIBGKW-DqIaBXQn.js} +1 -1
- package/dist/assets/{ishikawaDiagram-5VMMS53U-D2YAnuir.js → ishikawaDiagram-5VMMS53U-HlhsaZgv.js} +1 -1
- package/dist/assets/{journeyDiagram-3NMN7TZE-Bc2hSZGD.js → journeyDiagram-3NMN7TZE-BYO8-7rN.js} +1 -1
- package/dist/assets/{kanban-definition-UXKFOSKX-LRMGNonX.js → kanban-definition-UXKFOSKX-DXb8zHGT.js} +1 -1
- package/dist/assets/{layout-CcfWOlPV.js → layout-CREOVuEZ.js} +1 -1
- package/dist/assets/{linear-7JoS2gQO.js → linear-CVCrIRIY.js} +1 -1
- package/dist/assets/{mermaid.core-ijh_SsuN.js → mermaid.core-DFqeWtQR.js} +6 -6
- package/dist/assets/{mindmap-definition-YA3MSWOX-DCE4nzfF.js → mindmap-definition-YA3MSWOX-SUpXz333.js} +1 -1
- package/dist/assets/{pegDiagram-XKGWAZYB-SjaGa7lE.js → pegDiagram-XKGWAZYB-D-yWpve3.js} +1 -1
- package/dist/assets/{pieDiagram-E7YTZNPT-1xcRidSZ.js → pieDiagram-E7YTZNPT-BJX4Zp4Z.js} +1 -1
- package/dist/assets/{quadrantDiagram-AXDQQJYC-obHAHnYV.js → quadrantDiagram-AXDQQJYC-CHajheTu.js} +1 -1
- package/dist/assets/{railroadDiagram-O6MQD6OU-BsCVcRZb.js → railroadDiagram-O6MQD6OU-DHJWzW5A.js} +1 -1
- package/dist/assets/{requirementDiagram-BXWQKSXE-CUNkhvqb.js → requirementDiagram-BXWQKSXE-DigUIdso.js} +1 -1
- package/dist/assets/{sankeyDiagram-P5KCCOFB-DhezXMKm.js → sankeyDiagram-P5KCCOFB-CFUDDbA5.js} +1 -1
- package/dist/assets/{sequenceDiagram-WJ2MYXX4-CL9VswXH.js → sequenceDiagram-WJ2MYXX4-6tSaXfYD.js} +1 -1
- package/dist/assets/{sizeCapture-INFHLROL-t0g9jiEz.js → sizeCapture-INFHLROL-CXASEDKe.js} +1 -1
- package/dist/assets/{stateDiagram-D77RDMKH-BW-g-IOz.js → stateDiagram-D77RDMKH-DWTCAQfZ.js} +1 -1
- package/dist/assets/stateDiagram-v2-MP3YSRHH-HDQF4MGA.js +1 -0
- package/dist/assets/{swimlanes-42K2YHIH-LsuyeBte.js → swimlanes-42K2YHIH-7-oz5r-e.js} +1 -1
- package/dist/assets/swimlanesDiagram-VR7AAH4N-CJbgQFt8.js +8 -0
- package/dist/assets/{timeline-definition-24CTP7MA-B0cFka1R.js → timeline-definition-24CTP7MA-DVH5g4ty.js} +1 -1
- package/dist/assets/{vendor-codemirror-C5qqu4R7.js → vendor-codemirror-CjW0lHR5.js} +5 -5
- package/dist/assets/{vennDiagram-4TSXK5OY-D3HNCDR3.js → vennDiagram-4TSXK5OY-CQvY_ALZ.js} +1 -1
- package/dist/assets/{wardleyDiagram-VM6X3IG4-DJ3Co8nb.js → wardleyDiagram-VM6X3IG4-BdlTJgEc.js} +1 -1
- package/dist/assets/{xychartDiagram-S5SC5T6Z-Cz5E9New.js → xychartDiagram-S5SC5T6Z-CYn2sV77.js} +1 -1
- package/dist/favicon.png +0 -0
- package/dist/favicon.svg +1 -9
- package/dist/generate-icons.js +47 -46
- package/dist/icons/icon-128x128.png +0 -0
- package/dist/icons/icon-128x128.svg +1 -12
- package/dist/icons/icon-144x144.png +0 -0
- package/dist/icons/icon-144x144.svg +1 -12
- package/dist/icons/icon-152x152.png +0 -0
- package/dist/icons/icon-152x152.svg +1 -12
- package/dist/icons/icon-192x192.png +0 -0
- package/dist/icons/icon-192x192.svg +1 -12
- package/dist/icons/icon-384x384.png +0 -0
- package/dist/icons/icon-384x384.svg +1 -12
- package/dist/icons/icon-512x512.png +0 -0
- package/dist/icons/icon-512x512.svg +1 -12
- package/dist/icons/icon-72x72.png +0 -0
- package/dist/icons/icon-72x72.svg +1 -12
- package/dist/icons/icon-96x96.png +0 -0
- package/dist/icons/icon-96x96.svg +1 -12
- package/dist/index.html +3 -3
- package/dist/logo-128.png +0 -0
- package/dist/logo-256.png +0 -0
- package/dist/logo-32.png +0 -0
- package/dist/logo-512.png +0 -0
- package/dist/logo-64.png +0 -0
- package/dist/logo-sources/logo-black.png +0 -0
- package/dist/logo-sources/logo-white.png +0 -0
- package/dist/logo.svg +1 -17
- package/dist/sw.js +1 -1
- package/dist-server/plugins/piecemaker-dossier/src/institutional-terms.js +91 -0
- package/dist-server/plugins/piecemaker-dossier/src/institutional-terms.js.map +1 -0
- package/dist-server/plugins/piecemaker-dossier/src/knowledge.js +112 -10
- package/dist-server/plugins/piecemaker-dossier/src/knowledge.js.map +1 -1
- package/dist-server/plugins/piecemaker-dossier/src/scan-result.js +5 -1
- package/dist-server/plugins/piecemaker-dossier/src/scan-result.js.map +1 -1
- package/dist-server/server/index.js +60 -26
- package/dist-server/server/index.js.map +1 -1
- package/dist-server/server/modules/database/repositories/projects.db.js +28 -6
- package/dist-server/server/modules/database/repositories/projects.db.js.map +1 -1
- package/dist-server/server/modules/database/tests/projects.db.integration.test.js +58 -1
- package/dist-server/server/modules/database/tests/projects.db.integration.test.js.map +1 -1
- package/dist-server/server/modules/plugins/plugins.routes.js +8 -1
- package/dist-server/server/modules/plugins/plugins.routes.js.map +1 -1
- package/dist-server/server/modules/projects/services/projects-with-sessions-fetch.service.js +2 -0
- package/dist-server/server/modules/projects/services/projects-with-sessions-fetch.service.js.map +1 -1
- package/dist-server/server/piecemaker/bodacc-search.js +191 -0
- package/dist-server/server/piecemaker/bodacc-search.js.map +1 -0
- package/dist-server/server/piecemaker/company-search.js +145 -0
- package/dist-server/server/piecemaker/company-search.js.map +1 -0
- package/dist-server/server/piecemaker/company-search.test.js +18 -0
- package/dist-server/server/piecemaker/company-search.test.js.map +1 -0
- package/dist-server/server/piecemaker/index.js +6 -2
- package/dist-server/server/piecemaker/index.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/index.js +9 -4
- package/dist-server/server/piecemaker/knowledge/index.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/pipeline.js +68 -58
- package/dist-server/server/piecemaker/knowledge/pipeline.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/routes.js +2 -4
- package/dist-server/server/piecemaker/knowledge/routes.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/scan-jobs.js +91 -0
- package/dist-server/server/piecemaker/knowledge/scan-jobs.js.map +1 -0
- package/dist-server/server/piecemaker/knowledge/service.js +18 -36
- package/dist-server/server/piecemaker/knowledge/service.js.map +1 -1
- package/dist-server/server/piecemaker/library/connector-installation.js +224 -0
- package/dist-server/server/piecemaker/library/connector-installation.js.map +1 -0
- package/dist-server/server/piecemaker/library/index.js +11 -1
- package/dist-server/server/piecemaker/library/index.js.map +1 -1
- package/dist-server/server/piecemaker/library/marketplace.js +102 -43
- package/dist-server/server/piecemaker/library/marketplace.js.map +1 -1
- package/dist-server/server/piecemaker/library/migrate.js +8 -1
- package/dist-server/server/piecemaker/library/migrate.js.map +1 -1
- package/dist-server/server/piecemaker/library/provider-agents.js +50 -0
- package/dist-server/server/piecemaker/library/provider-agents.js.map +1 -0
- package/dist-server/server/piecemaker/library/provider-connectors.js +80 -0
- package/dist-server/server/piecemaker/library/provider-connectors.js.map +1 -0
- package/dist-server/server/piecemaker/library/provider-skills.js +11 -2
- package/dist-server/server/piecemaker/library/provider-skills.js.map +1 -1
- package/dist-server/server/piecemaker/library/store.js +125 -35
- package/dist-server/server/piecemaker/library/store.js.map +1 -1
- package/dist-server/server/piecemaker/library/workspace-installation.js +53 -29
- package/dist-server/server/piecemaker/library/workspace-installation.js.map +1 -1
- package/dist-server/server/piecemaker/timesheet/service.js +4 -21
- package/dist-server/server/piecemaker/timesheet/service.js.map +1 -1
- package/package.json +3 -2
- package/scripts/piecemaker/cli/install-command.mjs +35 -9
- package/scripts/piecemaker/cli/lib/config.mjs +1 -0
- package/scripts/piecemaker/cli/lib/piecemaker-anonymizer.mjs +78 -0
- package/scripts/piecemaker/cli/lib/plugins.mjs +2 -0
- package/scripts/piecemaker/cli/lib/pwa.mjs +46 -7
- package/scripts/piecemaker/cli/lib/services.mjs +65 -4
- package/scripts/piecemaker/cli/piecemaker-command.test.mjs +25 -0
- package/scripts/piecemaker/cli/piecemaker.mjs +45 -44
- package/scripts/piecemaker/cli/piecemaker.ps1 +8 -0
- package/scripts/piecemaker/cli/piecemaker.sh +6 -1
- package/server/index.ts +52 -23
- package/server/modules/database/repositories/projects.db.ts +28 -6
- package/server/modules/database/tests/projects.db.integration.test.ts +63 -1
- package/server/modules/plugins/plugins.routes.ts +8 -1
- package/server/modules/projects/services/projects-with-sessions-fetch.service.ts +5 -0
- package/server/piecemaker/anonymizer/client-config.cjs +10 -2
- package/server/piecemaker/anonymizer/dictionary.cjs +6 -72
- package/server/piecemaker/bodacc-search.ts +205 -0
- package/server/piecemaker/company-search.test.ts +26 -0
- package/server/piecemaker/company-search.ts +183 -0
- package/server/piecemaker/index.ts +8 -2
- package/server/piecemaker/knowledge/index.ts +9 -4
- package/server/piecemaker/knowledge/pipeline.ts +89 -57
- package/server/piecemaker/knowledge/routes.ts +2 -4
- package/server/piecemaker/knowledge/scan-jobs.ts +111 -0
- package/server/piecemaker/knowledge/service.ts +18 -33
- package/server/piecemaker/library/connector-installation.ts +248 -0
- package/server/piecemaker/library/index.ts +5 -1
- package/server/piecemaker/library/marketplace.ts +104 -39
- package/server/piecemaker/library/migrate.ts +8 -1
- package/server/piecemaker/library/provider-agents.ts +49 -0
- package/server/piecemaker/library/provider-connectors.ts +89 -0
- package/server/piecemaker/library/provider-skills.ts +10 -1
- package/server/piecemaker/library/store.ts +129 -36
- package/server/piecemaker/library/workspace-installation.ts +54 -29
- package/server/piecemaker/router.cjs +2 -1
- package/server/piecemaker/timesheet/service.ts +4 -16
- package/server/piecemaker/vendor/installer/steps/03-python-gliner.mjs +0 -27
- package/server/piecemaker/vendor/installer/steps/14-mxc-sandbox.mjs +1 -1
- package/server/piecemaker/vendor/piecemaker-plugin/scripts/lib/substitution.cjs +2 -3
- package/server/piecemaker/vendor/piecemaker-plugin/skills/conversion-md/SKILL.md +3 -3
- package/server/piecemaker/vendor/websocket-server/admin-routes.cjs +3 -10
- package/server/piecemaker/vendor/websocket-server/originals-pipeline.cjs +135 -4
- package/server/piecemaker/vendor/websocket-server/process-group.cjs +95 -0
- package/server/piecemaker/vendor/websocket-server/scripts/__pycache__/scan_utils.cpython-312.pyc +0 -0
- package/server/piecemaker/vendor/websocket-server/scripts/convert_and_scan_pipeline.py +280 -149
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/__pycache__/model_config.cpython-312.pyc +0 -0
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/__pycache__/scanner_worker.cpython-312.pyc +0 -0
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/scanner_worker.py +35 -29
- package/server/piecemaker/vendor/websocket-server/scripts/requirements.txt +0 -5
- package/server/shared/types.ts +1 -0
- package/dist/assets/channel-Dz3eE5vU.js +0 -1
- package/dist/assets/classDiagram-ZZMXUADV-BC1eKV2j.js +0 -1
- package/dist/assets/classDiagram-v2-VYDZK3BY-BC1eKV2j.js +0 -1
- package/dist/assets/index-DhwJpHCq.css +0 -1
- package/dist/assets/stateDiagram-v2-MP3YSRHH-BXuT4zDG.js +0 -1
- package/dist/assets/swimlanesDiagram-VR7AAH4N-B0eYrB4U.js +0 -8
- package/server/piecemaker/vendor/websocket-server/case-instructions.cjs +0 -111
- package/server/piecemaker/vendor/websocket-server/scripts/ADDRESS_ANONYMIZATION.md +0 -216
- package/server/piecemaker/vendor/websocket-server/scripts/DEDUPLICATION_CHANGES.md +0 -134
- package/server/piecemaker/vendor/websocket-server/scripts/GLINER2_README.md +0 -138
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/AUDIT_PERF_QUALITE.md +0 -246
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/MODELE_COREML.md +0 -62
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/__pycache__/coreml_runtime.cpython-312.pyc +0 -0
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/build_coreml.py +0 -118
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/coreml_runtime.py +0 -164
- package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/presidio-gliner.py +0 -479
package/server/piecemaker/vendor/websocket-server/scripts/presidio-gliner/presidio-gliner.py
DELETED
|
@@ -1,479 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Presidio-GLiNER2.5 PII Scanner — scans a Markdown file for sensitive data using
|
|
3
|
-
presidio-analyzer with a custom GLiNER2 recognizer (gliner2 pip package) and
|
|
4
|
-
the fastino/gliner2.5-multi-v1 boundary model.
|
|
5
|
-
|
|
6
|
-
Usage:
|
|
7
|
-
python presidio-gliner.py <md_file> -o <output_dir>
|
|
8
|
-
|
|
9
|
-
Pip deps:
|
|
10
|
-
presidio-analyzer>=2.2.0
|
|
11
|
-
gliner2[local]>=2.0.0
|
|
12
|
-
spacy>=3.7.0
|
|
13
|
-
# + spaCy models: fr_core_news_sm, en_core_web_sm
|
|
14
|
-
|
|
15
|
-
Language is auto-detected (fr/en) from the document content.
|
|
16
|
-
"""
|
|
17
|
-
|
|
18
|
-
import argparse
|
|
19
|
-
import json
|
|
20
|
-
import os
|
|
21
|
-
import sys
|
|
22
|
-
import warnings
|
|
23
|
-
from collections import defaultdict
|
|
24
|
-
from typing import Dict, List, Optional
|
|
25
|
-
|
|
26
|
-
# ---------------------------------------------------------------------------
|
|
27
|
-
# Import from the *real* installed presidio-analyzer BEFORE adding the parent
|
|
28
|
-
# dir to sys.path (the parent dir contains a vendored presidio_analyzer subset
|
|
29
|
-
# that would shadow the real package).
|
|
30
|
-
# ---------------------------------------------------------------------------
|
|
31
|
-
from presidio_analyzer import (
|
|
32
|
-
AnalysisExplanation,
|
|
33
|
-
AnalyzerEngine,
|
|
34
|
-
LocalRecognizer,
|
|
35
|
-
RecognizerResult,
|
|
36
|
-
)
|
|
37
|
-
from presidio_analyzer.nlp_engine import NlpArtifacts, NlpEngineProvider
|
|
38
|
-
|
|
39
|
-
try:
|
|
40
|
-
from gliner2 import AutoExtractor
|
|
41
|
-
|
|
42
|
-
GLINER2_AVAILABLE = True
|
|
43
|
-
except ImportError:
|
|
44
|
-
GLINER2_AVAILABLE = False
|
|
45
|
-
|
|
46
|
-
from model_config import PREFERRED_GLINER_MODEL
|
|
47
|
-
|
|
48
|
-
# ---------------------------------------------------------------------------
|
|
49
|
-
# Parent dir on sys.path so we can import scan_utils.
|
|
50
|
-
# scan_utils imports from a *vendored* presidio_analyzer subset that lives in
|
|
51
|
-
# the parent dir. The real package is already loaded above, so we temporarily
|
|
52
|
-
# hide it from sys.modules so the vendored copy is picked up by scan_utils,
|
|
53
|
-
# then restore the real modules afterwards.
|
|
54
|
-
# ---------------------------------------------------------------------------
|
|
55
|
-
_PARENT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
56
|
-
sys.path.insert(0, _PARENT_DIR)
|
|
57
|
-
|
|
58
|
-
_real_presidio_mods = {
|
|
59
|
-
k: sys.modules.pop(k)
|
|
60
|
-
for k in list(sys.modules)
|
|
61
|
-
if k.startswith("presidio_analyzer")
|
|
62
|
-
}
|
|
63
|
-
|
|
64
|
-
from scan_utils import ( # noqa: E402
|
|
65
|
-
build_output_payload,
|
|
66
|
-
build_pattern_recognizers,
|
|
67
|
-
detect_language,
|
|
68
|
-
extract_legal_form,
|
|
69
|
-
normalize_entity_text,
|
|
70
|
-
print_summary,
|
|
71
|
-
resolve_overlapping_spans,
|
|
72
|
-
run_pattern_recognizers,
|
|
73
|
-
validate_md_input,
|
|
74
|
-
)
|
|
75
|
-
|
|
76
|
-
sys.modules.update(_real_presidio_mods)
|
|
77
|
-
|
|
78
|
-
# Lives next to this file, not in the scripts/ directory added above.
|
|
79
|
-
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
80
|
-
import coreml_runtime # noqa: E402
|
|
81
|
-
|
|
82
|
-
# See scanner_worker.py for the measurements behind both settings.
|
|
83
|
-
try:
|
|
84
|
-
import torch
|
|
85
|
-
|
|
86
|
-
torch.set_num_threads(int(os.environ.get("PIECEMAKER_TORCH_THREADS", "6")))
|
|
87
|
-
except ImportError: # pragma: no cover - torch ships with gliner2
|
|
88
|
-
pass
|
|
89
|
-
|
|
90
|
-
DEBUG_ENTITIES = os.environ.get("PIECEMAKER_DEBUG_ENTITIES", "").lower() in ("1", "true", "yes")
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
# ---------------------------------------------------------------------------
|
|
94
|
-
# Constants
|
|
95
|
-
# ---------------------------------------------------------------------------
|
|
96
|
-
GLINER_MODEL = PREFERRED_GLINER_MODEL
|
|
97
|
-
|
|
98
|
-
ENTITY_MAPPING = {
|
|
99
|
-
"person": "PERSON",
|
|
100
|
-
"company": "ORGANIZATION", # explicit commercial entity
|
|
101
|
-
"organization": "ORGANIZATION", # non-profit / institutional
|
|
102
|
-
"location": "LOCATION",
|
|
103
|
-
}
|
|
104
|
-
|
|
105
|
-
ENTITY_DESCRIPTIONS = {
|
|
106
|
-
# Wording measured against the reference corpus: spelling out what each label
|
|
107
|
-
# EXCLUDES raises precision from 0.485 to 0.648 at equal (perfect) recall and
|
|
108
|
-
# nearly halves the number of words a false positive would rewrite. The previous
|
|
109
|
-
# one-line descriptions left the exclusions implicit and the model read job titles
|
|
110
|
-
# as people, generic bodies as organisations and nationalities as places.
|
|
111
|
-
"person": (
|
|
112
|
-
"Full name of a specific individual human being, such as Jean Dupont or "
|
|
113
|
-
"Mrs Laurence Vidal. Never a job title, never a role, never an acronym, "
|
|
114
|
-
"never a gene or a product name"
|
|
115
|
-
),
|
|
116
|
-
"company": (
|
|
117
|
-
"Name of a specific named commercial company, such as Novartis or Investis "
|
|
118
|
-
"Partners SAS. Never a generic word like company, group or shareholders"
|
|
119
|
-
),
|
|
120
|
-
"organization": (
|
|
121
|
-
"Name of a specific named institution, agency or regulator, such as FDA, "
|
|
122
|
-
"EMA or Inserm. Never a generic body such as board of directors, "
|
|
123
|
-
"committee or working group"
|
|
124
|
-
),
|
|
125
|
-
"location": (
|
|
126
|
-
"Name of a specific geographic place: a country, a city, a region or a postal "
|
|
127
|
-
"address. Never a nationality adjective such as French or European, never an "
|
|
128
|
-
"anatomical part"
|
|
129
|
-
),
|
|
130
|
-
}
|
|
131
|
-
|
|
132
|
-
# GLiNER2.5 has a different boundary head and its scores are not calibrated like
|
|
133
|
-
# those of the former span checkpoint. Start from Fastino's documented default;
|
|
134
|
-
# anonymisation favours recall, and the mapping remains reviewable for false positives.
|
|
135
|
-
GLINER_THRESHOLD = 0.5
|
|
136
|
-
|
|
137
|
-
# Legal forms are read literally from the text by scan_utils.extract_legal_form;
|
|
138
|
-
# the previous 30-label classify_text schema is gone (see scanner_worker.py).
|
|
139
|
-
|
|
140
|
-
# Réglage officiel Fastino pour les documents longs GLiNER2.5 : 384 mots avec
|
|
141
|
-
# 64 mots de recouvrement. Ces valeurs doivent rester identiques dans
|
|
142
|
-
# scanner_worker.py, qui est le point d'entrée persistant du scanner.
|
|
143
|
-
CHUNK_SIZE = 384 # words per chunk
|
|
144
|
-
CHUNK_OVERLAP = 64 # words of overlap between chunks
|
|
145
|
-
BATCH_SIZE = 8 # number of chunks to process in parallel
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
# ---------------------------------------------------------------------------
|
|
149
|
-
# Chunking utility
|
|
150
|
-
# ---------------------------------------------------------------------------
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
def chunk_text_by_words(
|
|
154
|
-
text: str,
|
|
155
|
-
max_words: int = CHUNK_SIZE,
|
|
156
|
-
overlap: int = CHUNK_OVERLAP,
|
|
157
|
-
) -> List[Dict]:
|
|
158
|
-
"""
|
|
159
|
-
Split text into overlapping chunks based on word count, preserving
|
|
160
|
-
original char offsets.
|
|
161
|
-
|
|
162
|
-
Uses regex to locate every word boundary in the original text so that
|
|
163
|
-
start_char / end_char are always exact offsets and original whitespace
|
|
164
|
-
(tabs, newlines, multiple spaces) is preserved inside each chunk.
|
|
165
|
-
"""
|
|
166
|
-
import re
|
|
167
|
-
|
|
168
|
-
word_spans = [(m.start(), m.end()) for m in re.finditer(r"\S+", text)]
|
|
169
|
-
if not word_spans:
|
|
170
|
-
return []
|
|
171
|
-
|
|
172
|
-
step = max(1, max_words - overlap)
|
|
173
|
-
chunks = []
|
|
174
|
-
for i in range(0, len(word_spans), step):
|
|
175
|
-
span_group = word_spans[i : i + max_words]
|
|
176
|
-
start_char = span_group[0][0]
|
|
177
|
-
end_char = span_group[-1][1]
|
|
178
|
-
|
|
179
|
-
chunks.append(
|
|
180
|
-
{
|
|
181
|
-
"text": text[start_char:end_char],
|
|
182
|
-
"start_char": start_char,
|
|
183
|
-
"end_char": end_char,
|
|
184
|
-
}
|
|
185
|
-
)
|
|
186
|
-
|
|
187
|
-
# Stop if this chunk already reached the end
|
|
188
|
-
if i + max_words >= len(word_spans):
|
|
189
|
-
break
|
|
190
|
-
|
|
191
|
-
return chunks
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
# ---------------------------------------------------------------------------
|
|
195
|
-
# Custom GLiNER2 Presidio recognizer
|
|
196
|
-
# ---------------------------------------------------------------------------
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
class GLiNER2Recognizer(LocalRecognizer):
|
|
200
|
-
"""Presidio recognizer wrapping the gliner2 package with batch processing."""
|
|
201
|
-
|
|
202
|
-
def __init__(
|
|
203
|
-
self,
|
|
204
|
-
model_name: str = GLINER_MODEL,
|
|
205
|
-
entity_mapping: Optional[Dict[str, str]] = None,
|
|
206
|
-
supported_language: str = "en",
|
|
207
|
-
threshold: float = GLINER_THRESHOLD,
|
|
208
|
-
batch_size: int = BATCH_SIZE,
|
|
209
|
-
):
|
|
210
|
-
self.model_name = model_name
|
|
211
|
-
self.model_to_presidio = entity_mapping or ENTITY_MAPPING
|
|
212
|
-
self.gliner_labels = list(self.model_to_presidio.keys())
|
|
213
|
-
self.threshold = threshold
|
|
214
|
-
self.batch_size = batch_size
|
|
215
|
-
self.model = None
|
|
216
|
-
|
|
217
|
-
supported_entities = list(set(self.model_to_presidio.values()))
|
|
218
|
-
|
|
219
|
-
super().__init__(
|
|
220
|
-
supported_entities=supported_entities,
|
|
221
|
-
name="GLiNER2Recognizer",
|
|
222
|
-
supported_language=supported_language,
|
|
223
|
-
)
|
|
224
|
-
|
|
225
|
-
def load(self) -> None:
|
|
226
|
-
if not GLINER2_AVAILABLE:
|
|
227
|
-
raise ImportError("gliner2 is not installed.")
|
|
228
|
-
self.model = AutoExtractor.from_pretrained(
|
|
229
|
-
self.model_name,
|
|
230
|
-
local_files_only=True,
|
|
231
|
-
)
|
|
232
|
-
# Same CoreML/GPU encoder as the worker (~2x, identical output). Falls back to
|
|
233
|
-
# torch on its own when coremltools or the compiled model is absent.
|
|
234
|
-
coreml_runtime.maybe_accelerate(self.model, self.model_name)
|
|
235
|
-
|
|
236
|
-
def analyze(
|
|
237
|
-
self,
|
|
238
|
-
text: str,
|
|
239
|
-
entities: List[str],
|
|
240
|
-
nlp_artifacts: Optional[NlpArtifacts] = None,
|
|
241
|
-
) -> List[RecognizerResult]:
|
|
242
|
-
if self.model is None:
|
|
243
|
-
self.load()
|
|
244
|
-
|
|
245
|
-
# Chunk the text
|
|
246
|
-
chunks = chunk_text_by_words(text, CHUNK_SIZE)
|
|
247
|
-
total_chunks = len(chunks)
|
|
248
|
-
print(f"⚡ Processing {total_chunks} chunks in batches of {self.batch_size}...")
|
|
249
|
-
|
|
250
|
-
# Extract batch texts
|
|
251
|
-
batch_texts = [chunk["text"] for chunk in chunks]
|
|
252
|
-
|
|
253
|
-
# Process one batch at a time so PROGRESS:CHUNKS is emitted as each
|
|
254
|
-
# batch actually finishes inference (real-time, not post-hoc).
|
|
255
|
-
batch_results = []
|
|
256
|
-
for batch_start in range(0, total_chunks, self.batch_size):
|
|
257
|
-
batch_end = min(batch_start + self.batch_size, total_chunks)
|
|
258
|
-
slice_results = self.model.batch_extract_entities(
|
|
259
|
-
batch_texts[batch_start:batch_end],
|
|
260
|
-
ENTITY_DESCRIPTIONS,
|
|
261
|
-
batch_size=self.batch_size,
|
|
262
|
-
threshold=self.threshold,
|
|
263
|
-
include_confidence=True,
|
|
264
|
-
include_spans=True,
|
|
265
|
-
)
|
|
266
|
-
batch_results.extend(slice_results)
|
|
267
|
-
pct = int(batch_end * 100 / total_chunks)
|
|
268
|
-
print(f"PROGRESS:CHUNKS:{pct}:{batch_end}:{total_chunks}", flush=True)
|
|
269
|
-
|
|
270
|
-
print(f"✓ Completed processing all {len(batch_results)} chunks")
|
|
271
|
-
|
|
272
|
-
# Deduplicate entities across chunks using text-based set
|
|
273
|
-
unique_entities = defaultdict(lambda: defaultdict(dict))
|
|
274
|
-
results = []
|
|
275
|
-
|
|
276
|
-
for chunk_idx, (chunk, batch_result) in enumerate(zip(chunks, batch_results)):
|
|
277
|
-
for label, matches in batch_result.get("entities", {}).items():
|
|
278
|
-
presidio_type = self.model_to_presidio.get(label, label.upper())
|
|
279
|
-
if entities and presidio_type not in entities:
|
|
280
|
-
continue
|
|
281
|
-
|
|
282
|
-
for match in matches:
|
|
283
|
-
# Calculate absolute positions
|
|
284
|
-
abs_start = chunk["start_char"] + match["start"]
|
|
285
|
-
abs_end = chunk["start_char"] + match["end"]
|
|
286
|
-
entity_text = match["text"]
|
|
287
|
-
|
|
288
|
-
# Create unique key for deduplication
|
|
289
|
-
entity_key = (entity_text, abs_start, abs_end)
|
|
290
|
-
|
|
291
|
-
# Keep only the highest confidence score for each unique entity
|
|
292
|
-
if entity_key not in unique_entities[presidio_type]:
|
|
293
|
-
unique_entities[presidio_type][entity_key] = {
|
|
294
|
-
"text": entity_text,
|
|
295
|
-
"start": abs_start,
|
|
296
|
-
"end": abs_end,
|
|
297
|
-
"score": match["confidence"],
|
|
298
|
-
}
|
|
299
|
-
else:
|
|
300
|
-
# Update if higher confidence
|
|
301
|
-
if (
|
|
302
|
-
match["confidence"]
|
|
303
|
-
> unique_entities[presidio_type][entity_key]["score"]
|
|
304
|
-
):
|
|
305
|
-
unique_entities[presidio_type][entity_key]["score"] = match[
|
|
306
|
-
"confidence"
|
|
307
|
-
]
|
|
308
|
-
|
|
309
|
-
# Read each ORGANIZATION's legal form straight out of the text, memoised
|
|
310
|
-
# by normalised name (entity keys are per-occurrence).
|
|
311
|
-
form_cache = {}
|
|
312
|
-
for entity_data in unique_entities.get("ORGANIZATION", {}).values():
|
|
313
|
-
name = normalize_entity_text(entity_data["text"])
|
|
314
|
-
if name not in form_cache:
|
|
315
|
-
trailing = text[entity_data["end"]:entity_data["end"] + 40]
|
|
316
|
-
form_cache[name] = extract_legal_form(name, trailing)
|
|
317
|
-
form, nationality = form_cache[name]
|
|
318
|
-
|
|
319
|
-
entity_data["presidio_type"] = f"ORGANIZATION_{form}" if form else "ORGANIZATION"
|
|
320
|
-
entity_data["nationality"] = nationality
|
|
321
|
-
|
|
322
|
-
# Convert deduplicated entities to RecognizerResult objects
|
|
323
|
-
for presidio_type, entity_dict in unique_entities.items():
|
|
324
|
-
for entity_data in entity_dict.values():
|
|
325
|
-
final_type = entity_data.get("presidio_type", presidio_type)
|
|
326
|
-
results.append(
|
|
327
|
-
RecognizerResult(
|
|
328
|
-
entity_type=final_type,
|
|
329
|
-
start=entity_data["start"],
|
|
330
|
-
end=entity_data["end"],
|
|
331
|
-
score=entity_data["score"],
|
|
332
|
-
analysis_explanation=AnalysisExplanation(
|
|
333
|
-
recognizer=self.name,
|
|
334
|
-
original_score=entity_data["score"],
|
|
335
|
-
textual_explanation=f"GLiNER2 batch processing",
|
|
336
|
-
),
|
|
337
|
-
)
|
|
338
|
-
)
|
|
339
|
-
|
|
340
|
-
return results
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
# ---------------------------------------------------------------------------
|
|
344
|
-
# Engine setup
|
|
345
|
-
# ---------------------------------------------------------------------------
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
def _build_analyzer(language: str):
|
|
349
|
-
"""Create an AnalyzerEngine with spaCy NLP + GLiNER2 recognizer."""
|
|
350
|
-
lang_models = {
|
|
351
|
-
"fr": "fr_core_news_sm",
|
|
352
|
-
"en": "en_core_web_sm",
|
|
353
|
-
}
|
|
354
|
-
model_name = lang_models.get(language, "en_core_web_sm")
|
|
355
|
-
|
|
356
|
-
nlp_provider = NlpEngineProvider(
|
|
357
|
-
nlp_configuration={
|
|
358
|
-
"nlp_engine_name": "spacy",
|
|
359
|
-
"models": [{"lang_code": language, "model_name": model_name}],
|
|
360
|
-
}
|
|
361
|
-
)
|
|
362
|
-
nlp_engine = nlp_provider.create_engine()
|
|
363
|
-
# Raise spaCy max_length so large documents don't get rejected.
|
|
364
|
-
# NER is handled by GLiNER2 in chunks, spaCy is only used for tokenization.
|
|
365
|
-
nlp_engine.nlp[language].max_length = 5_000_000
|
|
366
|
-
|
|
367
|
-
# ...so disable every other pipe. presidio loads the full pipeline by default and
|
|
368
|
-
# SpacyRecognizer is removed below, meaning tagger/parser/lemmatizer/ner all run
|
|
369
|
-
# over the document and their output is discarded (20.6 s vs 0.9 s per 300k chars).
|
|
370
|
-
_nlp = nlp_engine.nlp.get(language)
|
|
371
|
-
if _nlp is not None:
|
|
372
|
-
for _pipe_name in list(_nlp.pipe_names):
|
|
373
|
-
try:
|
|
374
|
-
_nlp.disable_pipe(_pipe_name)
|
|
375
|
-
except Exception as exc: # noqa: BLE001
|
|
376
|
-
warnings.warn(f"Could not disable spaCy pipe '{_pipe_name}': {exc}", stacklevel=1)
|
|
377
|
-
|
|
378
|
-
analyzer = AnalyzerEngine(
|
|
379
|
-
nlp_engine=nlp_engine,
|
|
380
|
-
supported_languages=[language],
|
|
381
|
-
)
|
|
382
|
-
|
|
383
|
-
gliner_recognizer = GLiNER2Recognizer(
|
|
384
|
-
model_name=GLINER_MODEL,
|
|
385
|
-
entity_mapping=ENTITY_MAPPING,
|
|
386
|
-
supported_language=language,
|
|
387
|
-
batch_size=BATCH_SIZE,
|
|
388
|
-
)
|
|
389
|
-
analyzer.registry.add_recognizer(gliner_recognizer)
|
|
390
|
-
analyzer.registry.remove_recognizer("SpacyRecognizer")
|
|
391
|
-
|
|
392
|
-
return analyzer
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
# ---------------------------------------------------------------------------
|
|
396
|
-
# Main
|
|
397
|
-
# ---------------------------------------------------------------------------
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
def main():
|
|
401
|
-
parser = argparse.ArgumentParser(
|
|
402
|
-
description="Presidio-GLiNER2 PII Scanner for .md files"
|
|
403
|
-
)
|
|
404
|
-
parser.add_argument("md_file", help="Path to the Markdown file to scan")
|
|
405
|
-
parser.add_argument(
|
|
406
|
-
"-o",
|
|
407
|
-
"--output",
|
|
408
|
-
default="./presidio_gliner_output",
|
|
409
|
-
help="Output directory for the JSON entity map",
|
|
410
|
-
)
|
|
411
|
-
args = parser.parse_args()
|
|
412
|
-
|
|
413
|
-
if not validate_md_input(args.md_file):
|
|
414
|
-
return 1
|
|
415
|
-
|
|
416
|
-
os.makedirs(args.output, exist_ok=True)
|
|
417
|
-
|
|
418
|
-
with open(args.md_file, "r", encoding="utf-8") as fh:
|
|
419
|
-
text = fh.read()
|
|
420
|
-
|
|
421
|
-
detected_lang = detect_language(text)
|
|
422
|
-
print(f"✓ Detected language: {detected_lang}")
|
|
423
|
-
|
|
424
|
-
pattern_recognizers = build_pattern_recognizers()
|
|
425
|
-
all_results = run_pattern_recognizers(text, pattern_recognizers)
|
|
426
|
-
|
|
427
|
-
extra_summary = {}
|
|
428
|
-
|
|
429
|
-
print("\U0001f50d Running NER with Presidio + GLiNER2...")
|
|
430
|
-
try:
|
|
431
|
-
analyzer = _build_analyzer(detected_lang)
|
|
432
|
-
ner_results = analyzer.analyze(
|
|
433
|
-
text=text,
|
|
434
|
-
language=detected_lang,
|
|
435
|
-
entities=["PERSON", "ORGANIZATION", "LOCATION"],
|
|
436
|
-
return_decision_process=False,
|
|
437
|
-
)
|
|
438
|
-
except Exception as exc: # noqa: BLE001
|
|
439
|
-
print(
|
|
440
|
-
f"❌ NER GLiNER2 indisponible, scan abandonné : {exc}",
|
|
441
|
-
file=sys.stderr,
|
|
442
|
-
)
|
|
443
|
-
print(
|
|
444
|
-
f" Interpréteur : {sys.executable}",
|
|
445
|
-
file=sys.stderr,
|
|
446
|
-
)
|
|
447
|
-
return 1
|
|
448
|
-
|
|
449
|
-
# Re-include ORGANIZATION_* results that classify_text may have re-typed;
|
|
450
|
-
# Presidio's engine only checks supported_entities at dispatch time, not on
|
|
451
|
-
# results, so ORGANIZATION_SA / ORGANIZATION_GMBH etc. pass through as-is.
|
|
452
|
-
print(f"✓ Presidio-GLiNER2 NER complete ({len(ner_results)} unique entities)")
|
|
453
|
-
for r in ner_results:
|
|
454
|
-
if DEBUG_ENTITIES:
|
|
455
|
-
print(f" [RAW] {r.entity_type}: \"{text[r.start:r.end]}\" score={r.score:.3f} ({r.start}-{r.end})")
|
|
456
|
-
else:
|
|
457
|
-
print(f" [RAW] {r.entity_type}: score={r.score:.3f} ({r.start}-{r.end})")
|
|
458
|
-
extra_summary["ner_engine"] = "presidio-gliner2"
|
|
459
|
-
all_results.extend(ner_results)
|
|
460
|
-
|
|
461
|
-
# Trust GLiNER scores as-is; arbitrate overlapping spans across types (presidio's
|
|
462
|
-
# remove_duplicates only handles same-type containment).
|
|
463
|
-
all_results = resolve_overlapping_spans(all_results)
|
|
464
|
-
|
|
465
|
-
payload = build_output_payload(all_results, text, args.md_file, extra_summary)
|
|
466
|
-
|
|
467
|
-
stem = os.path.splitext(os.path.basename(args.md_file))[0]
|
|
468
|
-
output_path = os.path.join(args.output, f"{stem}_sensitive_map.json")
|
|
469
|
-
|
|
470
|
-
with open(output_path, "w", encoding="utf-8") as fh:
|
|
471
|
-
json.dump(payload, fh, indent=2, ensure_ascii=False)
|
|
472
|
-
|
|
473
|
-
print_summary(payload["entities"], args.md_file, output_path)
|
|
474
|
-
|
|
475
|
-
return 0
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
if __name__ == "__main__":
|
|
479
|
-
sys.exit(main())
|