@piecemaker-legal/piecemaker 2.0.15 → 2.0.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{FindReplaceDialog-KU5TEIBJ-Cp2ZFN7K.js → FindReplaceDialog-KU5TEIBJ-B4N6W1_z.js} +1 -1
- package/dist/assets/{FootnotePropertiesDialog-W3YHM4BV-pWCpugJN.js → FootnotePropertiesDialog-W3YHM4BV-D_X_JkAZ.js} +1 -1
- package/dist/assets/{HyperlinkDialog-ZYH7ZHRF-CbCWgVs6.js → HyperlinkDialog-ZYH7ZHRF-EbyukgDx.js} +4 -4
- package/dist/assets/{ImagePositionDialog-HEWYR6Q3-Du2r0Wko.js → ImagePositionDialog-HEWYR6Q3-D9jEaxlE.js} +1 -1
- package/dist/assets/{ImagePropertiesDialog-RAB47C6M-DLpeqGpW.js → ImagePropertiesDialog-RAB47C6M-DvA1kZnM.js} +1 -1
- package/dist/assets/{PageSetupDialog-AIGIXO3E-B_CipGFt.js → PageSetupDialog-AIGIXO3E-ChfsKF43.js} +1 -1
- package/dist/assets/{SplitCellDialog-IFBKYCD6-xMFwbIsB.js → SplitCellDialog-IFBKYCD6-uM26FNXt.js} +1 -1
- package/dist/assets/{TablePropertiesDialog-TMLKGOOW-Fz1cRdQ9.js → TablePropertiesDialog-TMLKGOOW-By6EQI5w.js} +1 -1
- package/dist/assets/{WatermarkDialog-KRL7WMIS-7suU5cvL.js → WatermarkDialog-KRL7WMIS-Bss9F6Ej.js} +1 -1
- package/dist/assets/{abnfDiagram-VCTEODGH-BKCg5tMI.js → abnfDiagram-VCTEODGH-yLjNPHY7.js} +1 -1
- package/dist/assets/{arc-BxB5WTHo.js → arc-CMQiMSh4.js} +1 -1
- package/dist/assets/{architectureDiagram-5GKGNRK7-CM1c0ORG.js → architectureDiagram-5GKGNRK7-Cyna8IJZ.js} +1 -1
- package/dist/assets/{blockDiagram-I7D4REHJ-BH0vcN7C.js → blockDiagram-I7D4REHJ-DHAdjRYY.js} +1 -1
- package/dist/assets/{c4Diagram-7LVT6UL2-Z9_RzazA.js → c4Diagram-7LVT6UL2-qqZ0w4LC.js} +1 -1
- package/dist/assets/channel-B3AYuE-l.js +1 -0
- package/dist/assets/{chunk-2Q5K7J3B-OQRYTAt1.js → chunk-2Q5K7J3B-NS0halN6.js} +1 -1
- package/dist/assets/{chunk-5VM5RSS4-DBpdDz0k.js → chunk-5VM5RSS4-BF6-6Cah.js} +1 -1
- package/dist/assets/{chunk-F27PBJKO-npTlDSfn.js → chunk-F27PBJKO-Dnj2jFRB.js} +1 -1
- package/dist/assets/{chunk-IMKFNOWR-Cxrz4ZaK.js → chunk-IMKFNOWR-CmRgl1EE.js} +1 -1
- package/dist/assets/{chunk-JWPE2WC7-DK86pmtC.js → chunk-JWPE2WC7-BwC19UFc.js} +1 -1
- package/dist/assets/{chunk-POPQ4Y6H-uVCWOMNU.js → chunk-POPQ4Y6H-C1YWVD4-.js} +1 -1
- package/dist/assets/{chunk-SVP7TREG-hdcvsO4Z.js → chunk-SVP7TREG-D7XQTPRq.js} +1 -1
- package/dist/assets/{chunk-TICWLB2K-DQ2F5I7F.js → chunk-TICWLB2K-BkmTGk1O.js} +1 -1
- package/dist/assets/{chunk-XXDRQBXY-oLFixl7-.js → chunk-XXDRQBXY-CYW7YgXL.js} +1 -1
- package/dist/assets/classDiagram-ZZMXUADV-CE2Zkr7h.js +1 -0
- package/dist/assets/classDiagram-v2-VYDZK3BY-CE2Zkr7h.js +1 -0
- package/dist/assets/{cose-bilkent-JH36ORCC-b32g5AYD.js → cose-bilkent-JH36ORCC-BSktt6pW.js} +1 -1
- package/dist/assets/{cynefin-OW5HDTMX-CvzWSkfP.js → cynefin-OW5HDTMX-BNMRWXNG.js} +1 -1
- package/dist/assets/{cynefinDiagram-5FMLGOSQ-eNSRssDp.js → cynefinDiagram-5FMLGOSQ-Dk59YHuO.js} +1 -1
- package/dist/assets/{dagre-GXQ25YYZ-ByshmYfw.js → dagre-GXQ25YYZ-DFhWeh0x.js} +1 -1
- package/dist/assets/{diagram-S7CK7UJ4-DM734w6p.js → diagram-S7CK7UJ4-BH_8x-q5.js} +1 -1
- package/dist/assets/{diagram-UQ7AKVKN-CeIgNHVI.js → diagram-UQ7AKVKN-DmxmlXg5.js} +1 -1
- package/dist/assets/{diagram-VSXAHHWV-BCET7ibB.js → diagram-VSXAHHWV-DpG29cD4.js} +1 -1
- package/dist/assets/{diagram-VX7I27RA-BWH0JZL-.js → diagram-VX7I27RA-CvvYXTie.js} +1 -1
- package/dist/assets/{diagram-Z3DM3KII-C6BrIsgm.js → diagram-Z3DM3KII-xOXaaZqs.js} +1 -1
- package/dist/assets/{ebnfDiagram-PWID7BFC-B_n61ePe.js → ebnfDiagram-PWID7BFC-CtfFGqJB.js} +1 -1
- package/dist/assets/{erDiagram-RLTQ6QDP-BZJ1h673.js → erDiagram-RLTQ6QDP-AoCjoHUY.js} +1 -1
- package/dist/assets/{flowDiagram-HODETNUW-cymoZryb.js → flowDiagram-HODETNUW-Bl9_mq95.js} +1 -1
- package/dist/assets/{ganttDiagram-EL5Y4UJY-CPr5f3_S.js → ganttDiagram-EL5Y4UJY-DREJQ2BY.js} +1 -1
- package/dist/assets/{gitGraphDiagram-WWUBYQGX-S3PeYZkU.js → gitGraphDiagram-WWUBYQGX-C1q5-iep.js} +1 -1
- package/dist/assets/index-C1854KB7.css +1 -0
- package/dist/assets/{index-KoMJ5vMA.js → index-RKNvOXcq.js} +140 -140
- package/dist/assets/{infoDiagram-27XIBGKW-CHua8hTk.js → infoDiagram-27XIBGKW-Cz2KUKWU.js} +1 -1
- package/dist/assets/{ishikawaDiagram-5VMMS53U-BMPmgP2f.js → ishikawaDiagram-5VMMS53U-BU6SdWsw.js} +1 -1
- package/dist/assets/{journeyDiagram-3NMN7TZE-Dx8ezs5g.js → journeyDiagram-3NMN7TZE-CMMoiPf0.js} +1 -1
- package/dist/assets/{kanban-definition-UXKFOSKX-CnGlA00D.js → kanban-definition-UXKFOSKX-KpKnC3dd.js} +1 -1
- package/dist/assets/{layout-DdxskZT3.js → layout-6C8d-KNC.js} +1 -1
- package/dist/assets/{linear-CfdU6qlL.js → linear-CUINCO59.js} +1 -1
- package/dist/assets/{mermaid.core-D2KvaDg4.js → mermaid.core-ClgFTKfs.js} +6 -6
- package/dist/assets/{mindmap-definition-YA3MSWOX-CMoIPKDH.js → mindmap-definition-YA3MSWOX-oywqmpy6.js} +1 -1
- package/dist/assets/{pegDiagram-XKGWAZYB-DcemvKop.js → pegDiagram-XKGWAZYB-Uk65qoVL.js} +1 -1
- package/dist/assets/{pieDiagram-E7YTZNPT-BZEYiM_X.js → pieDiagram-E7YTZNPT-XxQhMRKu.js} +1 -1
- package/dist/assets/{quadrantDiagram-AXDQQJYC-CiAFc6a4.js → quadrantDiagram-AXDQQJYC-Djngo9IG.js} +1 -1
- package/dist/assets/{railroadDiagram-O6MQD6OU-DLmf01wa.js → railroadDiagram-O6MQD6OU-C4I8SNPI.js} +1 -1
- package/dist/assets/{requirementDiagram-BXWQKSXE-BMFvgV-y.js → requirementDiagram-BXWQKSXE-D62oiVFl.js} +1 -1
- package/dist/assets/{sankeyDiagram-P5KCCOFB-B4fcaIgK.js → sankeyDiagram-P5KCCOFB-HZuSub58.js} +1 -1
- package/dist/assets/{sequenceDiagram-WJ2MYXX4-BHpaOBPQ.js → sequenceDiagram-WJ2MYXX4-B8_t7SDx.js} +1 -1
- package/dist/assets/{sizeCapture-INFHLROL-BsOv244V.js → sizeCapture-INFHLROL-D2Lk0zlm.js} +1 -1
- package/dist/assets/{stateDiagram-D77RDMKH-DM2-SXRt.js → stateDiagram-D77RDMKH-DC3NnV2S.js} +1 -1
- package/dist/assets/{stateDiagram-v2-MP3YSRHH-DIJgYrCw.js → stateDiagram-v2-MP3YSRHH-DSsQOGCi.js} +1 -1
- package/dist/assets/{swimlanes-42K2YHIH-CZmZpYra.js → swimlanes-42K2YHIH-6RzklsF9.js} +1 -1
- package/dist/assets/swimlanesDiagram-VR7AAH4N-ZU4NRSFy.js +8 -0
- package/dist/assets/{timeline-definition-24CTP7MA-D1ckb3Ke.js → timeline-definition-24CTP7MA-Beuacf_x.js} +1 -1
- package/dist/assets/{vennDiagram-4TSXK5OY-CgU7Doxj.js → vennDiagram-4TSXK5OY-CMqthbQr.js} +1 -1
- package/dist/assets/{wardleyDiagram-VM6X3IG4-BACRuQOS.js → wardleyDiagram-VM6X3IG4-DKrdN08q.js} +1 -1
- package/dist/assets/{xychartDiagram-S5SC5T6Z-D8sG0Qsi.js → xychartDiagram-S5SC5T6Z-BtMW2Qlh.js} +1 -1
- package/dist/index.html +2 -2
- package/dist/sw.js +1 -1
- package/dist-server/server/piecemaker/desktop-update/index.js +3 -2
- package/dist-server/server/piecemaker/desktop-update/index.js.map +1 -1
- package/dist-server/server/piecemaker/harness/citation-instructions.js +2 -2
- package/dist-server/server/piecemaker/harness/citation-store.js +1 -1
- package/dist-server/server/piecemaker/harness/citation-store.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/local-routes.js +1 -1
- package/dist-server/server/piecemaker/knowledge/local-routes.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/pipeline.js +12 -1
- package/dist-server/server/piecemaker/knowledge/pipeline.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/routes.js +1 -1
- package/dist-server/server/piecemaker/knowledge/routes.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/scan-jobs.js +13 -0
- package/dist-server/server/piecemaker/knowledge/scan-jobs.js.map +1 -1
- package/dist-server/server/piecemaker/knowledge/service.js +3 -2
- package/dist-server/server/piecemaker/knowledge/service.js.map +1 -1
- package/package.json +2 -1
- package/scripts/piecemaker/cli/lib/plugins.mjs +28 -30
- package/scripts/piecemaker/cli/piecemaker-command.test.mjs +7 -0
- package/scripts/piecemaker/cli/piecemaker.mjs +9 -3
- package/server/piecemaker/desktop-update/index.ts +3 -2
- package/server/piecemaker/harness/citation-instructions.ts +2 -2
- package/server/piecemaker/harness/citation-store.ts +1 -1
- package/server/piecemaker/knowledge/local-routes.ts +1 -1
- package/server/piecemaker/knowledge/pipeline.ts +20 -2
- package/server/piecemaker/knowledge/routes.ts +1 -1
- package/server/piecemaker/knowledge/scan-jobs.ts +19 -1
- package/server/piecemaker/knowledge/service.ts +3 -2
- package/server/piecemaker/vendor/installer/steps/04-conversion-md.mjs +46 -9
- package/server/piecemaker/vendor/installer/steps/07-legifrance.mjs +3 -3
- package/server/piecemaker/vendor/piecemaker-plugin/README.md +8 -17
- package/server/piecemaker/vendor/piecemaker-plugin/skills/conversion-md/SKILL.md +3 -1
- package/server/piecemaker/vendor/piecemaker-plugin/skills/recherche-juridique/SKILL.md +19 -34
- package/server/piecemaker/vendor/websocket-server/admin-routes.cjs +6 -1
- package/server/piecemaker/vendor/websocket-server/originals-pipeline.cjs +10 -2
- package/server/piecemaker/vendor/websocket-server/scripts/convert_and_scan_pipeline.py +67 -13
- package/server/piecemaker/vendor/websocket-server/scripts/mineru_improved_1.py +17 -5
- package/server/piecemaker/vendor/websocket-server/scripts/smart_converter.py +20 -3
- package/dist/assets/channel-LQJzz1dz.js +0 -1
- package/dist/assets/classDiagram-ZZMXUADV-BaS6lPLe.js +0 -1
- package/dist/assets/classDiagram-v2-VYDZK3BY-BaS6lPLe.js +0 -1
- package/dist/assets/index-C5slQNvE.css +0 -1
- package/dist/assets/swimlanesDiagram-VR7AAH4N-kkl3-vaA.js +0 -8
|
@@ -22,51 +22,37 @@ effectivement disponibles, puis le format exact que le vérificateur accepte.
|
|
|
22
22
|
|
|
23
23
|
| Outil | Rôle |
|
|
24
24
|
| --- | --- |
|
|
25
|
-
| `
|
|
26
|
-
| `Search_Conseil_Etat` | Même recherche ciblée dans la jurisprudence du Conseil d'État, filtrable par publication au recueil Lebon. |
|
|
27
|
-
| `Search_Cour_Appel` | Même recherche dans les cours d'appel, filtrable par ville — bien cibler (ville + dates), le volume est important. |
|
|
28
|
-
| `Search_CAA` | Même recherche dans les cours administratives d'appel, filtrable par ville. |
|
|
29
|
-
| `Search_Premiere_Instance` | Recherche dans les juridictions de première instance — volume très limité (~50 décisions), opérateur OU par défaut. |
|
|
25
|
+
| `Search_Jurisprudence` | Seul outil de recherche jurisprudentielle : interroge en un appel Légifrance et Judilibre pour la Cour de cassation, les cours d'appel, la première instance, le Conseil d'État et les CAA, avec les mêmes filtres (`matiere` obligatoire en cassation, publication, villes, `types_premiere_instance` obligatoire en première instance, recueil Lebon, dates) et la même requête (guillemets, `ET`, `OU`, parenthèses, articles). Les doublons Légifrance/Judilibre sont fusionnés ; au-delà de 500 résultats cumulés, la recherche est refusée. |
|
|
30
26
|
| `Search_Code` | Recherche dans les codes juridiques français (Code civil, Code du travail…) — référence d'article exacte (`L. 1235-3`), mots-clés ou expression exacte. Renvoie les identifiants `LEGIARTI…` des articles trouvés. |
|
|
31
27
|
| `consulter_article` | Texte intégral et vigueur d'une version précise d'article, à partir de son identifiant `LEGIARTI…` rendu par `Search_Code`. |
|
|
32
|
-
| `consulter_decision` | Rapatrie le texte intégral d'une décision à partir de son identifiant (`JURITEXT…`, `CETATEXT…`). **Rapatrie le texte intégral et le met en cache** (voir plus bas) — c'est le moyen normal de lire une décision avant de la citer. |
|
|
33
|
-
| `
|
|
34
|
-
| `Build_Research_Corpus` | Construit un corpus exhaustif et reproductible sur une question de droit : plusieurs requêtes, déduplication, téléchargement et scan du texte intégral de chaque décision. **Rapatrie le texte intégral.** À utiliser pour une recherche large et systématique plutôt qu'un enchaînement manuel de `Search_*`. |
|
|
35
|
-
| `Validate_Research_Cards` | Valide mécaniquement (sans LLM) le résultat de `Build_Research_Corpus` : une fiche par décision, citations confirmées dans le texte intégral, rapport de couverture. |
|
|
28
|
+
| `consulter_decision` | Rapatrie le texte intégral d'une décision à partir de son identifiant Légifrance (`JURITEXT…`, `CETATEXT…`) ou Judilibre (24 caractères hexadécimaux). **Rapatrie le texte intégral et le met en cache** (voir plus bas) — c'est le moyen normal de lire une décision avant de la citer. |
|
|
29
|
+
| `Build_Research_Corpus` | Construit un corpus exhaustif et reproductible sur une question de droit : une formulation, recherchée par le même moteur et avec les mêmes filtres que `Search_Jurisprudence`, déduplication, téléchargement et scan du texte intégral de chaque décision. **Rapatrie le texte intégral.** À utiliser pour une recherche large et systématique plutôt qu'une suite de pages de `Search_Jurisprudence`. |
|
|
36
30
|
| `Tracking_BODACC` | Situation d'une entreprise (procédures collectives) via son SIREN — hors jurisprudence, utile pour qualifier une partie. |
|
|
37
31
|
|
|
38
32
|
### Le fait décisif : seuls deux chemins rendent une citation vérifiable
|
|
39
33
|
|
|
40
34
|
Le vérificateur ne peut confirmer une citation que si le texte intégral de sa
|
|
41
|
-
source est quelque part sur disque
|
|
42
|
-
une décision, **une seule ne rapatrie pas ce texte** :
|
|
35
|
+
source est quelque part sur disque :
|
|
43
36
|
|
|
44
37
|
- **`Build_Research_Corpus`** écrit `decisions.jsonl` (une décision par ligne,
|
|
45
38
|
champ `texte` = texte intégral réel) — vérifiable.
|
|
46
39
|
- **`consulter_decision`** renvoie le texte intégral dans sa réponse d'outil ;
|
|
47
40
|
le hook `decision-cache.mjs` le capte au passage et l'écrit dans
|
|
48
41
|
`~/.piecemaker/decisions/<id>.json` — vérifiable.
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
**Conséquence pratique** : `Download_Query_Results` sert à trier une liste de
|
|
56
|
-
résultats (lire les titres/sommaires, écarter le hors-sujet). Dès qu'une
|
|
57
|
-
décision de cette liste doit être citée, il faut d'abord la lire avec
|
|
58
|
-
`consulter_decision` (ou l'avoir dans un corpus `Build_Research_Corpus`) —
|
|
59
|
-
jamais citer directement depuis `results.json`.
|
|
42
|
+
|
|
43
|
+
**Conséquence pratique** : les résultats de `Search_Jurisprudence` (titre,
|
|
44
|
+
analyse, extraits) servent à trier. Dès qu'une décision doit être citée, il
|
|
45
|
+
faut d'abord la lire avec `consulter_decision` (ou l'avoir dans un corpus
|
|
46
|
+
`Build_Research_Corpus`).
|
|
60
47
|
|
|
61
48
|
## Déroulé de recherche
|
|
62
49
|
|
|
63
|
-
1. **Recherche large** avec
|
|
50
|
+
1. **Recherche large** avec `Search_Jurisprudence` sur les juridictions visées
|
|
64
51
|
(ou `Build_Research_Corpus` pour une question de droit qui mérite un
|
|
65
52
|
balayage systématique plutôt qu'une requête isolée). Pour un texte de loi,
|
|
66
53
|
`Search_Code`.
|
|
67
|
-
2. **Tri** des résultats sur les seuls éléments déjà renvoyés (titre,
|
|
68
|
-
date,
|
|
69
|
-
aide à trier un grand volume hors ligne.
|
|
54
|
+
2. **Tri** des résultats sur les seuls éléments déjà renvoyés (titre, analyse,
|
|
55
|
+
date, extraits) — sans lire chaque décision en entier.
|
|
70
56
|
3. **Lecture** des seules décisions/articles retenus comme potentiellement
|
|
71
57
|
cite-worthy : `consulter_decision` (jurisprudence) ou `consulter_article`
|
|
72
58
|
(texte de loi), ou relecture du `decisions.jsonl` d'un corpus déjà construit.
|
|
@@ -114,9 +100,9 @@ prend l'une des deux formes suivantes.
|
|
|
114
100
|
### Forme « jurisprudence » (`kind: "case"`)
|
|
115
101
|
|
|
116
102
|
Identifie la décision par `decision_id` — l'identifiant Légifrance
|
|
117
|
-
(`JURITEXT…`, `CETATEXT…`)
|
|
118
|
-
`Build_Research_Corpus`, et lu avec
|
|
119
|
-
`decisions.jsonl`).
|
|
103
|
+
(`JURITEXT…`, `CETATEXT…`) ou Judilibre (24 caractères hexadécimaux) tel que
|
|
104
|
+
rendu par `Search_Jurisprudence`/`Build_Research_Corpus`, et lu avec
|
|
105
|
+
`consulter_decision` (ou présent dans un `decisions.jsonl`).
|
|
120
106
|
|
|
121
107
|
```json
|
|
122
108
|
{
|
|
@@ -208,9 +194,8 @@ retombe sur un tableau `quotes` d'un seul élément :
|
|
|
208
194
|
|
|
209
195
|
1. Parse le bloc `<CITATIONS>` du dernier message.
|
|
210
196
|
2. Pour chaque citation, résout sa source réelle — texte intégral de la
|
|
211
|
-
décision (corpus `Build_Research_Corpus
|
|
212
|
-
|
|
213
|
-
Markdown du dossier.
|
|
197
|
+
décision (corpus `Build_Research_Corpus` ou cache `consulter_decision`) ou
|
|
198
|
+
pièce Markdown du dossier.
|
|
214
199
|
3. Localise mécaniquement chaque extrait dans ce texte. Un extrait qui a
|
|
215
200
|
légèrement dérivé (espace, casse, ponctuation) est **corrigé
|
|
216
201
|
automatiquement** avec l'extrait source exact — ce n'est pas une faute.
|
|
@@ -224,8 +209,8 @@ citation si elle ne peut pas être justifiée par une source effectivement lue.
|
|
|
224
209
|
|
|
225
210
|
## Ce qu'il ne faut pas faire
|
|
226
211
|
|
|
227
|
-
- Ne jamais citer une décision connue uniquement par un résultat
|
|
228
|
-
`
|
|
212
|
+
- Ne jamais citer une décision connue uniquement par un résultat de
|
|
213
|
+
`Search_Jurisprudence` sans l'avoir d'abord lue avec `consulter_decision`.
|
|
229
214
|
- Ne jamais paraphraser un extrait pour qu'il « sonne » comme la source — le
|
|
230
215
|
vérificateur tolère l'espace/la casse/la ponctuation, pas le sens.
|
|
231
216
|
- Ne jamais laisser un `ref` sauter un numéro ou repartir en désordre.
|
|
@@ -1638,7 +1638,12 @@ function startInstallJob(repoRoot, component) {
|
|
|
1638
1638
|
windowsHide: true,
|
|
1639
1639
|
stdio: ['ignore', 'pipe', 'pipe'],
|
|
1640
1640
|
// PIECEMAKER_YES=1 : accepte les valeurs par défaut (MinerU inclus) sans TTY.
|
|
1641
|
-
env: {
|
|
1641
|
+
env: {
|
|
1642
|
+
...process.env,
|
|
1643
|
+
PIECEMAKER_YES: '1',
|
|
1644
|
+
NO_COLOR: '1',
|
|
1645
|
+
...(component === 'mineru' ? { PIECEMAKER_INSTALL_MINERU: '1' } : {}),
|
|
1646
|
+
},
|
|
1642
1647
|
});
|
|
1643
1648
|
const tailStderr = [];
|
|
1644
1649
|
const onLine = (chunk) => {
|
|
@@ -255,6 +255,14 @@ function spawnTracked(job, script, args, progressScale = {}, io = {}) {
|
|
|
255
255
|
}
|
|
256
256
|
continue;
|
|
257
257
|
}
|
|
258
|
+
if (line.startsWith('OCR_REQUIRED:')) {
|
|
259
|
+
try {
|
|
260
|
+
io.onOcrRequired?.(JSON.parse(line.slice('OCR_REQUIRED:'.length)));
|
|
261
|
+
} catch (error) {
|
|
262
|
+
errorLines.push(`pièces à OCR illisibles : ${error.message}`);
|
|
263
|
+
}
|
|
264
|
+
continue;
|
|
265
|
+
}
|
|
258
266
|
const progress = /^PROGRESS:([A-Z]+):(\d+):(\d+):(\d+)/.exec(line.trim());
|
|
259
267
|
if (!progress) continue;
|
|
260
268
|
const [, marker, pct, current, total] = progress;
|
|
@@ -341,7 +349,7 @@ function spawnTracked(job, script, args, progressScale = {}, io = {}) {
|
|
|
341
349
|
* `onProgress`, si fourni, est appelé à chaque mise à jour de la progression
|
|
342
350
|
* avec `{ phase, percent, processed, total }`.
|
|
343
351
|
*/
|
|
344
|
-
function runManagedPythonJob({ action, script, args, onProgress, signal, input, onMapping } = {}) {
|
|
352
|
+
function runManagedPythonJob({ action, script, args, onProgress, signal, input, onMapping, onOcrRequired } = {}) {
|
|
345
353
|
if (!['convert', 'anonymize'].includes(action)) throw new Error('Action inconnue.');
|
|
346
354
|
if (!acceptingJobs) throw new Error('Le serveur est en cours d’arrêt : aucun nouveau traitement ne peut démarrer.');
|
|
347
355
|
const job = {
|
|
@@ -375,7 +383,7 @@ function runManagedPythonJob({ action, script, args, onProgress, signal, input,
|
|
|
375
383
|
if (signal.aborted) abort();
|
|
376
384
|
else signal.addEventListener('abort', abort, { once: true });
|
|
377
385
|
}
|
|
378
|
-
const run = () => spawnTracked(job, script, args, {}, { input, onMapping });
|
|
386
|
+
const run = () => spawnTracked(job, script, args, {}, { input, onMapping, onOcrRequired });
|
|
379
387
|
const descriptor = { job, run };
|
|
380
388
|
const completion = new Promise((resolve, reject) => {
|
|
381
389
|
descriptor.resolve = resolve;
|
|
@@ -50,7 +50,10 @@ from contextlib import closing
|
|
|
50
50
|
from pathlib import Path
|
|
51
51
|
from typing import List, Optional, Tuple, Dict, Set
|
|
52
52
|
|
|
53
|
+
from smart_converter import mineru_available, needs_ocr
|
|
53
54
|
|
|
55
|
+
|
|
56
|
+
EXIT_OCR_REQUIRED = 3
|
|
54
57
|
GLINER_CHUNK_SIZE = 384
|
|
55
58
|
GLINER_CHUNK_OVERLAP = 64
|
|
56
59
|
|
|
@@ -1575,7 +1578,9 @@ def source_is_anonymized(state: Dict, source_file: str, case_root: str) -> bool:
|
|
|
1575
1578
|
return source_is_processed(state, source_file, case_root, "scanned")
|
|
1576
1579
|
|
|
1577
1580
|
|
|
1578
|
-
def update_processing_state(
|
|
1581
|
+
def update_processing_state(
|
|
1582
|
+
state_path: Path, case_root: str, source_files: List[str], phase: str, ocr_missing: bool = False
|
|
1583
|
+
) -> None:
|
|
1579
1584
|
"""Atomically add successful conversions/scans without paths or entities."""
|
|
1580
1585
|
if not source_files:
|
|
1581
1586
|
return
|
|
@@ -1587,9 +1592,12 @@ def update_processing_state(state_path: Path, case_root: str, source_files: List
|
|
|
1587
1592
|
continue
|
|
1588
1593
|
fingerprint = {**source_fingerprint(source_file), "updatedAt": updated_at}
|
|
1589
1594
|
entry = state["files"].get(key, {})
|
|
1595
|
+
previous_ocr = entry.get("converted", {}).get("ocr")
|
|
1590
1596
|
entry[phase] = fingerprint
|
|
1597
|
+
if phase == "converted" and ocr_missing:
|
|
1598
|
+
entry["converted"] = {**fingerprint, "ocr": "missing"}
|
|
1591
1599
|
if phase == "scanned":
|
|
1592
|
-
entry["converted"] = fingerprint
|
|
1600
|
+
entry["converted"] = {**fingerprint, **({"ocr": previous_ocr} if previous_ocr else {})}
|
|
1593
1601
|
state["files"][key] = entry
|
|
1594
1602
|
state_path.parent.mkdir(parents=True, exist_ok=True)
|
|
1595
1603
|
temporary = state_path.with_name(f"{state_path.name}.piecemaker-{os.getpid()}.tmp")
|
|
@@ -1800,6 +1808,15 @@ def run_pipeline(resources: PipelineResources):
|
|
|
1800
1808
|
help="MinerU processing mode (only used if engine=mineru)",
|
|
1801
1809
|
)
|
|
1802
1810
|
parser.add_argument("--lang", help="OCR language code (only used if engine=mineru)")
|
|
1811
|
+
parser.add_argument(
|
|
1812
|
+
"--ocr-missing",
|
|
1813
|
+
choices=["ask", "continue"],
|
|
1814
|
+
default="continue",
|
|
1815
|
+
help=(
|
|
1816
|
+
"When scanned documents need OCR but MinerU is not installed: 'ask' stops before "
|
|
1817
|
+
"any work and prints OCR_REQUIRED:<json>; 'continue' converts them from their text layer only"
|
|
1818
|
+
),
|
|
1819
|
+
)
|
|
1803
1820
|
parser.add_argument(
|
|
1804
1821
|
"--document-id",
|
|
1805
1822
|
default="default",
|
|
@@ -1871,7 +1888,36 @@ def run_pipeline(resources: PipelineResources):
|
|
|
1871
1888
|
def markdown_path(input_file: str) -> Path:
|
|
1872
1889
|
return Path(args.output) / f"{Path(input_file).stem}.md"
|
|
1873
1890
|
|
|
1874
|
-
|
|
1891
|
+
mineru_ready = args.engine == "mineru" or (args.engine == "auto" and mineru_available())
|
|
1892
|
+
|
|
1893
|
+
def converted_without_ocr(input_file: str) -> bool:
|
|
1894
|
+
entry = source_state_entry(anonymization_state, input_file, state_case_root) or {}
|
|
1895
|
+
return entry.get("converted", {}).get("ocr") == "missing"
|
|
1896
|
+
|
|
1897
|
+
def markdown_is_reusable(input_file: str) -> bool:
|
|
1898
|
+
if not markdown_path(input_file).exists():
|
|
1899
|
+
return False
|
|
1900
|
+
if mineru_ready and converted_without_ocr(input_file):
|
|
1901
|
+
return False
|
|
1902
|
+
state_entry = source_state_entry(anonymization_state, input_file, state_case_root)
|
|
1903
|
+
return state_entry is None or source_is_converted(anonymization_state, input_file, state_case_root)
|
|
1904
|
+
|
|
1905
|
+
def case_relative_name(input_file: str) -> str:
|
|
1906
|
+
try:
|
|
1907
|
+
return Path(input_file).resolve().relative_to(Path(state_case_root).resolve()).as_posix()
|
|
1908
|
+
except ValueError:
|
|
1909
|
+
return Path(input_file).name
|
|
1910
|
+
|
|
1911
|
+
conversions = [f for f in input_files if not (args.skip_existing and markdown_is_reusable(f))]
|
|
1912
|
+
ocr_fallback = set() if mineru_ready else {f for f in conversions if needs_ocr(f)}
|
|
1913
|
+
reconverted_with_ocr = {f for f in conversions if mineru_ready and converted_without_ocr(f)}
|
|
1914
|
+
|
|
1915
|
+
if ocr_fallback and args.engine == "auto" and args.ocr_missing == "ask":
|
|
1916
|
+
files = sorted(case_relative_name(f) for f in ocr_fallback)
|
|
1917
|
+
print(f"OCR_REQUIRED:{json.dumps({'files': files}, ensure_ascii=False)}", flush=True)
|
|
1918
|
+
return EXIT_OCR_REQUIRED
|
|
1919
|
+
|
|
1920
|
+
scan_needed = bool(reconverted_with_ocr) or any(
|
|
1875
1921
|
not (args.skip_existing and source_is_anonymized(anonymization_state, f, state_case_root))
|
|
1876
1922
|
for f in input_files
|
|
1877
1923
|
)
|
|
@@ -1899,31 +1945,38 @@ def run_pipeline(resources: PipelineResources):
|
|
|
1899
1945
|
print_progress("CONVERT", i, len(input_files))
|
|
1900
1946
|
existing_md = markdown_path(input_file)
|
|
1901
1947
|
|
|
1902
|
-
|
|
1903
|
-
reusable_markdown = existing_md.exists() and (
|
|
1904
|
-
state_entry is None
|
|
1905
|
-
or source_is_converted(anonymization_state, input_file, state_case_root)
|
|
1906
|
-
)
|
|
1907
|
-
if args.skip_existing and reusable_markdown:
|
|
1948
|
+
if input_file not in conversions:
|
|
1908
1949
|
print(f"📄 [{i}/{len(input_files)}] Already converted, reusing: {existing_md.name}")
|
|
1909
1950
|
md_files.append(str(existing_md))
|
|
1910
1951
|
md_sources[str(existing_md)] = input_file
|
|
1911
|
-
update_processing_state(
|
|
1952
|
+
update_processing_state(
|
|
1953
|
+
state_target, state_case_root, [input_file], "converted",
|
|
1954
|
+
ocr_missing=converted_without_ocr(input_file),
|
|
1955
|
+
)
|
|
1912
1956
|
convert_success_count += 1
|
|
1913
1957
|
print()
|
|
1914
1958
|
continue
|
|
1915
1959
|
|
|
1916
1960
|
print(f"📄 [{i}/{len(input_files)}] Converting: {Path(input_file).name}")
|
|
1961
|
+
without_ocr = input_file in ocr_fallback
|
|
1962
|
+
if without_ocr:
|
|
1963
|
+
print(" ⚠️ Scanned document converted without OCR (MinerU not installed): text layer only")
|
|
1917
1964
|
|
|
1918
1965
|
success, md_path = convert_file(
|
|
1919
|
-
input_file,
|
|
1966
|
+
input_file,
|
|
1967
|
+
args.output,
|
|
1968
|
+
engine="markitdown" if without_ocr else args.engine,
|
|
1969
|
+
mode=args.mode,
|
|
1970
|
+
lang=args.lang,
|
|
1920
1971
|
)
|
|
1921
1972
|
|
|
1922
1973
|
if success and md_path:
|
|
1923
1974
|
print(f" ✅ Markdown generated: {Path(md_path).name}")
|
|
1924
1975
|
md_files.append(md_path)
|
|
1925
1976
|
md_sources[md_path] = input_file
|
|
1926
|
-
update_processing_state(
|
|
1977
|
+
update_processing_state(
|
|
1978
|
+
state_target, state_case_root, [input_file], "converted", ocr_missing=without_ocr
|
|
1979
|
+
)
|
|
1927
1980
|
convert_success_count += 1
|
|
1928
1981
|
else:
|
|
1929
1982
|
print(f" ❌ Conversion failed, skipping...")
|
|
@@ -1948,7 +2001,8 @@ def run_pipeline(resources: PipelineResources):
|
|
|
1948
2001
|
pending_scans = [
|
|
1949
2002
|
md_file
|
|
1950
2003
|
for md_file in md_files
|
|
1951
|
-
if
|
|
2004
|
+
if md_sources[md_file] in reconverted_with_ocr
|
|
2005
|
+
or not (
|
|
1952
2006
|
args.skip_existing
|
|
1953
2007
|
and source_is_anonymized(anonymization_state, md_sources[md_file], state_case_root)
|
|
1954
2008
|
)
|
|
@@ -5,19 +5,31 @@ Supports both pipeline (direct PDF) and VLM/hybrid (image-based) modes
|
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
7
|
import argparse
|
|
8
|
+
import importlib.util
|
|
8
9
|
import json
|
|
9
10
|
import os
|
|
11
|
+
import shutil
|
|
10
12
|
import subprocess
|
|
11
13
|
import sys
|
|
14
|
+
import sysconfig
|
|
12
15
|
from pathlib import Path
|
|
13
16
|
|
|
14
17
|
|
|
15
|
-
def
|
|
18
|
+
def find_mineru():
|
|
16
19
|
name = 'mineru.exe' if os.name == 'nt' else 'mineru'
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
20
|
+
for directory in (sysconfig.get_path('scripts'), Path(sys.executable).parent, Path(sys.executable).resolve().parent):
|
|
21
|
+
candidate = Path(directory) / name
|
|
22
|
+
if candidate.is_file():
|
|
23
|
+
return str(candidate)
|
|
24
|
+
return shutil.which(name)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def mineru_executable():
|
|
28
|
+
return find_mineru() or 'mineru'
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def mineru_available():
|
|
32
|
+
return importlib.util.find_spec('mineru') is not None and find_mineru() is not None
|
|
21
33
|
|
|
22
34
|
def parse_with_pipeline(pdf_path, output_dir, lang=None):
|
|
23
35
|
"""
|
|
@@ -24,6 +24,10 @@ import argparse
|
|
|
24
24
|
import subprocess
|
|
25
25
|
from pathlib import Path
|
|
26
26
|
|
|
27
|
+
from mineru_improved_1 import mineru_available
|
|
28
|
+
|
|
29
|
+
IMAGE_EXTENSIONS = {'.jpg', '.jpeg', '.png', '.tiff', '.tif', '.bmp', '.gif'}
|
|
30
|
+
|
|
27
31
|
# ---------------------------------------------------------------------------
|
|
28
32
|
# Markdown whitespace normalisation
|
|
29
33
|
# ---------------------------------------------------------------------------
|
|
@@ -188,6 +192,19 @@ def pdf_has_text_layer(pdf_path):
|
|
|
188
192
|
return False # Assume scanned if inspection fails
|
|
189
193
|
|
|
190
194
|
|
|
195
|
+
def needs_ocr(file_path):
|
|
196
|
+
ext = Path(file_path).suffix.lower()
|
|
197
|
+
if ext in IMAGE_EXTENSIONS:
|
|
198
|
+
return True
|
|
199
|
+
return ext == '.pdf' and not pdf_has_text_layer(file_path)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def ocr_engine(reason):
|
|
203
|
+
if HAS_MARKITDOWN and not mineru_available():
|
|
204
|
+
return 'markitdown', f'{reason}, but MinerU is not installed — text layer only'
|
|
205
|
+
return 'mineru', reason
|
|
206
|
+
|
|
207
|
+
|
|
191
208
|
def inspect_file(file_path):
|
|
192
209
|
"""Determine best conversion engine based on file inspection.
|
|
193
210
|
|
|
@@ -197,8 +214,8 @@ def inspect_file(file_path):
|
|
|
197
214
|
ext = Path(file_path).suffix.lower()
|
|
198
215
|
|
|
199
216
|
# Images always need OCR → MinerU
|
|
200
|
-
if ext in
|
|
201
|
-
return '
|
|
217
|
+
if ext in IMAGE_EXTENSIONS:
|
|
218
|
+
return ocr_engine('Image file requires OCR')
|
|
202
219
|
|
|
203
220
|
# Office docs have native text → markitdown
|
|
204
221
|
if ext in ['.docx', '.pptx', '.xlsx', '.xls', '.doc', '.ppt']:
|
|
@@ -214,7 +231,7 @@ def inspect_file(file_path):
|
|
|
214
231
|
if pdf_has_text_layer(file_path):
|
|
215
232
|
return 'markitdown', 'PDF has extractable text layer'
|
|
216
233
|
else:
|
|
217
|
-
return '
|
|
234
|
+
return ocr_engine('Scanned PDF requires OCR (no text layer)')
|
|
218
235
|
|
|
219
236
|
# HTML, TXT, etc. — try markitdown
|
|
220
237
|
if ext in ['.html', '.htm', '.txt', '.md', '.csv']:
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{U as a,Q as n}from"./mermaid.core-D2KvaDg4.js";const t=(r,o)=>a.lang.round(n.parse(r)[o]);export{t as c};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{s as a,c as s,a as e,C as t}from"./chunk-TICWLB2K-DQ2F5I7F.js";import{_ as i}from"./mermaid.core-D2KvaDg4.js";import"./chunk-5VM5RSS4-DBpdDz0k.js";import"./chunk-XXDRQBXY-oLFixl7-.js";import"./chunk-POPQ4Y6H-uVCWOMNU.js";import"./chunk-F27PBJKO-npTlDSfn.js";import"./index-KoMJ5vMA.js";import"./vendor-codemirror-CN-sJePf.js";import"./vendor-react-B22wqN3J.js";import"./vendor-xterm-BWUgpGtR.js";var c={parser:e,get db(){return new t},renderer:s,styles:a,init:i(r=>{r.class||(r.class={}),r.class.arrowMarkerAbsolute=r.arrowMarkerAbsolute},"init")};export{c as diagram};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{s as a,c as s,a as e,C as t}from"./chunk-TICWLB2K-DQ2F5I7F.js";import{_ as i}from"./mermaid.core-D2KvaDg4.js";import"./chunk-5VM5RSS4-DBpdDz0k.js";import"./chunk-XXDRQBXY-oLFixl7-.js";import"./chunk-POPQ4Y6H-uVCWOMNU.js";import"./chunk-F27PBJKO-npTlDSfn.js";import"./index-KoMJ5vMA.js";import"./vendor-codemirror-CN-sJePf.js";import"./vendor-react-B22wqN3J.js";import"./vendor-xterm-BWUgpGtR.js";var c={parser:e,get db(){return new t},renderer:s,styles:a,init:i(r=>{r.class||(r.class={}),r.class.arrowMarkerAbsolute=r.arrowMarkerAbsolute},"init")};export{c as diagram};
|