@tricoteuses/senat 3.1.22 → 3.1.24
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +32 -32
- package/README.md +326 -326
- package/lib/src/index.d.ts +1 -0
- package/lib/src/other_types/collaborateurs.d.ts +5 -0
- package/lib/src/parsers/collaborateurs.d.ts +44 -0
- package/lib/src/parsers/collaborateurs.js +157 -0
- package/lib/src/rich_types/sens.d.ts +5 -0
- package/lib/src/scripts/data-download.js +1 -0
- package/lib/src/scripts/retrieve_collaborateurs.js +166 -0
- package/lib/src/scripts/retrieve_cr_seance.js +34 -25
- package/lib/src/scripts/retrieve_videos.js +13 -8
- package/lib/src/scripts/shared/incremental_import_sql.js +866 -859
- package/lib/src/scripts/shared/schema_version.js +84 -84
- package/lib/src/scripts/shared/staging_metadata_sql.js +214 -214
- package/lib/src/scripts/validate_prefixed_tables.js +12 -12
- package/lib/src/server/ameli.js +14 -13
- package/lib/src/server/conversion_textes.js +106 -106
- package/lib/src/server/debats.js +10 -10
- package/lib/src/server/documents.js +22 -22
- package/lib/src/server/dosleg.js +33 -33
- package/lib/src/server/questions.js +10 -10
- package/lib/src/server/scrutins.js +3 -3
- package/lib/src/server/sens.js +19 -19
- package/lib/src/utils/reunion_odj_building.js +0 -4
- package/lib/src/videos/match.js +7 -5
- package/lib/src/videos/pipeline.js +26 -19
- package/lib/tests/collaborateurs.test.js +115 -0
- package/lib/tests/incrementalImportSql.test.js +4 -1
- package/package.json +121 -119
- package/lib/add-js-extensions-v2.js +0 -23
- package/lib/add-js-extensions.js +0 -17
- package/lib/aggregates.d.ts +0 -52
- package/lib/aggregates.js +0 -930
- package/lib/aggregates.mjs +0 -713
- package/lib/aggregates.ts +0 -833
- package/lib/config.d.ts +0 -10
- package/lib/config.js +0 -16
- package/lib/config.mjs +0 -16
- package/lib/config.ts +0 -26
- package/lib/databases.d.ts +0 -2
- package/lib/databases.js +0 -26
- package/lib/databases.mjs +0 -57
- package/lib/databases.ts +0 -71
- package/lib/datasets.d.ts +0 -34
- package/lib/datasets.js +0 -233
- package/lib/datasets.mjs +0 -78
- package/lib/datasets.ts +0 -118
- package/lib/fields.d.ts +0 -10
- package/lib/fields.js +0 -68
- package/lib/fields.mjs +0 -22
- package/lib/fields.ts +0 -29
- package/lib/git.d.ts +0 -26
- package/lib/git.js +0 -167
- package/lib/index.d.ts +0 -13
- package/lib/index.js +0 -1
- package/lib/index.mjs +0 -7
- package/lib/index.ts +0 -64
- package/lib/inserters.d.ts +0 -98
- package/lib/inserters.js +0 -500
- package/lib/inserters.mjs +0 -360
- package/lib/inserters.ts +0 -521
- package/lib/legislatures.json +0 -38
- package/lib/loaders.d.ts +0 -58
- package/lib/loaders.js +0 -286
- package/lib/loaders.mjs +0 -158
- package/lib/loaders.ts +0 -271
- package/lib/model/agenda.d.ts +0 -6
- package/lib/model/agenda.js +0 -148
- package/lib/model/ameli.d.ts +0 -51
- package/lib/model/ameli.js +0 -149
- package/lib/model/ameli.mjs +0 -84
- package/lib/model/ameli.ts +0 -100
- package/lib/model/commission.d.ts +0 -18
- package/lib/model/commission.js +0 -269
- package/lib/model/debats.d.ts +0 -67
- package/lib/model/debats.js +0 -95
- package/lib/model/debats.mjs +0 -43
- package/lib/model/debats.ts +0 -68
- package/lib/model/documents.d.ts +0 -12
- package/lib/model/documents.js +0 -151
- package/lib/model/dosleg.d.ts +0 -7
- package/lib/model/dosleg.js +0 -326
- package/lib/model/dosleg.mjs +0 -196
- package/lib/model/dosleg.ts +0 -240
- package/lib/model/index.d.ts +0 -7
- package/lib/model/index.js +0 -7
- package/lib/model/index.mjs +0 -5
- package/lib/model/index.ts +0 -15
- package/lib/model/questions.d.ts +0 -45
- package/lib/model/questions.js +0 -89
- package/lib/model/questions.mjs +0 -71
- package/lib/model/questions.ts +0 -93
- package/lib/model/scrutins.d.ts +0 -13
- package/lib/model/scrutins.js +0 -114
- package/lib/model/seance.d.ts +0 -3
- package/lib/model/seance.js +0 -267
- package/lib/model/sens.d.ts +0 -146
- package/lib/model/sens.js +0 -454
- package/lib/model/sens.mjs +0 -415
- package/lib/model/sens.ts +0 -516
- package/lib/model/texte.d.ts +0 -7
- package/lib/model/texte.js +0 -256
- package/lib/model/texte.mjs +0 -208
- package/lib/model/texte.ts +0 -229
- package/lib/model/util.d.ts +0 -9
- package/lib/model/util.js +0 -38
- package/lib/model/util.mjs +0 -19
- package/lib/model/util.ts +0 -32
- package/lib/parsers/texte.d.ts +0 -7
- package/lib/parsers/texte.js +0 -228
- package/lib/raw_types/ameli.d.ts +0 -914
- package/lib/raw_types/ameli.js +0 -5
- package/lib/raw_types/ameli.mjs +0 -163
- package/lib/raw_types/debats.d.ts +0 -207
- package/lib/raw_types/debats.js +0 -5
- package/lib/raw_types/debats.mjs +0 -58
- package/lib/raw_types/dosleg.d.ts +0 -1619
- package/lib/raw_types/dosleg.js +0 -5
- package/lib/raw_types/dosleg.mjs +0 -438
- package/lib/raw_types/questions.d.ts +0 -419
- package/lib/raw_types/questions.js +0 -5
- package/lib/raw_types/questions.mjs +0 -11
- package/lib/raw_types/senat.d.ts +0 -11368
- package/lib/raw_types/senat.js +0 -5
- package/lib/raw_types/sens.d.ts +0 -8248
- package/lib/raw_types/sens.js +0 -5
- package/lib/raw_types/sens.mjs +0 -508
- package/lib/raw_types_kysely/ameli.d.ts +0 -915
- package/lib/raw_types_kysely/ameli.js +0 -7
- package/lib/raw_types_kysely/ameli.mjs +0 -5
- package/lib/raw_types_kysely/ameli.ts +0 -951
- package/lib/raw_types_kysely/debats.d.ts +0 -207
- package/lib/raw_types_kysely/debats.js +0 -7
- package/lib/raw_types_kysely/debats.mjs +0 -5
- package/lib/raw_types_kysely/debats.ts +0 -222
- package/lib/raw_types_kysely/dosleg.d.ts +0 -3532
- package/lib/raw_types_kysely/dosleg.js +0 -7
- package/lib/raw_types_kysely/dosleg.mjs +0 -5
- package/lib/raw_types_kysely/dosleg.ts +0 -3621
- package/lib/raw_types_kysely/questions.d.ts +0 -414
- package/lib/raw_types_kysely/questions.js +0 -7
- package/lib/raw_types_kysely/questions.mjs +0 -5
- package/lib/raw_types_kysely/questions.ts +0 -426
- package/lib/raw_types_kysely/sens.d.ts +0 -4394
- package/lib/raw_types_kysely/sens.js +0 -7
- package/lib/raw_types_kysely/sens.mjs +0 -5
- package/lib/raw_types_kysely/sens.ts +0 -4499
- package/lib/raw_types_schemats/ameli.d.ts +0 -539
- package/lib/raw_types_schemats/ameli.js +0 -2
- package/lib/raw_types_schemats/ameli.mjs +0 -2
- package/lib/raw_types_schemats/ameli.ts +0 -601
- package/lib/raw_types_schemats/debats.d.ts +0 -127
- package/lib/raw_types_schemats/debats.js +0 -2
- package/lib/raw_types_schemats/debats.mjs +0 -2
- package/lib/raw_types_schemats/debats.ts +0 -145
- package/lib/raw_types_schemats/dosleg.d.ts +0 -977
- package/lib/raw_types_schemats/dosleg.js +0 -2
- package/lib/raw_types_schemats/dosleg.mjs +0 -2
- package/lib/raw_types_schemats/dosleg.ts +0 -2193
- package/lib/raw_types_schemats/questions.d.ts +0 -235
- package/lib/raw_types_schemats/questions.js +0 -2
- package/lib/raw_types_schemats/questions.mjs +0 -2
- package/lib/raw_types_schemats/questions.ts +0 -249
- package/lib/raw_types_schemats/sens.d.ts +0 -6915
- package/lib/raw_types_schemats/sens.js +0 -2
- package/lib/raw_types_schemats/sens.mjs +0 -2
- package/lib/raw_types_schemats/sens.ts +0 -2907
- package/lib/scripts/convert_data.js +0 -354
- package/lib/scripts/convert_data.mjs +0 -181
- package/lib/scripts/convert_data.ts +0 -243
- package/lib/scripts/data-download.d.ts +0 -1
- package/lib/scripts/data-download.js +0 -12
- package/lib/scripts/datautil.d.ts +0 -8
- package/lib/scripts/datautil.js +0 -34
- package/lib/scripts/datautil.mjs +0 -16
- package/lib/scripts/datautil.ts +0 -19
- package/lib/scripts/images/transparent_150x192.jpg +0 -0
- package/lib/scripts/images/transparent_155x225.jpg +0 -0
- package/lib/scripts/parse_textes.d.ts +0 -1
- package/lib/scripts/parse_textes.js +0 -44
- package/lib/scripts/parse_textes.mjs +0 -46
- package/lib/scripts/parse_textes.ts +0 -65
- package/lib/scripts/retrieve_agenda.d.ts +0 -1
- package/lib/scripts/retrieve_agenda.js +0 -132
- package/lib/scripts/retrieve_cr_commission.d.ts +0 -1
- package/lib/scripts/retrieve_cr_commission.js +0 -364
- package/lib/scripts/retrieve_cr_seance.d.ts +0 -6
- package/lib/scripts/retrieve_cr_seance.js +0 -347
- package/lib/scripts/retrieve_documents.d.ts +0 -3
- package/lib/scripts/retrieve_documents.js +0 -219
- package/lib/scripts/retrieve_documents.mjs +0 -249
- package/lib/scripts/retrieve_documents.ts +0 -298
- package/lib/scripts/retrieve_open_data.d.ts +0 -1
- package/lib/scripts/retrieve_open_data.js +0 -315
- package/lib/scripts/retrieve_open_data.mjs +0 -217
- package/lib/scripts/retrieve_open_data.ts +0 -268
- package/lib/scripts/retrieve_senateurs_photos.d.ts +0 -1
- package/lib/scripts/retrieve_senateurs_photos.js +0 -147
- package/lib/scripts/retrieve_senateurs_photos.mjs +0 -147
- package/lib/scripts/retrieve_senateurs_photos.ts +0 -177
- package/lib/scripts/retrieve_videos.d.ts +0 -1
- package/lib/scripts/retrieve_videos.js +0 -461
- package/lib/scripts/shared/cli_helpers.d.ts +0 -95
- package/lib/scripts/shared/cli_helpers.js +0 -91
- package/lib/scripts/shared/cli_helpers.ts +0 -36
- package/lib/scripts/shared/util.d.ts +0 -4
- package/lib/scripts/shared/util.js +0 -35
- package/lib/scripts/shared/util.ts +0 -33
- package/lib/scripts/test_iter_load.d.ts +0 -1
- package/lib/scripts/test_iter_load.js +0 -12
- package/lib/src/ameli.d.ts +0 -66
- package/lib/src/ameli.js +0 -1
- package/lib/src/config.d.ts +0 -43
- package/lib/src/config.js +0 -37
- package/lib/src/conversion_textes.d.ts +0 -11
- package/lib/src/conversion_textes.js +0 -320
- package/lib/src/databases.d.ts +0 -3
- package/lib/src/databases.js +0 -26
- package/lib/src/databases_postgres.d.ts +0 -4
- package/lib/src/databases_postgres.js +0 -23
- package/lib/src/datasets.d.ts +0 -38
- package/lib/src/datasets.js +0 -247
- package/lib/src/db_types/ameli.d.ts +0 -1762
- package/lib/src/db_types/ameli.js +0 -1074
- package/lib/src/db_types/debats.d.ts +0 -380
- package/lib/src/db_types/debats.js +0 -266
- package/lib/src/db_types/dosleg.d.ts +0 -2954
- package/lib/src/db_types/dosleg.js +0 -2005
- package/lib/src/db_types/questions.d.ts +0 -699
- package/lib/src/db_types/questions.js +0 -493
- package/lib/src/db_types/sens.d.ts +0 -7843
- package/lib/src/db_types/sens.js +0 -4691
- package/lib/src/debats.d.ts +0 -38
- package/lib/src/debats.js +0 -1
- package/lib/src/dosleg.d.ts +0 -142
- package/lib/src/dosleg.js +0 -193
- package/lib/src/git.d.ts +0 -27
- package/lib/src/git.js +0 -251
- package/lib/src/loaders.d.ts +0 -52
- package/lib/src/loaders.js +0 -260
- package/lib/src/model/agenda.d.ts +0 -6
- package/lib/src/model/agenda.js +0 -148
- package/lib/src/model/ameli.d.ts +0 -67
- package/lib/src/model/ameli.js +0 -150
- package/lib/src/model/ameli_postgres.d.ts +0 -67
- package/lib/src/model/ameli_postgres.js +0 -150
- package/lib/src/model/commission.d.ts +0 -19
- package/lib/src/model/commission.js +0 -269
- package/lib/src/model/debats.d.ts +0 -39
- package/lib/src/model/debats.js +0 -112
- package/lib/src/model/documents.d.ts +0 -32
- package/lib/src/model/documents.js +0 -182
- package/lib/src/model/dosleg.d.ts +0 -144
- package/lib/src/model/dosleg.js +0 -468
- package/lib/src/model/index.d.ts +0 -7
- package/lib/src/model/index.js +0 -7
- package/lib/src/model/questions.d.ts +0 -54
- package/lib/src/model/questions.js +0 -91
- package/lib/src/model/scrutins.d.ts +0 -48
- package/lib/src/model/scrutins.js +0 -121
- package/lib/src/model/seance.d.ts +0 -3
- package/lib/src/model/seance.js +0 -267
- package/lib/src/model/sens.d.ts +0 -112
- package/lib/src/model/sens.js +0 -385
- package/lib/src/model/util.d.ts +0 -1
- package/lib/src/model/util.js +0 -15
- package/lib/src/other_types/questions.d.ts +0 -2
- package/lib/src/other_types/questions.js +0 -1
- package/lib/src/questions.d.ts +0 -53
- package/lib/src/questions.js +0 -1
- package/lib/src/raw_types/ameli.d.ts +0 -1762
- package/lib/src/raw_types/ameli.js +0 -1074
- package/lib/src/raw_types/debats.d.ts +0 -380
- package/lib/src/raw_types/debats.js +0 -266
- package/lib/src/raw_types/dosleg.d.ts +0 -2954
- package/lib/src/raw_types/dosleg.js +0 -2005
- package/lib/src/raw_types/questions.d.ts +0 -699
- package/lib/src/raw_types/questions.js +0 -493
- package/lib/src/raw_types/senat.d.ts +0 -11372
- package/lib/src/raw_types/senat.js +0 -5
- package/lib/src/raw_types/sens.d.ts +0 -7843
- package/lib/src/raw_types/sens.js +0 -4691
- package/lib/src/raw_types_schemats/ameli.d.ts +0 -541
- package/lib/src/raw_types_schemats/ameli.js +0 -2
- package/lib/src/raw_types_schemats/debats.d.ts +0 -127
- package/lib/src/raw_types_schemats/debats.js +0 -2
- package/lib/src/raw_types_schemats/dosleg.d.ts +0 -977
- package/lib/src/raw_types_schemats/dosleg.js +0 -2
- package/lib/src/raw_types_schemats/questions.d.ts +0 -237
- package/lib/src/raw_types_schemats/questions.js +0 -2
- package/lib/src/raw_types_schemats/sens.d.ts +0 -2709
- package/lib/src/raw_types_schemats/sens.js +0 -2
- package/lib/src/rich_types/agenda.d.ts +0 -45
- package/lib/src/rich_types/agenda.js +0 -1
- package/lib/src/rich_types/compte_rendu.d.ts +0 -83
- package/lib/src/rich_types/compte_rendu.js +0 -1
- package/lib/src/rich_types/sessions.d.ts +0 -6
- package/lib/src/rich_types/sessions.js +0 -19
- package/lib/src/rich_types/texte.d.ts +0 -72
- package/lib/src/rich_types/texte.js +0 -15
- package/lib/src/scripts/test_iter_load.d.ts +0 -1
- package/lib/src/scripts/test_iter_load.js +0 -12
- package/lib/src/sens.d.ts +0 -104
- package/lib/src/sens.js +0 -1
- package/lib/src/types/agenda.d.ts +0 -45
- package/lib/src/types/agenda.js +0 -1
- package/lib/src/types/ameli.d.ts +0 -1762
- package/lib/src/types/ameli.js +0 -1074
- package/lib/src/types/compte_rendu.d.ts +0 -83
- package/lib/src/types/compte_rendu.js +0 -1
- package/lib/src/types/debats.d.ts +0 -380
- package/lib/src/types/debats.js +0 -266
- package/lib/src/types/dosleg.d.ts +0 -2954
- package/lib/src/types/dosleg.js +0 -2005
- package/lib/src/types/questions.d.ts +0 -699
- package/lib/src/types/questions.js +0 -493
- package/lib/src/types/sens.d.ts +0 -7843
- package/lib/src/types/sens.js +0 -4691
- package/lib/src/types/sessions.d.ts +0 -6
- package/lib/src/types/sessions.js +0 -19
- package/lib/src/types/texte.d.ts +0 -72
- package/lib/src/types/texte.js +0 -15
- package/lib/src/validators/config.d.ts +0 -9
- package/lib/src/validators/config.js +0 -10
- package/lib/strings.d.ts +0 -1
- package/lib/strings.js +0 -18
- package/lib/strings.mjs +0 -18
- package/lib/strings.ts +0 -26
- package/lib/tsconfig.tsbuildinfo +0 -1
- package/lib/types/agenda.d.ts +0 -44
- package/lib/types/agenda.js +0 -1
- package/lib/types/ameli.d.ts +0 -5
- package/lib/types/ameli.js +0 -1
- package/lib/types/ameli.mjs +0 -13
- package/lib/types/ameli.ts +0 -21
- package/lib/types/compte_rendu.d.ts +0 -83
- package/lib/types/compte_rendu.js +0 -1
- package/lib/types/debats.d.ts +0 -2
- package/lib/types/debats.js +0 -1
- package/lib/types/debats.mjs +0 -2
- package/lib/types/debats.ts +0 -6
- package/lib/types/dosleg.d.ts +0 -70
- package/lib/types/dosleg.js +0 -1
- package/lib/types/dosleg.mjs +0 -151
- package/lib/types/dosleg.ts +0 -284
- package/lib/types/questions.d.ts +0 -2
- package/lib/types/questions.js +0 -1
- package/lib/types/questions.mjs +0 -1
- package/lib/types/questions.ts +0 -3
- package/lib/types/sens.d.ts +0 -10
- package/lib/types/sens.js +0 -1
- package/lib/types/sens.mjs +0 -1
- package/lib/types/sens.ts +0 -12
- package/lib/types/sessions.d.ts +0 -5
- package/lib/types/sessions.js +0 -84
- package/lib/types/sessions.mjs +0 -43
- package/lib/types/sessions.ts +0 -42
- package/lib/types/texte.d.ts +0 -74
- package/lib/types/texte.js +0 -16
- package/lib/types/texte.mjs +0 -16
- package/lib/types/texte.ts +0 -76
- package/lib/typings/windows-1252.d.js +0 -2
- package/lib/typings/windows-1252.d.mjs +0 -2
- package/lib/typings/windows-1252.d.ts +0 -11
- package/lib/utils/cr_spliting.d.ts +0 -28
- package/lib/utils/cr_spliting.js +0 -265
- package/lib/utils/date.d.ts +0 -10
- package/lib/utils/date.js +0 -100
- package/lib/utils/nvs-timecode.d.ts +0 -7
- package/lib/utils/nvs-timecode.js +0 -79
- package/lib/utils/reunion_grouping.d.ts +0 -9
- package/lib/utils/reunion_grouping.js +0 -361
- package/lib/utils/reunion_odj_building.d.ts +0 -5
- package/lib/utils/reunion_odj_building.js +0 -154
- package/lib/utils/reunion_parsing.d.ts +0 -23
- package/lib/utils/reunion_parsing.js +0 -209
- package/lib/utils/scoring.d.ts +0 -14
- package/lib/utils/scoring.js +0 -147
- package/lib/utils/string_cleaning.d.ts +0 -7
- package/lib/utils/string_cleaning.js +0 -57
- package/lib/validators/config.d.ts +0 -9
- package/lib/validators/config.js +0 -10
- package/lib/validators/config.mjs +0 -54
- package/lib/validators/config.ts +0 -79
- package/lib/validators/senat.d.ts +0 -0
- package/lib/validators/senat.js +0 -28
- package/lib/validators/senat.mjs +0 -24
- package/lib/validators/senat.ts +0 -26
- /package/lib/{add-js-extensions-v2.d.ts → src/other_types/collaborateurs.js} +0 -0
- /package/lib/{add-js-extensions.d.ts → src/scripts/retrieve_collaborateurs.d.ts} +0 -0
- /package/lib/{scripts/convert_data.d.ts → tests/collaborateurs.test.d.ts} +0 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure functions for parsing the HR PDF listing senators' collaborators
|
|
3
|
+
* (https://www.senat.fr/pubagas/liste_senateurs_collaborateurs.pdf).
|
|
4
|
+
*
|
|
5
|
+
* The PDF is a two-column table per page:
|
|
6
|
+
* - "Employeur" (the senator) on the left, e.g. "Mme AESCHLIMANN Marie-Do" ;
|
|
7
|
+
* - "Nom Collaborateur" on the right, e.g. "M. CHAREF Dahmane".
|
|
8
|
+
* A line without an employer extends the previous senator.
|
|
9
|
+
*
|
|
10
|
+
* Edge cases handled here: no matricule in the PDF, truncated senator first names
|
|
11
|
+
* ("Marie-Do"), compound names ("DI FOLCO", "MAZENOT CHAPPUY"), accents.
|
|
12
|
+
*/
|
|
13
|
+
/** Positioned text element extracted from the PDF (x goes right, y goes up). */
|
|
14
|
+
export type PdfTextItem = {
|
|
15
|
+
str: string;
|
|
16
|
+
x: number;
|
|
17
|
+
y: number;
|
|
18
|
+
page: number;
|
|
19
|
+
};
|
|
20
|
+
export type PersonName = {
|
|
21
|
+
civilite: string;
|
|
22
|
+
nom: string;
|
|
23
|
+
prenom: string;
|
|
24
|
+
};
|
|
25
|
+
export type CollaboratorRow = {
|
|
26
|
+
senateur: PersonName;
|
|
27
|
+
collaborateur: PersonName;
|
|
28
|
+
};
|
|
29
|
+
/** Normalizes a string for comparison: uppercase, no accents, compacted spaces. */
|
|
30
|
+
export declare function normalizeForComparison(value: string): string;
|
|
31
|
+
/**
|
|
32
|
+
* Splits a "Civility LAST_NAME(S) FirstName" cell into {civilite, nom, prenom}.
|
|
33
|
+
* Uppercase head tokens form the name (handles compound names), the rest is the first name.
|
|
34
|
+
* Returns null if the cell is not a usable person name.
|
|
35
|
+
*/
|
|
36
|
+
export declare function parsePersonCell(cell: string): PersonName | null;
|
|
37
|
+
/** Extracts the edition date printed in the footer ("Edition du JJ/MM/AAAA"). */
|
|
38
|
+
export declare function extractEditionDate(strings: string[]): Date | null;
|
|
39
|
+
/**
|
|
40
|
+
* Reconstructs the {senator, collaborator} list from positioned text items.
|
|
41
|
+
* Iterates pages then lines top to bottom; tracks the current senator and
|
|
42
|
+
* attaches collaborators from subsequent lines until a new employer appears.
|
|
43
|
+
*/
|
|
44
|
+
export declare function buildCollaboratorRows(items: PdfTextItem[]): CollaboratorRow[];
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure functions for parsing the HR PDF listing senators' collaborators
|
|
3
|
+
* (https://www.senat.fr/pubagas/liste_senateurs_collaborateurs.pdf).
|
|
4
|
+
*
|
|
5
|
+
* The PDF is a two-column table per page:
|
|
6
|
+
* - "Employeur" (the senator) on the left, e.g. "Mme AESCHLIMANN Marie-Do" ;
|
|
7
|
+
* - "Nom Collaborateur" on the right, e.g. "M. CHAREF Dahmane".
|
|
8
|
+
* A line without an employer extends the previous senator.
|
|
9
|
+
*
|
|
10
|
+
* Edge cases handled here: no matricule in the PDF, truncated senator first names
|
|
11
|
+
* ("Marie-Do"), compound names ("DI FOLCO", "MAZENOT CHAPPUY"), accents.
|
|
12
|
+
*/
|
|
13
|
+
// Horizontal bands of the two columns (measured on the PDF: employer ≈ 147, collaborator ≈ 285).
|
|
14
|
+
const EMPLOYER_COLUMN = { min: 130, max: 225 };
|
|
15
|
+
const COLLABORATOR_COLUMN = { min: 255, max: 360 };
|
|
16
|
+
const CIVILITIES = ["Mme", "M.", "Mlle", "M"];
|
|
17
|
+
// Lowercase nobiliary/patronymic particles that are part of the name when they precede it
|
|
18
|
+
// (e.g. "de CIDRAC", "de LA GONTRIE", "de LEGGE"). The Senate reference keeps them in Nom_usuel.
|
|
19
|
+
const PARTICLES = new Set([
|
|
20
|
+
"de",
|
|
21
|
+
"du",
|
|
22
|
+
"des",
|
|
23
|
+
"d'",
|
|
24
|
+
"le",
|
|
25
|
+
"la",
|
|
26
|
+
"les",
|
|
27
|
+
"von",
|
|
28
|
+
"van",
|
|
29
|
+
"der",
|
|
30
|
+
"den",
|
|
31
|
+
"da",
|
|
32
|
+
"di",
|
|
33
|
+
"del",
|
|
34
|
+
"dos",
|
|
35
|
+
]);
|
|
36
|
+
// Header / footer lines to ignore.
|
|
37
|
+
const IGNORED_LINES = [
|
|
38
|
+
/liste des collaborateurs/i,
|
|
39
|
+
/^employeur$/i,
|
|
40
|
+
/^nom collaborateur$/i,
|
|
41
|
+
/^a\.?g\.?a\.?s\.?/i,
|
|
42
|
+
/edition du/i,
|
|
43
|
+
/congé non rémunéré/i,
|
|
44
|
+
/^-\s*\d+\s*-$/,
|
|
45
|
+
];
|
|
46
|
+
/** Normalizes a string for comparison: uppercase, no accents, compacted spaces. */
|
|
47
|
+
export function normalizeForComparison(value) {
|
|
48
|
+
return value
|
|
49
|
+
.normalize("NFD")
|
|
50
|
+
.replace(/\p{Diacritic}/gu, "")
|
|
51
|
+
.toUpperCase()
|
|
52
|
+
.replace(/['']/g, "'")
|
|
53
|
+
.replace(/\s+/g, " ")
|
|
54
|
+
.trim();
|
|
55
|
+
}
|
|
56
|
+
function isIgnored(str) {
|
|
57
|
+
const t = str.trim();
|
|
58
|
+
if (!t)
|
|
59
|
+
return true;
|
|
60
|
+
return IGNORED_LINES.some((re) => re.test(t));
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Splits a "Civility LAST_NAME(S) FirstName" cell into {civilite, nom, prenom}.
|
|
64
|
+
* Uppercase head tokens form the name (handles compound names), the rest is the first name.
|
|
65
|
+
* Returns null if the cell is not a usable person name.
|
|
66
|
+
*/
|
|
67
|
+
export function parsePersonCell(cell) {
|
|
68
|
+
const raw = cell
|
|
69
|
+
.replace(/\(\*\)/g, "")
|
|
70
|
+
.replace(/\s+/g, " ")
|
|
71
|
+
.trim();
|
|
72
|
+
if (!raw)
|
|
73
|
+
return null;
|
|
74
|
+
let civilite = "";
|
|
75
|
+
let rest = raw;
|
|
76
|
+
for (const civ of CIVILITIES) {
|
|
77
|
+
if (raw === civ)
|
|
78
|
+
return null;
|
|
79
|
+
if (raw.startsWith(`${civ} `)) {
|
|
80
|
+
civilite = civ;
|
|
81
|
+
rest = raw.slice(civ.length + 1).trim();
|
|
82
|
+
break;
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
if (!rest)
|
|
86
|
+
return null;
|
|
87
|
+
const tokens = rest.split(" ");
|
|
88
|
+
const isUppercase = (token) => {
|
|
89
|
+
const letters = token.replace(/[^A-Za-zÀ-ÿ]/g, "");
|
|
90
|
+
return letters.length > 0 && letters === letters.toUpperCase();
|
|
91
|
+
};
|
|
92
|
+
const nameTokens = [];
|
|
93
|
+
let i = 0;
|
|
94
|
+
// Leading lowercase particles (e.g. "de", "de la") that are part of the name.
|
|
95
|
+
while (i < tokens.length && PARTICLES.has(tokens[i].toLowerCase())) {
|
|
96
|
+
nameTokens.push(tokens[i]);
|
|
97
|
+
i++;
|
|
98
|
+
}
|
|
99
|
+
// Then uppercase name tokens.
|
|
100
|
+
while (i < tokens.length && isUppercase(tokens[i])) {
|
|
101
|
+
nameTokens.push(tokens[i]);
|
|
102
|
+
i++;
|
|
103
|
+
}
|
|
104
|
+
// A lone particle without an uppercase name is not valid: reset.
|
|
105
|
+
if (nameTokens.length > 0 && nameTokens.every((j) => PARTICLES.has(j.toLowerCase()))) {
|
|
106
|
+
nameTokens.length = 0;
|
|
107
|
+
i = 0;
|
|
108
|
+
}
|
|
109
|
+
// If no uppercase token (unexpected case), take the first as name.
|
|
110
|
+
if (nameTokens.length === 0 && tokens.length > 0) {
|
|
111
|
+
nameTokens.push(tokens[0]);
|
|
112
|
+
i = 1;
|
|
113
|
+
}
|
|
114
|
+
const nom = nameTokens.join(" ");
|
|
115
|
+
const prenom = tokens.slice(i).join(" ");
|
|
116
|
+
if (!nom)
|
|
117
|
+
return null;
|
|
118
|
+
return { civilite, nom, prenom };
|
|
119
|
+
}
|
|
120
|
+
/** Extracts the edition date printed in the footer ("Edition du JJ/MM/AAAA"). */
|
|
121
|
+
export function extractEditionDate(strings) {
|
|
122
|
+
for (const s of strings) {
|
|
123
|
+
const m = s.match(/Edition du\s+(\d{2})\/(\d{2})\/(\d{4})/i);
|
|
124
|
+
if (m) {
|
|
125
|
+
const [, dd, mm, yyyy] = m;
|
|
126
|
+
return new Date(Date.UTC(Number(yyyy), Number(mm) - 1, Number(dd)));
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
return null;
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* Reconstructs the {senator, collaborator} list from positioned text items.
|
|
133
|
+
* Iterates pages then lines top to bottom; tracks the current senator and
|
|
134
|
+
* attaches collaborators from subsequent lines until a new employer appears.
|
|
135
|
+
*/
|
|
136
|
+
export function buildCollaboratorRows(items) {
|
|
137
|
+
const useful = items.filter((it) => !isIgnored(it.str));
|
|
138
|
+
// Reading order: increasing page, then decreasing y (top → bottom), then increasing x (left → right).
|
|
139
|
+
const sorted = [...useful].sort((a, b) => a.page - b.page || b.y - a.y || a.x - b.x);
|
|
140
|
+
const rows = [];
|
|
141
|
+
let currentSenator = null;
|
|
142
|
+
for (const it of sorted) {
|
|
143
|
+
const inEmployer = it.x >= EMPLOYER_COLUMN.min && it.x <= EMPLOYER_COLUMN.max;
|
|
144
|
+
const inCollaborator = it.x >= COLLABORATOR_COLUMN.min && it.x <= COLLABORATOR_COLUMN.max;
|
|
145
|
+
if (inEmployer) {
|
|
146
|
+
const person = parsePersonCell(it.str);
|
|
147
|
+
if (person)
|
|
148
|
+
currentSenator = person;
|
|
149
|
+
}
|
|
150
|
+
else if (inCollaborator && currentSenator) {
|
|
151
|
+
const collaborateur = parsePersonCell(it.str);
|
|
152
|
+
if (collaborateur)
|
|
153
|
+
rows.push({ senateur: currentSenator, collaborateur });
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
return rows;
|
|
157
|
+
}
|
|
@@ -79,6 +79,11 @@ export interface SenateurResult {
|
|
|
79
79
|
siege: string | null;
|
|
80
80
|
url_hatvp: string | null;
|
|
81
81
|
urls: UrlRow[];
|
|
82
|
+
collaborateurs?: Array<{
|
|
83
|
+
civilite: string;
|
|
84
|
+
nom: string;
|
|
85
|
+
prenom: string;
|
|
86
|
+
}>;
|
|
82
87
|
}
|
|
83
88
|
export interface CirconscriptionResult {
|
|
84
89
|
article: string | null;
|
|
@@ -18,3 +18,4 @@ runScript(`cross-env TZ='Etc/UTC' tsx src/scripts/retrieve_agenda.ts ${args} --p
|
|
|
18
18
|
runScript(`tsx src/scripts/retrieve_cr_seance.ts ${args} --parseDebats --silent`);
|
|
19
19
|
runScript(`tsx src/scripts/retrieve_cr_commission.ts ${args} --parseDebats --silent`);
|
|
20
20
|
runScript(`tsx src/scripts/retrieve_videos.ts ${args} --silent`);
|
|
21
|
+
runScript(`tsx src/scripts/retrieve_collaborateurs.ts ${args} --silent`);
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
import fs from "fs-extra";
|
|
2
|
+
import path from "path";
|
|
3
|
+
import { getDocumentProxy } from "unpdf";
|
|
4
|
+
import { buildCollaboratorRows, extractEditionDate, normalizeForComparison, } from "../parsers/collaborateurs.js";
|
|
5
|
+
import { assertExistingDirectory } from "./shared/cli_helpers.js";
|
|
6
|
+
import { iterFilePaths } from "../server/loaders.js";
|
|
7
|
+
const DEFAULT_PDF_URL = "https://www.senat.fr/pubagas/liste_senateurs_collaborateurs.pdf";
|
|
8
|
+
const USER_AGENT = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120 Safari/537.36";
|
|
9
|
+
async function downloadPdf(url) {
|
|
10
|
+
const cookies = new Map();
|
|
11
|
+
let current = url;
|
|
12
|
+
for (let i = 0; i < 8; i += 1) {
|
|
13
|
+
const headers = { "User-Agent": USER_AGENT };
|
|
14
|
+
if (cookies.size > 0) {
|
|
15
|
+
headers["Cookie"] = [...cookies].map(([k, v]) => `${k}=${v}`).join("; ");
|
|
16
|
+
}
|
|
17
|
+
const response = await fetch(current, { redirect: "manual", headers });
|
|
18
|
+
for (const raw of response.headers.getSetCookie?.() ?? []) {
|
|
19
|
+
const [pair] = raw.split(";");
|
|
20
|
+
const idx = pair.indexOf("=");
|
|
21
|
+
if (idx > 0)
|
|
22
|
+
cookies.set(pair.slice(0, idx).trim(), pair.slice(idx + 1).trim());
|
|
23
|
+
}
|
|
24
|
+
if (response.status >= 300 && response.status < 400) {
|
|
25
|
+
const location = response.headers.get("location");
|
|
26
|
+
if (!location)
|
|
27
|
+
throw new Error(`Redirect without Location header (HTTP ${response.status})`);
|
|
28
|
+
current = new URL(location, current).toString();
|
|
29
|
+
continue;
|
|
30
|
+
}
|
|
31
|
+
if (!response.ok)
|
|
32
|
+
throw new Error(`PDF download failed: HTTP ${response.status}`);
|
|
33
|
+
const buffer = Buffer.from(await response.arrayBuffer());
|
|
34
|
+
if (buffer.subarray(0, 5).toString("latin1") !== "%PDF-") {
|
|
35
|
+
throw new Error("Downloaded content is not a PDF");
|
|
36
|
+
}
|
|
37
|
+
return buffer;
|
|
38
|
+
}
|
|
39
|
+
throw new Error("Too many redirects during HR PDF download");
|
|
40
|
+
}
|
|
41
|
+
async function extractPdfItems(buffer) {
|
|
42
|
+
const pdf = await getDocumentProxy(new Uint8Array(buffer));
|
|
43
|
+
const items = [];
|
|
44
|
+
const strings = [];
|
|
45
|
+
for (let p = 1; p <= pdf.numPages; p += 1) {
|
|
46
|
+
const page = await pdf.getPage(p);
|
|
47
|
+
const content = await page.getTextContent();
|
|
48
|
+
for (const item of content.items) {
|
|
49
|
+
if (typeof item.str !== "string" || !item.str.trim() || !item.transform)
|
|
50
|
+
continue;
|
|
51
|
+
items.push({ str: item.str, x: item.transform[4], y: item.transform[5], page: p });
|
|
52
|
+
strings.push(item.str);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
return { items, strings };
|
|
56
|
+
}
|
|
57
|
+
function loadSenators(dataDir) {
|
|
58
|
+
const senatorsDir = path.join(dataDir, "sens", "senateurs");
|
|
59
|
+
const files = new Map();
|
|
60
|
+
const indexByName = new Map();
|
|
61
|
+
if (!fs.existsSync(senatorsDir))
|
|
62
|
+
return { files, indexByName };
|
|
63
|
+
for (const filePath of iterFilePaths(senatorsDir)) {
|
|
64
|
+
const senator = JSON.parse(fs.readFileSync(filePath, "utf-8"));
|
|
65
|
+
files.set(senator.matricule, senator);
|
|
66
|
+
const key = normalizeForComparison(senator.nom_usuel);
|
|
67
|
+
const list = indexByName.get(key);
|
|
68
|
+
if (list)
|
|
69
|
+
list.push(senator);
|
|
70
|
+
else
|
|
71
|
+
indexByName.set(key, [senator]);
|
|
72
|
+
}
|
|
73
|
+
return { files, indexByName };
|
|
74
|
+
}
|
|
75
|
+
async function main() {
|
|
76
|
+
const dataDir = process.argv[2];
|
|
77
|
+
assertExistingDirectory(dataDir, "dataDir");
|
|
78
|
+
// 1. Download + parse PDF
|
|
79
|
+
console.log(`Downloading HR PDF: ${DEFAULT_PDF_URL}`);
|
|
80
|
+
const buffer = await downloadPdf(DEFAULT_PDF_URL);
|
|
81
|
+
const { items, strings } = await extractPdfItems(buffer);
|
|
82
|
+
const rows = buildCollaboratorRows(items);
|
|
83
|
+
const editionDate = extractEditionDate(strings);
|
|
84
|
+
console.log(`PDF parsed: ${rows.length} collaborator rows, edition ${editionDate?.toISOString().slice(0, 10) ?? "unknown"}`);
|
|
85
|
+
if (rows.length === 0) {
|
|
86
|
+
throw new Error("No collaborator rows extracted from PDF: unexpected format, aborting without changes.");
|
|
87
|
+
}
|
|
88
|
+
// 2. Load existing senator files
|
|
89
|
+
const { indexByName } = loadSenators(dataDir);
|
|
90
|
+
if (indexByName.size === 0) {
|
|
91
|
+
throw new Error("No senator files found. Run data:download first to generate the files.");
|
|
92
|
+
}
|
|
93
|
+
// 3. Resolve PDF senators → files
|
|
94
|
+
let nbEnriched = 0;
|
|
95
|
+
let nbUnresolved = 0;
|
|
96
|
+
// Group collaborators by senator matricule
|
|
97
|
+
const collabsByMatricule = new Map();
|
|
98
|
+
const unresolvedNames = new Set();
|
|
99
|
+
for (const row of rows) {
|
|
100
|
+
const nameKey = normalizeForComparison(row.senateur.nom);
|
|
101
|
+
const candidates = indexByName.get(nameKey);
|
|
102
|
+
if (!candidates || candidates.length === 0) {
|
|
103
|
+
unresolvedNames.add(`${row.senateur.civilite} ${row.senateur.nom} ${row.senateur.prenom}`.trim());
|
|
104
|
+
nbUnresolved += 1;
|
|
105
|
+
continue;
|
|
106
|
+
}
|
|
107
|
+
// Disambiguation by first name prefix
|
|
108
|
+
const targetFirstName = normalizeForComparison(row.senateur.prenom);
|
|
109
|
+
const byFirstName = candidates.filter((c) => normalizeForComparison(c.prenom_usuel).startsWith(targetFirstName));
|
|
110
|
+
let senator = null;
|
|
111
|
+
if (candidates.length === 1) {
|
|
112
|
+
senator = candidates[0];
|
|
113
|
+
}
|
|
114
|
+
else if (byFirstName.length === 1) {
|
|
115
|
+
senator = byFirstName[0];
|
|
116
|
+
}
|
|
117
|
+
else {
|
|
118
|
+
unresolvedNames.add(`${row.senateur.civilite} ${row.senateur.nom} ${row.senateur.prenom}`.trim());
|
|
119
|
+
nbUnresolved += 1;
|
|
120
|
+
continue;
|
|
121
|
+
}
|
|
122
|
+
let collabs = collabsByMatricule.get(senator.matricule);
|
|
123
|
+
if (!collabs) {
|
|
124
|
+
collabs = [];
|
|
125
|
+
collabsByMatricule.set(senator.matricule, collabs);
|
|
126
|
+
}
|
|
127
|
+
// Deduplication
|
|
128
|
+
const nName = normalizeForComparison(row.collaborateur.nom);
|
|
129
|
+
const nFirstName = normalizeForComparison(row.collaborateur.prenom);
|
|
130
|
+
if (!collabs.some((c) => normalizeForComparison(c.nom) === nName && normalizeForComparison(c.prenom) === nFirstName)) {
|
|
131
|
+
collabs.push({
|
|
132
|
+
civilite: row.collaborateur.civilite,
|
|
133
|
+
nom: row.collaborateur.nom,
|
|
134
|
+
prenom: row.collaborateur.prenom,
|
|
135
|
+
});
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
// 4. Write collaborators into senator files
|
|
139
|
+
for (const [matricule, collaborateurs] of collabsByMatricule) {
|
|
140
|
+
const filePath = path.join(dataDir, "sens", "senateurs", `${matricule}.json`);
|
|
141
|
+
if (!fs.existsSync(filePath))
|
|
142
|
+
continue;
|
|
143
|
+
const senator = JSON.parse(fs.readFileSync(filePath, "utf-8"));
|
|
144
|
+
senator.collaborateurs = collaborateurs;
|
|
145
|
+
fs.writeFileSync(filePath, JSON.stringify(senator, null, 2) + "\n");
|
|
146
|
+
nbEnriched += 1;
|
|
147
|
+
}
|
|
148
|
+
// 5. Clean collaborators from senators absent from the PDF
|
|
149
|
+
const enrichedMatricules = new Set(collabsByMatricule.keys());
|
|
150
|
+
const { files: allFiles } = loadSenators(dataDir);
|
|
151
|
+
for (const [matricule, senator] of allFiles) {
|
|
152
|
+
if (!enrichedMatricules.has(matricule) && senator.collaborateurs) {
|
|
153
|
+
delete senator.collaborateurs;
|
|
154
|
+
const filePath = path.join(dataDir, "sens", "senateurs", `${matricule}.json`);
|
|
155
|
+
fs.writeFileSync(filePath, JSON.stringify(senator, null, 2) + "\n");
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
console.log(`${nbEnriched} senator(s) enriched with their collaborators`);
|
|
159
|
+
if (unresolvedNames.size > 0) {
|
|
160
|
+
console.warn(`${nbUnresolved} PDF senator(s) could not be matched: ${[...unresolvedNames].slice(0, 10).join(", ")}${unresolvedNames.size > 10 ? "…" : ""}`);
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
main().catch((error) => {
|
|
164
|
+
console.error(error);
|
|
165
|
+
process.exit(1);
|
|
166
|
+
});
|
|
@@ -28,18 +28,25 @@ const optionsDefinitions = [
|
|
|
28
28
|
const options = commandLineArgs(optionsDefinitions);
|
|
29
29
|
const CRI_ZIP_URL = "https://data.senat.fr/data/debats/cri.zip";
|
|
30
30
|
let exitCode = 10; // 0: some data changed, 10: no modification
|
|
31
|
+
function log(options, ...args) {
|
|
32
|
+
if (!options["silent"])
|
|
33
|
+
console.log(...args);
|
|
34
|
+
}
|
|
35
|
+
function warn(options, ...args) {
|
|
36
|
+
if (!options["silent"])
|
|
37
|
+
console.warn(...args);
|
|
38
|
+
}
|
|
31
39
|
class CompteRenduError extends Error {
|
|
32
40
|
constructor(message, url) {
|
|
33
41
|
super(`An error occurred while retrieving ${url}: ${message}`);
|
|
34
42
|
}
|
|
35
43
|
}
|
|
36
|
-
async function downloadCriZip(zipPath) {
|
|
37
|
-
|
|
38
|
-
console.log(`Downloading CRI zip ${CRI_ZIP_URL}…`);
|
|
44
|
+
async function downloadCriZip(zipPath, options) {
|
|
45
|
+
log(options, `Downloading CRI zip ${CRI_ZIP_URL}…`);
|
|
39
46
|
const response = await fetchWithRetry(CRI_ZIP_URL);
|
|
40
47
|
if (!response.ok) {
|
|
41
48
|
if (response.status === 404) {
|
|
42
|
-
|
|
49
|
+
warn(options, `CRI zip ${CRI_ZIP_URL} not found`);
|
|
43
50
|
return;
|
|
44
51
|
}
|
|
45
52
|
throw new CompteRenduError(String(response.status), CRI_ZIP_URL);
|
|
@@ -48,7 +55,7 @@ async function downloadCriZip(zipPath) {
|
|
|
48
55
|
await fs.writeFile(zipPath, buf);
|
|
49
56
|
if (!options["silent"]) {
|
|
50
57
|
const mb = (buf.length / (1024 * 1024)).toFixed(1);
|
|
51
|
-
|
|
58
|
+
log(options, `[CRI] Downloaded ${mb} MB → ${zipPath}`);
|
|
52
59
|
}
|
|
53
60
|
}
|
|
54
61
|
async function extractAndDistributeXmlBySession(zipPath, originalRoot) {
|
|
@@ -97,9 +104,9 @@ export async function retrieveCriXmlDump(dataDir, options = {}) {
|
|
|
97
104
|
const sessions = getSessionsFromStart((options["fromSession"] ?? UNDEFINED_SESSION));
|
|
98
105
|
// 1) Download ZIP global + distribut by session
|
|
99
106
|
const zipPath = path.join(dataDir, "cri.zip");
|
|
100
|
-
|
|
101
|
-
await downloadCriZip(zipPath);
|
|
102
|
-
|
|
107
|
+
log(options, "[CRI] Downloading global CRI zip…");
|
|
108
|
+
await downloadCriZip(zipPath, options);
|
|
109
|
+
log(options, "[CRI] Extracting + distributing XMLs by session…");
|
|
103
110
|
for (const session of sessions) {
|
|
104
111
|
const dir = path.join(originalRoot, String(session));
|
|
105
112
|
if (await fs.pathExists(dir)) {
|
|
@@ -110,13 +117,13 @@ export async function retrieveCriXmlDump(dataDir, options = {}) {
|
|
|
110
117
|
}
|
|
111
118
|
const n = await extractAndDistributeXmlBySession(zipPath, originalRoot);
|
|
112
119
|
if (n === 0) {
|
|
113
|
-
|
|
120
|
+
warn(options, "[CRI] No XML extracted. Archive empty or layout changed?");
|
|
114
121
|
}
|
|
115
122
|
else {
|
|
116
|
-
|
|
123
|
+
log(options, `[CRI] Distributed ${n} XML file(s) into session folders.`);
|
|
117
124
|
}
|
|
118
125
|
if (!options["parseDebats"]) {
|
|
119
|
-
|
|
126
|
+
log(options, "[CRI] parseDebats not requested → done.");
|
|
120
127
|
return;
|
|
121
128
|
}
|
|
122
129
|
for (const session of sessions) {
|
|
@@ -147,10 +154,10 @@ export async function retrieveCriXmlDump(dataDir, options = {}) {
|
|
|
147
154
|
const crPath = path.join(transformedSessionDir, fn);
|
|
148
155
|
try {
|
|
149
156
|
const cr = await fs.readJSON(crPath);
|
|
150
|
-
await linkCriEventIntoAgenda(dataDir, yyyymmdd, eventId, cr.uid, cr, session);
|
|
157
|
+
await linkCriEventIntoAgenda(dataDir, yyyymmdd, eventId, cr.uid, cr, session, options);
|
|
151
158
|
}
|
|
152
159
|
catch (e) {
|
|
153
|
-
|
|
160
|
+
warn(options, `[CR] [${session}] Could not relink existing CR into a reunion for ${yyyymmdd} event=${eventId}:`, e);
|
|
154
161
|
}
|
|
155
162
|
}
|
|
156
163
|
continue;
|
|
@@ -160,7 +167,7 @@ export async function retrieveCriXmlDump(dataDir, options = {}) {
|
|
|
160
167
|
// === Charger les events SP du jour depuis les agendas groupés ===
|
|
161
168
|
const dayEvents = await loadAgendaSpEventsForDate(dataDir, yyyymmdd, session);
|
|
162
169
|
if (dayEvents.length === 0) {
|
|
163
|
-
|
|
170
|
+
warn(options, `[CRI] [${session}] No agenda SP events found for ${yyyymmdd} → skip split/link`);
|
|
164
171
|
continue;
|
|
165
172
|
}
|
|
166
173
|
// === Lire XML + construire index DOM ===
|
|
@@ -175,14 +182,14 @@ export async function retrieveCriXmlDump(dataDir, options = {}) {
|
|
|
175
182
|
idx = new Map(order.map((el, i) => [el, i]));
|
|
176
183
|
}
|
|
177
184
|
catch (e) {
|
|
178
|
-
|
|
185
|
+
warn(options, `[CRI] [${session}] Cannot read/parse ${f}:`, e);
|
|
179
186
|
continue;
|
|
180
187
|
}
|
|
181
188
|
// === Extraire sommaire + matcher vers events agenda ===
|
|
182
189
|
const blocks = extractSommaireBlocks($, idx);
|
|
183
190
|
const intervals = buildIntervalsByAgendaEvents($, idx, order, blocks, dayEvents);
|
|
184
191
|
if (!intervals.length) {
|
|
185
|
-
|
|
192
|
+
warn(options, `[CRI] [${session}] No confident split intervals for ${yyyymmdd} → skip`);
|
|
186
193
|
continue;
|
|
187
194
|
}
|
|
188
195
|
// === Parser / écrire / linker chaque segment par event ===
|
|
@@ -191,16 +198,16 @@ export async function retrieveCriXmlDump(dataDir, options = {}) {
|
|
|
191
198
|
const outPath = path.join(transformedSessionDir, outName);
|
|
192
199
|
const cr = await parseCompteRenduIntervalFromFile(xmlPath, iv.startIndex, iv.endIndex, iv.agendaEventId);
|
|
193
200
|
if (!cr) {
|
|
194
|
-
|
|
201
|
+
warn(options, `[CRI] [${session}] Empty or no points for ${yyyymmdd} event=${iv.agendaEventId} → skip`);
|
|
195
202
|
continue;
|
|
196
203
|
}
|
|
197
204
|
await fs.ensureDir(transformedSessionDir);
|
|
198
205
|
await fs.writeJSON(outPath, cr, { spaces: 2 });
|
|
199
206
|
try {
|
|
200
|
-
await linkCriEventIntoAgenda(dataDir, yyyymmdd, iv.agendaEventId, cr.uid, cr, session);
|
|
207
|
+
await linkCriEventIntoAgenda(dataDir, yyyymmdd, iv.agendaEventId, cr.uid, cr, session, options);
|
|
201
208
|
}
|
|
202
209
|
catch (e) {
|
|
203
|
-
|
|
210
|
+
warn(options, `[CR] [${session}] Could not link CR into agenda for ${yyyymmdd} event=${iv.agendaEventId}:`, e);
|
|
204
211
|
}
|
|
205
212
|
}
|
|
206
213
|
}
|
|
@@ -218,7 +225,7 @@ function commitAndPushGit(datasetDir, options) {
|
|
|
218
225
|
}
|
|
219
226
|
}
|
|
220
227
|
}
|
|
221
|
-
async function linkCriEventIntoAgenda(dataDir, yyyymmdd, agendaEventId, crUid, cr, session) {
|
|
228
|
+
async function linkCriEventIntoAgenda(dataDir, yyyymmdd, agendaEventId, crUid, cr, session, options) {
|
|
222
229
|
const agendadDir = path.join(dataDir, AGENDA_FOLDER, DATA_TRANSFORMED_FOLDER, session.toString());
|
|
223
230
|
fs.ensureDirSync(agendadDir);
|
|
224
231
|
const dateISO = `${yyyymmdd.slice(0, 4)}-${yyyymmdd.slice(4, 6)}-${yyyymmdd.slice(6, 8)}`;
|
|
@@ -230,17 +237,17 @@ async function linkCriEventIntoAgenda(dataDir, yyyymmdd, agendaEventId, crUid, c
|
|
|
230
237
|
agenda = await fs.readJSON(agendaPath);
|
|
231
238
|
}
|
|
232
239
|
catch (e) {
|
|
233
|
-
|
|
240
|
+
warn(options, `[CR] unreadable reunion JSON → ${agendaPath} (${e})`);
|
|
234
241
|
agenda = null;
|
|
235
242
|
}
|
|
236
243
|
}
|
|
237
244
|
if (!agenda) {
|
|
238
|
-
|
|
245
|
+
warn(options, `[CR] Missing reunion file for SP event=${agendaEventId}: ${agendaPath}`);
|
|
239
246
|
return;
|
|
240
247
|
}
|
|
241
248
|
agenda.compteRenduRefUid = crUid;
|
|
242
249
|
await fs.writeJSON(agendaPath, agenda, { spaces: 2 });
|
|
243
|
-
|
|
250
|
+
log(options, `[CR] Linked CR ${crUid} → ${path.basename(agendaPath)} (event=${agendaEventId})`);
|
|
244
251
|
}
|
|
245
252
|
function buildIntervalsByAgendaEvents($, idx, order, blocks, dayEvents) {
|
|
246
253
|
const MIN_SCORE = 0.65;
|
|
@@ -348,9 +355,11 @@ function resolveTargetIndex($, idx, targetId) {
|
|
|
348
355
|
}
|
|
349
356
|
async function main() {
|
|
350
357
|
const dataDir = assertExistingDirectory(options["dataDir"], "data directory");
|
|
351
|
-
|
|
358
|
+
if (!options["silent"])
|
|
359
|
+
console.time("CRI processing time");
|
|
352
360
|
await retrieveCriXmlDump(dataDir, options);
|
|
353
|
-
|
|
361
|
+
if (!options["silent"])
|
|
362
|
+
console.timeEnd("CRI processing time");
|
|
354
363
|
}
|
|
355
364
|
main()
|
|
356
365
|
.then(() => process.exit(exitCode))
|
|
@@ -17,6 +17,10 @@ import { processBisIfNeeded, processOneReunionMatch, writeIfChanged } from "../v
|
|
|
17
17
|
const optionsDefinitions = [...commonOptions];
|
|
18
18
|
const options = commandLineArgs(optionsDefinitions);
|
|
19
19
|
let exitCode = 10; // 0: some data changed, 10: no modification
|
|
20
|
+
function log(...args) {
|
|
21
|
+
if (!options["silent"])
|
|
22
|
+
console.log(...args);
|
|
23
|
+
}
|
|
20
24
|
function shouldSkipAgenda(agenda) {
|
|
21
25
|
if (!agenda.date || !agenda.startTime)
|
|
22
26
|
return true;
|
|
@@ -113,12 +117,12 @@ async function processGroupedReunion(agenda, session, dataDir, lastByVideo) {
|
|
|
113
117
|
STATS.total++;
|
|
114
118
|
const candidates = await fetchCandidatesForAgenda(agenda, options);
|
|
115
119
|
if (!candidates) {
|
|
116
|
-
|
|
120
|
+
log(`[warn] ${agenda.uid} No candidate found for this reunion. Probably VOD not published yet.`);
|
|
117
121
|
return;
|
|
118
122
|
}
|
|
119
123
|
const match = await matchAgendaToVideo({ agenda, agendaTs: ctx.agendaTs, candidates, options });
|
|
120
124
|
if (!match) {
|
|
121
|
-
|
|
125
|
+
log(`[miss] ${agenda.uid} No match found for this reunion`);
|
|
122
126
|
return;
|
|
123
127
|
}
|
|
124
128
|
;
|
|
@@ -127,8 +131,7 @@ async function processGroupedReunion(agenda, session, dataDir, lastByVideo) {
|
|
|
127
131
|
await writeMatchArtifacts({ agenda, ctx, best, secondBest });
|
|
128
132
|
}
|
|
129
133
|
if (best && isAmbiguousTimeOriginal(agenda.events[0].timeOriginal)) {
|
|
130
|
-
|
|
131
|
-
console.log("If the time is ambiguous, update agenda startTime from matched video");
|
|
134
|
+
log("If the time is ambiguous, update agenda startTime from matched video");
|
|
132
135
|
agenda = { ...agenda, startTime: epochToParisDateTime(best.epoch)?.startTime ?? agenda.startTime };
|
|
133
136
|
}
|
|
134
137
|
// 3) Always update BEST agenda JSON from local NVS
|
|
@@ -158,7 +161,7 @@ async function processGroupedReunion(agenda, session, dataDir, lastByVideo) {
|
|
|
158
161
|
});
|
|
159
162
|
}
|
|
160
163
|
async function processAll(dataDir, sessions) {
|
|
161
|
-
|
|
164
|
+
log("Process all Agendas and fetch video's url");
|
|
162
165
|
for (const session of sessions) {
|
|
163
166
|
const lastByVideo = new Map();
|
|
164
167
|
for (const { item: agenda } of iterLoadSenatAgendas(dataDir, session)) {
|
|
@@ -186,12 +189,14 @@ async function main() {
|
|
|
186
189
|
const dataDir = assertExistingDirectory(options["dataDir"], "data directory");
|
|
187
190
|
const sessions = getSessionsFromStart((options["fromSession"] ?? UNDEFINED_SESSION));
|
|
188
191
|
const TIMER = "senat-agendas→videos processing time";
|
|
189
|
-
|
|
192
|
+
if (!options["silent"])
|
|
193
|
+
console.time(TIMER);
|
|
190
194
|
await processAll(dataDir, sessions);
|
|
191
|
-
|
|
195
|
+
if (!options["silent"])
|
|
196
|
+
console.timeEnd(TIMER);
|
|
192
197
|
const { total, accepted } = STATS;
|
|
193
198
|
const ratio = total ? ((accepted / total) * 100).toFixed(1) : "0.0";
|
|
194
|
-
|
|
199
|
+
log(`[summary] accepted=${accepted} / total=${total} (${ratio}%)`);
|
|
195
200
|
}
|
|
196
201
|
if (import.meta.url === pathToFileURL(process.argv[1]).href) {
|
|
197
202
|
main()
|