sih-br-mcp 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,837 @@
1
+ /**
2
+ * Módulo de conexão DuckDB para consulta aos cubos de dados Parquet
3
+ * Suporta múltiplos arquivos Parquet por ano (sih_causas_YYYY.parquet)
4
+ */
5
+ // Cliente DuckDB: `@duckdb/node-api` ("Node Neo"), o cliente oficial atual,
6
+ // desde 05/09/2026 (PLAN-002). Antes era o binding legado `duckdb`, que pinava
7
+ // `node-gyp ^9` em runtime e arrastava tar 6/cacache 16/glob 7 para o lock — a
8
+ // origem de 60 dos 72 alertas do Dependabot zerados em 04/09. O Neo traz
9
+ // binário pré-compilado por plataforma (optionalDependencies de
10
+ // @duckdb/node-bindings): zero node-gyp, zero tar. A API é toda de Promise;
11
+ // o funil `query()` continua sendo o único ponto que toca a conexão.
12
+ import { DuckDBInstance } from "@duckdb/node-api";
13
+ import { fileURLToPath } from "url";
14
+ import { dirname, join, resolve } from "path";
15
+ import { existsSync, mkdirSync, readdirSync } from "fs";
16
+ import { CUBES_BASE_URL, CUBES_CACHE_ENABLED, cubesCacheDir } from "../cache.js";
17
+ // Obtém diretório do projeto
18
+ const __filename = fileURLToPath(import.meta.url);
19
+ const __dirname = dirname(__filename);
20
+ const PROJECT_ROOT = join(__dirname, "..", "..");
21
+ // SIH_DATA_DIR aponta o servidor para outra pasta de cubos — é a costura que o
22
+ // smoke stdio usa para ler a fixture versionada em tests/fixtures/sih, já que
23
+ // data/*.parquet não vai para o git. Sem a variável, nada muda: data/ do projeto.
24
+ const DATA_DIR = process.env.SIH_DATA_DIR
25
+ ? resolve(process.env.SIH_DATA_DIR)
26
+ : join(PROJECT_ROOT, "data");
27
+ const TEST_DATA_DIR = join(DATA_DIR, "test");
28
+ // Singleton do banco de dados. `connecting` segura a Promise da primeira
29
+ // abertura para que chamadas concorrentes (as tools de taxa disparam várias
30
+ // consultas) não criem duas instâncias — o legado tinha essa corrida.
31
+ let instance = null;
32
+ let conn = null;
33
+ let connecting = null;
34
+ // Detecta se estamos em modo teste (data/test tem arquivos)
35
+ let dataDirectory = null;
36
+ /**
37
+ * Determina qual diretório de dados usar.
38
+ * - Padrão: data/ (dados reais)
39
+ * - Teste: data/test/ (somente se SIH_TEST_MODE=1 estiver definido)
40
+ */
41
+ /**
42
+ * A pasta de cubos CONFIGURADA (SIH_DATA_DIR ou data/ do projeto), exista
43
+ * cubo nela ou não. Quem precisa só do sidecar de proveniência — o frescor
44
+ * num checkout limpo, onde data/*.parquet é gitignored e só o JSON está
45
+ * versionado — lê daqui; quem precisa de Parquet usa getDataDirectory().
46
+ */
47
+ export function configuredDataDirectory() {
48
+ return DATA_DIR;
49
+ }
50
+ export function getDataDirectory() {
51
+ if (dataDirectory)
52
+ return dataDirectory;
53
+ // Usa data/test/ somente em modo teste explícito
54
+ if (process.env.SIH_TEST_MODE === "1" && existsSync(TEST_DATA_DIR)) {
55
+ const testFiles = readdirSync(TEST_DATA_DIR).filter((f) => f.endsWith(".parquet"));
56
+ if (testFiles.length > 0) {
57
+ console.error(`[DuckDB] Modo TESTE: usando ${TEST_DATA_DIR}`);
58
+ dataDirectory = TEST_DATA_DIR;
59
+ return dataDirectory;
60
+ }
61
+ }
62
+ // Padrão: dados reais em data/
63
+ if (existsSync(DATA_DIR)) {
64
+ const prodFiles = readdirSync(DATA_DIR).filter((f) => f.startsWith("sih_") && f.endsWith(".parquet"));
65
+ if (prodFiles.length > 0) {
66
+ console.error(`[DuckDB] Usando dados em ${DATA_DIR}`);
67
+ dataDirectory = DATA_DIR;
68
+ return dataDirectory;
69
+ }
70
+ }
71
+ // Sem cubo na pasta do projeto (instalação pelo npm, que não embarca cubo):
72
+ // a pasta passa a ser o cache local, que `src/cache.ts` enche sob demanda a
73
+ // partir do canal público (data.sidneybissoli.com/sih/cubos/). A pasta pode
74
+ // estar vazia agora; os handlers chamam ensureYears() antes de consultar.
75
+ if (CUBES_CACHE_ENABLED) {
76
+ const cacheDir = cubesCacheDir();
77
+ mkdirSync(cacheDir, { recursive: true });
78
+ console.error(`[DuckDB] Sem cubos em ${DATA_DIR}; usando o cache ${cacheDir} (baixa de ${CUBES_BASE_URL} sob demanda)`);
79
+ dataDirectory = cacheDir;
80
+ return dataDirectory;
81
+ }
82
+ throw new Error(`Nenhum arquivo SIH Parquet encontrado em ${DATA_DIR} e o cache está desligado (SIH_CUBES_CACHE=off). Execute os scripts R de agregação ou baixe os cubos de ${CUBES_BASE_URL}.`);
83
+ }
84
+ /**
85
+ * Retorna o padrão glob para um tipo de cubo
86
+ */
87
+ export function getParquetPattern(cube) {
88
+ const dir = getDataDirectory();
89
+ return join(dir, `sih_${cube}_*.parquet`).replace(/\\/g, "/");
90
+ }
91
+ /**
92
+ * Lista os anos disponíveis nos dados
93
+ */
94
+ export function getAvailableYears() {
95
+ const dir = getDataDirectory();
96
+ const files = readdirSync(dir).filter((f) => f.startsWith("sih_causas_"));
97
+ const years = files
98
+ .map((f) => {
99
+ const match = f.match(/sih_causas_(\d{4})\.parquet/);
100
+ return match ? parseInt(match[1]) : null;
101
+ })
102
+ .filter((y) => y !== null)
103
+ .sort((a, b) => a - b);
104
+ return years;
105
+ }
106
+ /**
107
+ * Inicializa conexão com DuckDB
108
+ */
109
+ export async function getDatabase() {
110
+ if (conn) {
111
+ return conn;
112
+ }
113
+ if (connecting) {
114
+ return connecting;
115
+ }
116
+ connecting = (async () => {
117
+ // Usa banco em memória (consultas diretas aos Parquet)
118
+ try {
119
+ instance = await DuckDBInstance.create(":memory:");
120
+ }
121
+ catch (err) {
122
+ const message = err instanceof Error ? err.message : String(err);
123
+ throw new Error(`Erro ao criar banco DuckDB: ${message}`);
124
+ }
125
+ const connection = await instance.connect();
126
+ // Configura DuckDB para melhor performance com Parquet
127
+ try {
128
+ await connection.run("SET threads TO 4");
129
+ }
130
+ catch {
131
+ console.error("Aviso: não foi possível configurar threads");
132
+ }
133
+ conn = connection;
134
+ return connection;
135
+ })();
136
+ try {
137
+ return await connecting;
138
+ }
139
+ finally {
140
+ connecting = null;
141
+ }
142
+ }
143
+ /**
144
+ * Fecha conexão com o banco
145
+ */
146
+ export async function closeDatabase() {
147
+ if (conn) {
148
+ conn.closeSync();
149
+ conn = null;
150
+ }
151
+ if (instance) {
152
+ instance.closeSync();
153
+ instance = null;
154
+ }
155
+ dataDirectory = null; // Reset para próxima conexão
156
+ }
157
+ /**
158
+ * Converte BigInt para Number em objetos aninhados
159
+ */
160
+ function convertBigIntToNumber(obj) {
161
+ if (obj === null || obj === undefined) {
162
+ return obj;
163
+ }
164
+ if (typeof obj === "bigint") {
165
+ return Number(obj);
166
+ }
167
+ if (Array.isArray(obj)) {
168
+ return obj.map(convertBigIntToNumber);
169
+ }
170
+ if (typeof obj === "object") {
171
+ const result = {};
172
+ for (const [key, value] of Object.entries(obj)) {
173
+ result[key] = convertBigIntToNumber(value);
174
+ }
175
+ return result;
176
+ }
177
+ return obj;
178
+ }
179
+ /**
180
+ * Executa query SQL e retorna resultados
181
+ */
182
+ export async function query(sql) {
183
+ const connection = await getDatabase();
184
+ let rows;
185
+ try {
186
+ const result = await connection.runAndReadAll(sql);
187
+ // `getRowObjectsJS()` e não `getRowObjects()`: o segundo devolve DECIMAL
188
+ // como {width, scale, value} e DATE como {days} (classes do Neo), que
189
+ // virariam objetos no JSON da ferramenta; o primeiro entrega tipos JS
190
+ // nativos (DECIMAL → number, DATE → ISO). Medido em 05/09/2026. Hoje os
191
+ // cubos só têm INTEGER/DOUBLE/VARCHAR/BOOLEAN, mas o funil é um só e a
192
+ // defesa fica aqui. BIGINT e HUGEINT (SUM de INTEGER) continuam chegando
193
+ // como bigint nos dois leitores — é o que `convertBigIntToNumber` trata.
194
+ rows = result.getRowObjectsJS();
195
+ }
196
+ catch (err) {
197
+ const message = err instanceof Error ? err.message : String(err);
198
+ throw new Error(`Erro na query: ${message}\nSQL: ${sql}`);
199
+ }
200
+ // Converte BigInt para Number para serialização JSON
201
+ return convertBigIntToNumber(rows);
202
+ }
203
+ // =============================================================================
204
+ // FUNÇÕES DE QUERY ESPECÍFICAS
205
+ // =============================================================================
206
+ /**
207
+ * Constrói cláusula WHERE a partir de filtros
208
+ */
209
+ function buildWhereClause(filters) {
210
+ const conditions = [];
211
+ if ("years" in filters && filters.years && filters.years.length > 0) {
212
+ conditions.push(`year IN (${filters.years.join(", ")})`);
213
+ }
214
+ if ("months" in filters && filters.months && filters.months.length > 0) {
215
+ conditions.push(`month IN (${filters.months.join(", ")})`);
216
+ }
217
+ if (filters.ufs && filters.ufs.length > 0) {
218
+ const ufs = filters.ufs.map((u) => `'${u}'`).join(", ");
219
+ conditions.push(`uf IN (${ufs})`);
220
+ }
221
+ if ("municipalityCodes" in filters &&
222
+ filters.municipalityCodes &&
223
+ filters.municipalityCodes.length > 0) {
224
+ const codes = filters.municipalityCodes.map((c) => `'${c}'`).join(", ");
225
+ conditions.push(`municipality_code IN (${codes})`);
226
+ }
227
+ if ("cidChapters" in filters &&
228
+ filters.cidChapters &&
229
+ filters.cidChapters.length > 0) {
230
+ conditions.push(`cid_chapter IN (${filters.cidChapters.join(", ")})`);
231
+ }
232
+ if ("cidGroups" in filters && filters.cidGroups && filters.cidGroups.length > 0) {
233
+ const groups = filters.cidGroups.map((g) => `'${g}'`).join(", ");
234
+ conditions.push(`cid_group IN (${groups})`);
235
+ }
236
+ if (filters.sex) {
237
+ conditions.push(`sex = '${filters.sex}'`);
238
+ }
239
+ if (filters.ageMin !== undefined) {
240
+ conditions.push(`age >= ${filters.ageMin}`);
241
+ }
242
+ if (filters.ageMax !== undefined) {
243
+ conditions.push(`age <= ${filters.ageMax}`);
244
+ }
245
+ if (filters.races && filters.races.length > 0) {
246
+ const races = filters.races.map((r) => `'${r}'`).join(", ");
247
+ conditions.push(`race IN (${races})`);
248
+ }
249
+ if ("isCsap" in filters && filters.isCsap !== undefined) {
250
+ conditions.push(`is_csap = ${filters.isCsap}`);
251
+ }
252
+ if ("csapGroups" in filters && filters.csapGroups && filters.csapGroups.length > 0) {
253
+ const groups = filters.csapGroups.map((g) => `'${g}'`).join(", ");
254
+ conditions.push(`csap_group IN (${groups})`);
255
+ }
256
+ return conditions.length > 0 ? conditions.join(" AND ") : "1=1";
257
+ }
258
+ /**
259
+ * Expressão de uma coluna de agrupamento no SELECT e no GROUP BY. `cid_revision`
260
+ * só existe nos cubos do builder >= 2.5.0: os cubos são abertos por glob com
261
+ * `union_by_name`, então um cubo anterior (cache local misto) a traz NULA — e
262
+ * nula significa CID-10 (a coluna é constante 10 em 1998+). A defesa vale só
263
+ * para caches mistos; a série publicada é toda 2.5.0.
264
+ */
265
+ /**
266
+ * Soma de `value` como DECIMAL(18,2) e não como DOUBLE: o SUM paralelo de
267
+ * doubles do DuckDB muda o último dígito conforme a ordem em que as threads
268
+ * terminam (golden de 07/09/2026: 9746572.610000005 numa execução,
269
+ * 9746572.61000001 na seguinte); em decimal a soma é exata e determinística.
270
+ * `value` é dinheiro com 2 casas — nada se perde.
271
+ */
272
+ function groupSelect(column) {
273
+ return column === "cid_revision" ? "COALESCE(cid_revision, 10) AS cid_revision" : column;
274
+ }
275
+ function groupKey(column) {
276
+ return column === "cid_revision" ? "COALESCE(cid_revision, 10)" : column;
277
+ }
278
+ /**
279
+ * ORDER BY determinístico para consultas agrupadas.
280
+ *
281
+ * Sem isto, duas chamadas iguais devolvem respostas diferentes: GROUP BY sem
282
+ * ORDER BY sai na ordem em que as 4 threads do DuckDB terminam, e ORDER BY
283
+ * por métrica empata (dois grupos com o mesmo SUM) sem critério de desempate —
284
+ * medido em 05/09/2026 sobre a fixture: `compare_regions` dava posição 5 a GO
285
+ * numa chamada e a PB na seguinte, `rank_csap_groups` trocava g01 e g13, e
286
+ * qualquer `limit` em cima disso devolvia SUBCONJUNTOS diferentes. O golden
287
+ * das ferramentas (scripts/golden-tools.mjs) foi o que expôs; o conserto é
288
+ * aqui, no funil, e não tool a tool: toda coluna de agrupamento que o chamador
289
+ * não ordenou explicitamente entra como desempate, na ordem do GROUP BY.
290
+ */
291
+ function buildOrderBy(orderBy, groupBy) {
292
+ const mentioned = new Set((orderBy ?? "")
293
+ .split(",")
294
+ .map((part) => part.trim().split(/\s+/)[0])
295
+ .filter(Boolean));
296
+ const tieBreak = groupBy.filter((column) => !mentioned.has(column));
297
+ const parts = [orderBy, ...tieBreak].filter((part) => !!part);
298
+ return parts.length > 0 ? ` ORDER BY ${parts.join(", ")}` : "";
299
+ }
300
+ /**
301
+ * Query no cubo de causas com agregação flexível
302
+ */
303
+ export async function queryCausas(options) {
304
+ const pattern = getParquetPattern("causas");
305
+ const whereClause = buildWhereClause(options.filters || {});
306
+ // Métricas a calcular
307
+ const metricsMap = {
308
+ n: "SUM(n) as n_hospitalizations",
309
+ days: "SUM(days) as total_days",
310
+ value: "SUM(CAST(value AS DECIMAL(18,2))) as total_value",
311
+ deaths: "SUM(deaths) as deaths",
312
+ };
313
+ const metrics = options.metrics || ["n"];
314
+ const selectMetrics = metrics.map((m) => metricsMap[m]).join(", ");
315
+ // Group by
316
+ const groupBy = options.groupBy || [];
317
+ const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
318
+ const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
319
+ let sql = `
320
+ SELECT ${selectGroups}${selectMetrics}
321
+ FROM read_parquet('${pattern}', union_by_name = true)
322
+ WHERE ${whereClause}
323
+ ${groupByClause}
324
+ `;
325
+ sql += buildOrderBy(options.orderBy, groupBy);
326
+ if (options.limit) {
327
+ sql += ` LIMIT ${options.limit}`;
328
+ }
329
+ return query(sql);
330
+ }
331
+ /**
332
+ * Query no cubo de ICSAP com agregação flexível
333
+ */
334
+ export async function queryIcsap(options) {
335
+ // 0.9.0: toda agregação do cubo ICSAP passa por icsapAggregateSql (o
336
+ // denominador por estratos distintos); `metrics` fica só pela assinatura —
337
+ // as seis medidas saem sempre.
338
+ return query(icsapAggregateSql(options));
339
+ }
340
+ /** Chaves do ESTRATO do cubo ICSAP: `n_total` é o total do estrato, repetido em cada linha dele. */
341
+ // `exclusion` (builder >= 2.6.0) é chave do estrato: sem ela, dois estratos que
342
+ // só diferem no motivo de exclusão e têm o mesmo n_total colapsariam no
343
+ // DISTINCT (2023/RR com universe = "all" dava 48.122 em vez de 48.480).
344
+ const ICSAP_STRATUM_KEYS = ["year", "uf", "municipality_code", "cid_revision", "sex", "age", "race", "exclusion"];
345
+ function universeCondition(universe) {
346
+ return (universe ?? "csapaih") === "csapaih" ? "AND exclusion IS NULL" : "";
347
+ }
348
+ function icsapAggregateSql(options) {
349
+ const pattern = getParquetPattern("icsap");
350
+ const { csapGroups, ...strataFilters } = options.filters ?? {};
351
+ const whereStrata = `${buildWhereClause(strataFilters)} ${universeCondition(options.universe)}`;
352
+ const csapCond = csapGroups && csapGroups.length > 0 ? `AND csap_group IN (${csapGroups.map((g) => `'${g}'`).join(", ")})` : "";
353
+ const groupBy = options.groupBy ?? [];
354
+ const byGroup = groupBy.includes("csap_group");
355
+ const strataGroup = groupBy.filter((g) => g !== "csap_group");
356
+ const groupClause = (cols) => (cols.length > 0 ? `GROUP BY ${cols.join(", ")}` : "");
357
+ const joinOn = strataGroup.length > 0 ? strataGroup.map((k) => `icsap.${k} IS NOT DISTINCT FROM total.${k}`).join(" AND ") : "TRUE";
358
+ // Sem csap_group no agrupamento, a linha-mestre é a de total: um grupo com
359
+ // estratos mas sem ICSAP ainda aparece, com 0. Com csap_group, é a de ICSAP.
360
+ const fromClause = byGroup ? `icsap JOIN total ON ${joinOn}` : `total LEFT JOIN icsap ON ${joinOn}`;
361
+ const selectCols = groupBy
362
+ .map((g) => (byGroup || g === "csap_group" ? `icsap.${g} AS ${g}` : `total.${g} AS ${g}`))
363
+ .join(", ");
364
+ const strataKeys = ICSAP_STRATUM_KEYS.map(groupSelect);
365
+ let sql = `
366
+ WITH linhas AS (
367
+ SELECT * FROM read_parquet('${pattern}', union_by_name = true)
368
+ WHERE ${whereStrata}
369
+ ),
370
+ estratos AS (SELECT DISTINCT ${strataKeys.join(", ")}, n_total FROM linhas),
371
+ total AS (
372
+ SELECT ${strataGroup.length > 0 ? strataGroup.join(", ") + ", " : ""}SUM(n_total) AS n_total
373
+ FROM estratos ${groupClause(strataGroup)}
374
+ ),
375
+ icsap AS (
376
+ SELECT ${groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : ""}SUM(n) AS n_icsap, SUM(days) AS total_days, SUM(CAST(value AS DECIMAL(18,2))) AS total_value, SUM(deaths) AS deaths
377
+ FROM linhas
378
+ WHERE csap_group IS NOT NULL ${csapCond}
379
+ ${groupClause(groupBy.map(groupKey))}
380
+ )
381
+ SELECT ${selectCols}${groupBy.length > 0 ? ", " : ""}
382
+ COALESCE(icsap.n_icsap, 0) AS n_icsap,
383
+ total.n_total AS n_total,
384
+ ROUND(COALESCE(icsap.n_icsap, 0) * 100.0 / NULLIF(total.n_total, 0), 2) AS icsap_percentage,
385
+ COALESCE(icsap.total_days, 0) AS total_days,
386
+ COALESCE(icsap.total_value, 0) AS total_value,
387
+ COALESCE(icsap.deaths, 0) AS deaths
388
+ FROM ${fromClause}
389
+ `;
390
+ sql += buildOrderBy(options.orderBy, groupBy);
391
+ if (options.limit)
392
+ sql += ` LIMIT ${options.limit}`;
393
+ return sql;
394
+ }
395
+ /**
396
+ * Query no cubo de séries temporais
397
+ */
398
+ export async function querySeries(options) {
399
+ const pattern = getParquetPattern("series");
400
+ const conditions = [];
401
+ if (options.filters) {
402
+ if (options.filters.yearMonthStart) {
403
+ conditions.push(`year_month >= '${options.filters.yearMonthStart}'`);
404
+ }
405
+ if (options.filters.yearMonthEnd) {
406
+ conditions.push(`year_month <= '${options.filters.yearMonthEnd}'`);
407
+ }
408
+ if (options.filters.ufs && options.filters.ufs.length > 0) {
409
+ const ufs = options.filters.ufs.map((u) => `'${u}'`).join(", ");
410
+ conditions.push(`uf IN (${ufs})`);
411
+ }
412
+ if (options.filters.cidChapters && options.filters.cidChapters.length > 0) {
413
+ conditions.push(`cid_chapter IN (${options.filters.cidChapters.join(", ")})`);
414
+ }
415
+ }
416
+ const whereClause = conditions.length > 0 ? conditions.join(" AND ") : "1=1";
417
+ // Group by
418
+ const groupBy = options.groupBy || [];
419
+ const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
420
+ const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
421
+ let sql = `
422
+ SELECT ${selectGroups}SUM(n) as n, SUM(deaths) as deaths
423
+ FROM read_parquet('${pattern}', union_by_name = true)
424
+ WHERE ${whereClause}
425
+ ${groupByClause}
426
+ `;
427
+ sql += buildOrderBy(options.orderBy, groupBy);
428
+ return query(sql);
429
+ }
430
+ /**
431
+ * Query para ranking de grupos CSAP
432
+ */
433
+ export async function rankCsapGroups(options) {
434
+ const pattern = getParquetPattern("icsap");
435
+ const whereClause = `${buildWhereClause(options.filters || {})} ${universeCondition(options.universe)}`;
436
+ const metric = options.metric || "n";
437
+ const metricsMap = {
438
+ n: "SUM(n)",
439
+ days: "SUM(days)",
440
+ value: "SUM(CAST(value AS DECIMAL(18,2)))",
441
+ deaths: "SUM(deaths)",
442
+ };
443
+ const sql = `
444
+ SELECT
445
+ csap_group,
446
+ ${metricsMap[metric]} as metric_value,
447
+ SUM(n) as n_hospitalizations,
448
+ SUM(days) as total_days,
449
+ SUM(CAST(value AS DECIMAL(18,2))) as total_value,
450
+ SUM(deaths) as deaths
451
+ FROM read_parquet('${pattern}', union_by_name = true)
452
+ WHERE ${whereClause} AND csap_group IS NOT NULL
453
+ GROUP BY csap_group
454
+ ORDER BY metric_value DESC, csap_group
455
+ ${options.limit ? `LIMIT ${options.limit}` : ""}
456
+ `;
457
+ return query(sql);
458
+ }
459
+ /**
460
+ * Calcula indicadores de ICSAP (percentual)
461
+ */
462
+ export async function calculateIcsapIndicators(options) {
463
+ // 0.9.0: denominador por estratos distintos (ver icsapAggregateSql)
464
+ return query(icsapAggregateSql({ filters: options.filters, groupBy: options.groupBy, universe: options.universe }));
465
+ }
466
+ // =============================================================================
467
+ // FUNÇÕES DE POPULAÇÃO
468
+ // =============================================================================
469
+ /**
470
+ * Onde os arquivos de população vivem (0.12.0, sih:populacao-no-canal): a
471
+ * pasta configurada (SIH_DATA_DIR ou data/ do projeto; nunca data/test/) tem
472
+ * precedência sempre que tiver pop_uf.parquet — é o caso da fixture do smoke
473
+ * e do golden e de quem gerou a população à mão; sem nada nela e com o cache
474
+ * ligado, o cache local, que `ensurePopulation()` (src/cache.ts) enche a
475
+ * partir do canal sih/cubos/ (pop_uf, pop_uf_agregado, pop_municipios +
476
+ * pop_provenance.json, assinados no bloco `population` do manifesto).
477
+ */
478
+ export function getPopulationDir() {
479
+ if (existsSync(join(DATA_DIR, "pop_uf.parquet")))
480
+ return DATA_DIR;
481
+ if (CUBES_CACHE_ENABLED)
482
+ return cubesCacheDir();
483
+ return DATA_DIR;
484
+ }
485
+ /** Padrão de arquivo para dados populacionais (pasta de getPopulationDir()). */
486
+ export function getPopulationPattern(type) {
487
+ const filename = type === "municipios" ? "pop_municipios.parquet" :
488
+ type === "uf" ? "pop_uf.parquet" :
489
+ "pop_uf_agregado.parquet";
490
+ return join(getPopulationDir(), filename).replace(/\\/g, "/");
491
+ }
492
+ /** Verifica se algum arquivo de população está disponível em getPopulationDir(). */
493
+ export function hasPopulationData() {
494
+ try {
495
+ const dir = getPopulationDir();
496
+ if (!existsSync(dir))
497
+ return false;
498
+ const files = readdirSync(dir);
499
+ return files.some(f => f.startsWith("pop_") && f.endsWith(".parquet"));
500
+ }
501
+ catch {
502
+ return false;
503
+ }
504
+ }
505
+ /** Mensagem única para "sem denominador" — aponta o canal, não um script R. */
506
+ export const POPULATION_MISSING_MESSAGE = `Dados populacionais não disponíveis: nem pop_uf.parquet na pasta de dados nem no cache local. Com o cache ligado o servidor baixa pop_uf, pop_uf_agregado e pop_municipios de ${CUBES_BASE_URL} (bloco population do manifest.json); confira a rede ou a variável SIH_CUBES_CACHE.`;
507
+ /**
508
+ * Constrói cláusula WHERE para população UF
509
+ */
510
+ function buildPopulationWhereClause(filters) {
511
+ const conditions = [];
512
+ if (filters.years && filters.years.length > 0) {
513
+ conditions.push(`year IN (${filters.years.join(", ")})`);
514
+ }
515
+ if (filters.ufs && filters.ufs.length > 0) {
516
+ const ufs = filters.ufs.map((u) => `'${u}'`).join(", ");
517
+ conditions.push(`uf IN (${ufs})`);
518
+ }
519
+ if (filters.sex) {
520
+ conditions.push(`sex = '${filters.sex}'`);
521
+ }
522
+ if (filters.ageMin !== undefined) {
523
+ conditions.push(`age >= ${filters.ageMin}`);
524
+ }
525
+ if (filters.ageMax !== undefined) {
526
+ conditions.push(`age <= ${filters.ageMax}`);
527
+ }
528
+ return conditions.length > 0 ? conditions.join(" AND ") : "1=1";
529
+ }
530
+ /**
531
+ * Query população por UF (dados com idade simples)
532
+ */
533
+ export async function queryPopulationUf(options) {
534
+ const pattern = getPopulationPattern("uf");
535
+ const whereClause = buildPopulationWhereClause(options.filters || {});
536
+ const groupBy = options.groupBy || [];
537
+ const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
538
+ const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
539
+ const sql = `
540
+ SELECT
541
+ ${selectGroups}
542
+ SUM(population) as population
543
+ FROM read_parquet('${pattern}', union_by_name = true)
544
+ WHERE ${whereClause}
545
+ ${groupByClause}${buildOrderBy(undefined, groupBy)}
546
+ `;
547
+ return query(sql);
548
+ }
549
+ /**
550
+ * Query população agregada por UF (para anos 1991-1999, com faixa etária)
551
+ */
552
+ export async function queryPopulationUfAgregado(options) {
553
+ const pattern = getPopulationPattern("uf_agregado");
554
+ const conditions = [];
555
+ if (options.years && options.years.length > 0) {
556
+ conditions.push(`year IN (${options.years.join(", ")})`);
557
+ }
558
+ if (options.ageGroups && options.ageGroups.length > 0) {
559
+ // As linhas de idade ignorada (age_group nulo; POPBR "I000") ficam fora de
560
+ // qualquer recorte etário — e dentro do total, que só bate com a
561
+ // estimativa do IBGE com elas (Brasil 1993: 151.556.521).
562
+ conditions.push(`age_group IN (${options.ageGroups.map((g) => `'${g}'`).join(", ")})`);
563
+ }
564
+ if (options.ufs && options.ufs.length > 0) {
565
+ const ufs = options.ufs.map((u) => `'${u}'`).join(", ");
566
+ conditions.push(`uf IN (${ufs})`);
567
+ }
568
+ if (options.sex) {
569
+ conditions.push(`sex = '${options.sex}'`);
570
+ }
571
+ const whereClause = conditions.length > 0 ? conditions.join(" AND ") : "1=1";
572
+ const groupBy = options.groupBy || [];
573
+ const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
574
+ const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
575
+ const sql = `
576
+ SELECT
577
+ ${selectGroups}
578
+ SUM(population) as population
579
+ FROM read_parquet('${pattern}', union_by_name = true)
580
+ WHERE ${whereClause}
581
+ ${groupByClause}${buildOrderBy(undefined, groupBy)}
582
+ `;
583
+ return query(sql);
584
+ }
585
+ /**
586
+ * Verifica se pop_uf.parquet existe
587
+ */
588
+ function hasPopUf() {
589
+ const path = join(getPopulationDir(), "pop_uf.parquet");
590
+ return existsSync(path);
591
+ }
592
+ /** Intervalo de anos de um arquivo de população, lido do próprio arquivo. */
593
+ async function yearRangeOf(file) {
594
+ const path = join(getPopulationDir(), file);
595
+ if (!existsSync(path))
596
+ return null;
597
+ const rows = await query(`SELECT CAST(min(year) AS INTEGER) AS first_year, CAST(max(year) AS INTEGER) AS last_year FROM read_parquet('${path.replace(/\\/g, "/")}')`);
598
+ const r = rows[0];
599
+ return r && r.first_year != null && r.last_year != null
600
+ ? { first_year: Number(r.first_year), last_year: Number(r.last_year) }
601
+ : null;
602
+ }
603
+ /** Faixas etárias quinquenais de pop_uf_agregado.parquet (1991–1999), na ordem. */
604
+ export const AGGREGATED_AGE_GROUPS = [
605
+ "0-4", "5-9", "10-14", "15-19", "20-24", "25-29", "30-34", "35-39", "40-44",
606
+ "45-49", "50-54", "55-59", "60-64", "65-69", "70-74", "75-79", "80 e +",
607
+ ];
608
+ /**
609
+ * Faixas de pop_uf_agregado.parquet que cobrem EXATAMENTE [ageMin, ageMax].
610
+ * Devolve null quando o intervalo não cai nos limites das faixas (ex.: 0–14
611
+ * serve, 0–17 não): antes de 2000 só há população por faixa quinquenal, e uma
612
+ * taxa com denominador aproximado seria número errado com cara de certo.
613
+ * Sem filtro de idade devolve undefined (todas as faixas).
614
+ */
615
+ export function aggregatedAgeGroupsFor(ageMin, ageMax) {
616
+ if (ageMin === undefined && ageMax === undefined)
617
+ return undefined;
618
+ const lo = ageMin ?? 0;
619
+ const hi = ageMax ?? Infinity;
620
+ if (lo % 5 !== 0)
621
+ return null;
622
+ if (hi !== Infinity && hi < 79 && (hi + 1) % 5 !== 0)
623
+ return null;
624
+ const out = [];
625
+ for (const g of AGGREGATED_AGE_GROUPS) {
626
+ const start = parseInt(g, 10);
627
+ const end = g === "80 e +" ? Infinity : start + 4;
628
+ if (start >= lo && end <= hi)
629
+ out.push(g);
630
+ else if (start >= lo && g === "80 e +" && hi >= 80)
631
+ out.push(g);
632
+ }
633
+ if (hi !== Infinity && hi >= 80 && hi < Infinity && !out.includes("80 e +"))
634
+ return null;
635
+ return out.length > 0 ? out : null;
636
+ }
637
+ let popCoverageCache;
638
+ /**
639
+ * Esquece a cobertura memoizada — chamada depois que ensurePopulation() baixa
640
+ * algum arquivo: uma chamada anterior pode ter memoizado `null` (sem
641
+ * população) antes do download.
642
+ */
643
+ export function resetPopulationCoverageCache() {
644
+ popCoverageCache = undefined;
645
+ }
646
+ /**
647
+ * Cobertura populacional, lida dos próprios arquivos — nunca fixada no código.
648
+ * `first_year`/`last_year` é a união de pop_uf.parquet (idade simples, 2000+;
649
+ * a regra do CONTEXT.md diz que vai até o último ano de cubo FECHADO do SIH,
650
+ * e quem a cumpre é o build-population.R do healthbr-data) e de pop_uf_agregado.parquet
651
+ * (faixa etária, 1991–1999; sih:taxas-1992-1999). Antes de 2000 a taxa por
652
+ * idade só sai nas faixas do arquivo (aggregatedAgeGroupsFor). Devolve null
653
+ * sem nenhum dos dois arquivos.
654
+ */
655
+ export async function getPopulationCoverage() {
656
+ if (popCoverageCache !== undefined)
657
+ return popCoverageCache;
658
+ const detailed = await yearRangeOf("pop_uf.parquet");
659
+ const aggregated = await yearRangeOf("pop_uf_agregado.parquet");
660
+ if (!detailed && !aggregated) {
661
+ popCoverageCache = null;
662
+ return null;
663
+ }
664
+ const ranges = [detailed, aggregated].filter((r) => !!r);
665
+ popCoverageCache = {
666
+ first_year: Math.min(...ranges.map((r) => r.first_year)),
667
+ last_year: Math.max(...ranges.map((r) => r.last_year)),
668
+ detailed: detailed ? { ...detailed, source: "pop_uf.parquet", age: "idade simples" } : null,
669
+ aggregated: aggregated
670
+ ? { ...aggregated, source: "pop_uf_agregado.parquet", age: "faixa etária quinquenal", age_groups: AGGREGATED_AGE_GROUPS }
671
+ : null,
672
+ };
673
+ return popCoverageCache;
674
+ }
675
+ /** Qual arquivo de população serve a um ano (null = nenhum). */
676
+ export async function populationSourceFor(year) {
677
+ const c = await getPopulationCoverage();
678
+ if (!c)
679
+ return null;
680
+ if (c.detailed && year >= c.detailed.first_year && year <= c.detailed.last_year)
681
+ return "detailed";
682
+ if (c.aggregated && year >= c.aggregated.first_year && year <= c.aggregated.last_year)
683
+ return "aggregated";
684
+ return null;
685
+ }
686
+ /**
687
+ * Compatibilidade: o intervalo aceito pelas taxas (união dos dois arquivos).
688
+ * Devolve null sem população.
689
+ */
690
+ export async function getPopulationYearRange() {
691
+ const c = await getPopulationCoverage();
692
+ return c ? { first_year: c.first_year, last_year: c.last_year } : null;
693
+ }
694
+ /**
695
+ * Query população de pop_municipios.parquet agregando por UF
696
+ */
697
+ async function queryPopulationFromMunicipios(options) {
698
+ const pattern = getPopulationPattern("municipios");
699
+ const conditions = [];
700
+ if (options.years && options.years.length > 0) {
701
+ conditions.push(`year IN (${options.years.join(", ")})`);
702
+ }
703
+ if (options.ufs && options.ufs.length > 0) {
704
+ const ufs = options.ufs.map((u) => `'${u}'`).join(", ");
705
+ conditions.push(`uf IN (${ufs})`);
706
+ }
707
+ if (options.sex) {
708
+ conditions.push(`sex = '${options.sex}'`);
709
+ }
710
+ const whereClause = conditions.length > 0 ? conditions.join(" AND ") : "1=1";
711
+ const groupBy = options.groupBy || [];
712
+ const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
713
+ const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
714
+ const sql = `
715
+ SELECT
716
+ ${selectGroups}
717
+ SUM(population) as population
718
+ FROM read_parquet('${pattern}', union_by_name = true)
719
+ WHERE ${whereClause}
720
+ ${groupByClause}${buildOrderBy(undefined, groupBy)}
721
+ `;
722
+ return query(sql);
723
+ }
724
+ /**
725
+ * Obtém população total para um filtro específico.
726
+ * Hierarquia de fontes:
727
+ * 1. pop_uf.parquet (idade simples, anos >= 2000)
728
+ * 2. pop_uf_agregado.parquet (faixa etária, anos 1991-1999)
729
+ * 3. pop_municipios.parquet (fallback, agregando por UF)
730
+ */
731
+ export async function getPopulation(options) {
732
+ if (!hasPopulationData()) {
733
+ throw new Error(POPULATION_MISSING_MESSAGE);
734
+ }
735
+ // Tenta pop_uf.parquet para anos >= 2000 (idade simples)
736
+ if (options.year >= 2000 && hasPopUf()) {
737
+ const result = await queryPopulationUf({
738
+ filters: {
739
+ years: [options.year],
740
+ ufs: options.uf,
741
+ sex: options.sex,
742
+ ageMin: options.ageMin,
743
+ ageMax: options.ageMax,
744
+ },
745
+ });
746
+ if (result[0]?.population)
747
+ return result[0].population;
748
+ }
749
+ // pop_uf_agregado.parquet para anos < 2000 (sih:taxas-1992-1999): idade só
750
+ // nas faixas do arquivo — intervalo fora dos limites das faixas é erro, não
751
+ // aproximação; e com filtro de idade não há fallback para os municípios,
752
+ // que não têm idade.
753
+ if (options.year < 2000) {
754
+ const ageGroups = aggregatedAgeGroupsFor(options.ageMin, options.ageMax);
755
+ if (ageGroups === null) {
756
+ throw new Error(`Antes de 2000 a população por UF só existe em faixas etárias quinquenais (${AGGREGATED_AGE_GROUPS.join(", ")}): ` +
757
+ `o intervalo de idade ${options.ageMin ?? 0}–${options.ageMax ?? "+"} não cai nos limites das faixas. Use age_min múltiplo de 5 e age_max terminado em 4 ou 9 (ou 80+).`);
758
+ }
759
+ const result = await queryPopulationUfAgregado({
760
+ years: [options.year],
761
+ ufs: options.uf,
762
+ sex: options.sex,
763
+ ageGroups,
764
+ });
765
+ if (result[0]?.population)
766
+ return result[0].population;
767
+ if (ageGroups)
768
+ return 0;
769
+ }
770
+ // Fallback: pop_municipios.parquet (agregando por UF)
771
+ const result = await queryPopulationFromMunicipios({
772
+ years: [options.year],
773
+ ufs: options.uf,
774
+ sex: options.sex,
775
+ });
776
+ return result[0]?.population || 0;
777
+ }
778
+ /**
779
+ * Obtém população por UF para um ano.
780
+ * Usa pop_uf.parquet se disponível, senão agrega pop_municipios.parquet.
781
+ */
782
+ export async function getPopulationByUf(options) {
783
+ if (!hasPopulationData()) {
784
+ throw new Error(POPULATION_MISSING_MESSAGE);
785
+ }
786
+ let result;
787
+ if (options.year >= 2000 && hasPopUf()) {
788
+ result = await queryPopulationUf({
789
+ filters: {
790
+ years: [options.year],
791
+ sex: options.sex,
792
+ ageMin: options.ageMin,
793
+ ageMax: options.ageMax,
794
+ },
795
+ groupBy: ["uf"],
796
+ });
797
+ }
798
+ else if (options.year < 2000) {
799
+ const ageGroups = aggregatedAgeGroupsFor(options.ageMin, options.ageMax);
800
+ if (ageGroups === null) {
801
+ throw new Error(`Antes de 2000 a população por UF só existe em faixas etárias quinquenais (${AGGREGATED_AGE_GROUPS.join(", ")}): ` +
802
+ `o intervalo de idade ${options.ageMin ?? 0}–${options.ageMax ?? "+"} não cai nos limites das faixas.`);
803
+ }
804
+ try {
805
+ result = await queryPopulationUfAgregado({
806
+ years: [options.year],
807
+ sex: options.sex,
808
+ ageGroups,
809
+ groupBy: ["uf"],
810
+ });
811
+ }
812
+ catch {
813
+ if (ageGroups)
814
+ throw new Error("pop_uf_agregado.parquet indisponível: sem população por faixa etária antes de 2000.");
815
+ // Fallback (sem idade)
816
+ result = await queryPopulationFromMunicipios({
817
+ years: [options.year],
818
+ sex: options.sex,
819
+ groupBy: ["uf"],
820
+ });
821
+ }
822
+ }
823
+ else {
824
+ // Fallback: pop_municipios.parquet agregado por UF
825
+ result = await queryPopulationFromMunicipios({
826
+ years: [options.year],
827
+ sex: options.sex,
828
+ groupBy: ["uf"],
829
+ });
830
+ }
831
+ const byUf = {};
832
+ for (const row of result) {
833
+ byUf[row.uf] = row.population;
834
+ }
835
+ return byUf;
836
+ }
837
+ //# sourceMappingURL=duckdb.js.map