sih-br-mcp 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +101 -0
- package/dist/cache.d.ts +108 -0
- package/dist/cache.d.ts.map +1 -0
- package/dist/cache.js +270 -0
- package/dist/cache.js.map +1 -0
- package/dist/data/brazil-regions.json +68 -0
- package/dist/data/cid-chapters.json +163 -0
- package/dist/data/csap-groups-cid9.json +4281 -0
- package/dist/data/csap-groups.json +262 -0
- package/dist/db/duckdb.d.ts +263 -0
- package/dist/db/duckdb.d.ts.map +1 -0
- package/dist/db/duckdb.js +837 -0
- package/dist/db/duckdb.js.map +1 -0
- package/dist/freshness.d.ts +101 -0
- package/dist/freshness.d.ts.map +1 -0
- package/dist/freshness.js +303 -0
- package/dist/freshness.js.map +1 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +1504 -0
- package/dist/index.js.map +1 -0
- package/dist/provenance.d.ts +217 -0
- package/dist/provenance.d.ts.map +1 -0
- package/dist/provenance.js +534 -0
- package/dist/provenance.js.map +1 -0
- package/dist/utils/age-groups.d.ts +164 -0
- package/dist/utils/age-groups.d.ts.map +1 -0
- package/dist/utils/age-groups.js +112 -0
- package/dist/utils/age-groups.js.map +1 -0
- package/dist/utils/population.d.ts +39 -0
- package/dist/utils/population.d.ts.map +1 -0
- package/dist/utils/population.js +114 -0
- package/dist/utils/population.js.map +1 -0
- package/package.json +64 -0
|
@@ -0,0 +1,837 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Módulo de conexão DuckDB para consulta aos cubos de dados Parquet
|
|
3
|
+
* Suporta múltiplos arquivos Parquet por ano (sih_causas_YYYY.parquet)
|
|
4
|
+
*/
|
|
5
|
+
// Cliente DuckDB: `@duckdb/node-api` ("Node Neo"), o cliente oficial atual,
|
|
6
|
+
// desde 05/09/2026 (PLAN-002). Antes era o binding legado `duckdb`, que pinava
|
|
7
|
+
// `node-gyp ^9` em runtime e arrastava tar 6/cacache 16/glob 7 para o lock — a
|
|
8
|
+
// origem de 60 dos 72 alertas do Dependabot zerados em 04/09. O Neo traz
|
|
9
|
+
// binário pré-compilado por plataforma (optionalDependencies de
|
|
10
|
+
// @duckdb/node-bindings): zero node-gyp, zero tar. A API é toda de Promise;
|
|
11
|
+
// o funil `query()` continua sendo o único ponto que toca a conexão.
|
|
12
|
+
import { DuckDBInstance } from "@duckdb/node-api";
|
|
13
|
+
import { fileURLToPath } from "url";
|
|
14
|
+
import { dirname, join, resolve } from "path";
|
|
15
|
+
import { existsSync, mkdirSync, readdirSync } from "fs";
|
|
16
|
+
import { CUBES_BASE_URL, CUBES_CACHE_ENABLED, cubesCacheDir } from "../cache.js";
|
|
17
|
+
// Obtém diretório do projeto
|
|
18
|
+
const __filename = fileURLToPath(import.meta.url);
|
|
19
|
+
const __dirname = dirname(__filename);
|
|
20
|
+
const PROJECT_ROOT = join(__dirname, "..", "..");
|
|
21
|
+
// SIH_DATA_DIR aponta o servidor para outra pasta de cubos — é a costura que o
|
|
22
|
+
// smoke stdio usa para ler a fixture versionada em tests/fixtures/sih, já que
|
|
23
|
+
// data/*.parquet não vai para o git. Sem a variável, nada muda: data/ do projeto.
|
|
24
|
+
const DATA_DIR = process.env.SIH_DATA_DIR
|
|
25
|
+
? resolve(process.env.SIH_DATA_DIR)
|
|
26
|
+
: join(PROJECT_ROOT, "data");
|
|
27
|
+
const TEST_DATA_DIR = join(DATA_DIR, "test");
|
|
28
|
+
// Singleton do banco de dados. `connecting` segura a Promise da primeira
|
|
29
|
+
// abertura para que chamadas concorrentes (as tools de taxa disparam várias
|
|
30
|
+
// consultas) não criem duas instâncias — o legado tinha essa corrida.
|
|
31
|
+
let instance = null;
|
|
32
|
+
let conn = null;
|
|
33
|
+
let connecting = null;
|
|
34
|
+
// Detecta se estamos em modo teste (data/test tem arquivos)
|
|
35
|
+
let dataDirectory = null;
|
|
36
|
+
/**
|
|
37
|
+
* Determina qual diretório de dados usar.
|
|
38
|
+
* - Padrão: data/ (dados reais)
|
|
39
|
+
* - Teste: data/test/ (somente se SIH_TEST_MODE=1 estiver definido)
|
|
40
|
+
*/
|
|
41
|
+
/**
|
|
42
|
+
* A pasta de cubos CONFIGURADA (SIH_DATA_DIR ou data/ do projeto), exista
|
|
43
|
+
* cubo nela ou não. Quem precisa só do sidecar de proveniência — o frescor
|
|
44
|
+
* num checkout limpo, onde data/*.parquet é gitignored e só o JSON está
|
|
45
|
+
* versionado — lê daqui; quem precisa de Parquet usa getDataDirectory().
|
|
46
|
+
*/
|
|
47
|
+
export function configuredDataDirectory() {
|
|
48
|
+
return DATA_DIR;
|
|
49
|
+
}
|
|
50
|
+
export function getDataDirectory() {
|
|
51
|
+
if (dataDirectory)
|
|
52
|
+
return dataDirectory;
|
|
53
|
+
// Usa data/test/ somente em modo teste explícito
|
|
54
|
+
if (process.env.SIH_TEST_MODE === "1" && existsSync(TEST_DATA_DIR)) {
|
|
55
|
+
const testFiles = readdirSync(TEST_DATA_DIR).filter((f) => f.endsWith(".parquet"));
|
|
56
|
+
if (testFiles.length > 0) {
|
|
57
|
+
console.error(`[DuckDB] Modo TESTE: usando ${TEST_DATA_DIR}`);
|
|
58
|
+
dataDirectory = TEST_DATA_DIR;
|
|
59
|
+
return dataDirectory;
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
// Padrão: dados reais em data/
|
|
63
|
+
if (existsSync(DATA_DIR)) {
|
|
64
|
+
const prodFiles = readdirSync(DATA_DIR).filter((f) => f.startsWith("sih_") && f.endsWith(".parquet"));
|
|
65
|
+
if (prodFiles.length > 0) {
|
|
66
|
+
console.error(`[DuckDB] Usando dados em ${DATA_DIR}`);
|
|
67
|
+
dataDirectory = DATA_DIR;
|
|
68
|
+
return dataDirectory;
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
// Sem cubo na pasta do projeto (instalação pelo npm, que não embarca cubo):
|
|
72
|
+
// a pasta passa a ser o cache local, que `src/cache.ts` enche sob demanda a
|
|
73
|
+
// partir do canal público (data.sidneybissoli.com/sih/cubos/). A pasta pode
|
|
74
|
+
// estar vazia agora; os handlers chamam ensureYears() antes de consultar.
|
|
75
|
+
if (CUBES_CACHE_ENABLED) {
|
|
76
|
+
const cacheDir = cubesCacheDir();
|
|
77
|
+
mkdirSync(cacheDir, { recursive: true });
|
|
78
|
+
console.error(`[DuckDB] Sem cubos em ${DATA_DIR}; usando o cache ${cacheDir} (baixa de ${CUBES_BASE_URL} sob demanda)`);
|
|
79
|
+
dataDirectory = cacheDir;
|
|
80
|
+
return dataDirectory;
|
|
81
|
+
}
|
|
82
|
+
throw new Error(`Nenhum arquivo SIH Parquet encontrado em ${DATA_DIR} e o cache está desligado (SIH_CUBES_CACHE=off). Execute os scripts R de agregação ou baixe os cubos de ${CUBES_BASE_URL}.`);
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Retorna o padrão glob para um tipo de cubo
|
|
86
|
+
*/
|
|
87
|
+
export function getParquetPattern(cube) {
|
|
88
|
+
const dir = getDataDirectory();
|
|
89
|
+
return join(dir, `sih_${cube}_*.parquet`).replace(/\\/g, "/");
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Lista os anos disponíveis nos dados
|
|
93
|
+
*/
|
|
94
|
+
export function getAvailableYears() {
|
|
95
|
+
const dir = getDataDirectory();
|
|
96
|
+
const files = readdirSync(dir).filter((f) => f.startsWith("sih_causas_"));
|
|
97
|
+
const years = files
|
|
98
|
+
.map((f) => {
|
|
99
|
+
const match = f.match(/sih_causas_(\d{4})\.parquet/);
|
|
100
|
+
return match ? parseInt(match[1]) : null;
|
|
101
|
+
})
|
|
102
|
+
.filter((y) => y !== null)
|
|
103
|
+
.sort((a, b) => a - b);
|
|
104
|
+
return years;
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* Inicializa conexão com DuckDB
|
|
108
|
+
*/
|
|
109
|
+
export async function getDatabase() {
|
|
110
|
+
if (conn) {
|
|
111
|
+
return conn;
|
|
112
|
+
}
|
|
113
|
+
if (connecting) {
|
|
114
|
+
return connecting;
|
|
115
|
+
}
|
|
116
|
+
connecting = (async () => {
|
|
117
|
+
// Usa banco em memória (consultas diretas aos Parquet)
|
|
118
|
+
try {
|
|
119
|
+
instance = await DuckDBInstance.create(":memory:");
|
|
120
|
+
}
|
|
121
|
+
catch (err) {
|
|
122
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
123
|
+
throw new Error(`Erro ao criar banco DuckDB: ${message}`);
|
|
124
|
+
}
|
|
125
|
+
const connection = await instance.connect();
|
|
126
|
+
// Configura DuckDB para melhor performance com Parquet
|
|
127
|
+
try {
|
|
128
|
+
await connection.run("SET threads TO 4");
|
|
129
|
+
}
|
|
130
|
+
catch {
|
|
131
|
+
console.error("Aviso: não foi possível configurar threads");
|
|
132
|
+
}
|
|
133
|
+
conn = connection;
|
|
134
|
+
return connection;
|
|
135
|
+
})();
|
|
136
|
+
try {
|
|
137
|
+
return await connecting;
|
|
138
|
+
}
|
|
139
|
+
finally {
|
|
140
|
+
connecting = null;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* Fecha conexão com o banco
|
|
145
|
+
*/
|
|
146
|
+
export async function closeDatabase() {
|
|
147
|
+
if (conn) {
|
|
148
|
+
conn.closeSync();
|
|
149
|
+
conn = null;
|
|
150
|
+
}
|
|
151
|
+
if (instance) {
|
|
152
|
+
instance.closeSync();
|
|
153
|
+
instance = null;
|
|
154
|
+
}
|
|
155
|
+
dataDirectory = null; // Reset para próxima conexão
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Converte BigInt para Number em objetos aninhados
|
|
159
|
+
*/
|
|
160
|
+
function convertBigIntToNumber(obj) {
|
|
161
|
+
if (obj === null || obj === undefined) {
|
|
162
|
+
return obj;
|
|
163
|
+
}
|
|
164
|
+
if (typeof obj === "bigint") {
|
|
165
|
+
return Number(obj);
|
|
166
|
+
}
|
|
167
|
+
if (Array.isArray(obj)) {
|
|
168
|
+
return obj.map(convertBigIntToNumber);
|
|
169
|
+
}
|
|
170
|
+
if (typeof obj === "object") {
|
|
171
|
+
const result = {};
|
|
172
|
+
for (const [key, value] of Object.entries(obj)) {
|
|
173
|
+
result[key] = convertBigIntToNumber(value);
|
|
174
|
+
}
|
|
175
|
+
return result;
|
|
176
|
+
}
|
|
177
|
+
return obj;
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Executa query SQL e retorna resultados
|
|
181
|
+
*/
|
|
182
|
+
export async function query(sql) {
|
|
183
|
+
const connection = await getDatabase();
|
|
184
|
+
let rows;
|
|
185
|
+
try {
|
|
186
|
+
const result = await connection.runAndReadAll(sql);
|
|
187
|
+
// `getRowObjectsJS()` e não `getRowObjects()`: o segundo devolve DECIMAL
|
|
188
|
+
// como {width, scale, value} e DATE como {days} (classes do Neo), que
|
|
189
|
+
// virariam objetos no JSON da ferramenta; o primeiro entrega tipos JS
|
|
190
|
+
// nativos (DECIMAL → number, DATE → ISO). Medido em 05/09/2026. Hoje os
|
|
191
|
+
// cubos só têm INTEGER/DOUBLE/VARCHAR/BOOLEAN, mas o funil é um só e a
|
|
192
|
+
// defesa fica aqui. BIGINT e HUGEINT (SUM de INTEGER) continuam chegando
|
|
193
|
+
// como bigint nos dois leitores — é o que `convertBigIntToNumber` trata.
|
|
194
|
+
rows = result.getRowObjectsJS();
|
|
195
|
+
}
|
|
196
|
+
catch (err) {
|
|
197
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
198
|
+
throw new Error(`Erro na query: ${message}\nSQL: ${sql}`);
|
|
199
|
+
}
|
|
200
|
+
// Converte BigInt para Number para serialização JSON
|
|
201
|
+
return convertBigIntToNumber(rows);
|
|
202
|
+
}
|
|
203
|
+
// =============================================================================
|
|
204
|
+
// FUNÇÕES DE QUERY ESPECÍFICAS
|
|
205
|
+
// =============================================================================
|
|
206
|
+
/**
|
|
207
|
+
* Constrói cláusula WHERE a partir de filtros
|
|
208
|
+
*/
|
|
209
|
+
function buildWhereClause(filters) {
|
|
210
|
+
const conditions = [];
|
|
211
|
+
if ("years" in filters && filters.years && filters.years.length > 0) {
|
|
212
|
+
conditions.push(`year IN (${filters.years.join(", ")})`);
|
|
213
|
+
}
|
|
214
|
+
if ("months" in filters && filters.months && filters.months.length > 0) {
|
|
215
|
+
conditions.push(`month IN (${filters.months.join(", ")})`);
|
|
216
|
+
}
|
|
217
|
+
if (filters.ufs && filters.ufs.length > 0) {
|
|
218
|
+
const ufs = filters.ufs.map((u) => `'${u}'`).join(", ");
|
|
219
|
+
conditions.push(`uf IN (${ufs})`);
|
|
220
|
+
}
|
|
221
|
+
if ("municipalityCodes" in filters &&
|
|
222
|
+
filters.municipalityCodes &&
|
|
223
|
+
filters.municipalityCodes.length > 0) {
|
|
224
|
+
const codes = filters.municipalityCodes.map((c) => `'${c}'`).join(", ");
|
|
225
|
+
conditions.push(`municipality_code IN (${codes})`);
|
|
226
|
+
}
|
|
227
|
+
if ("cidChapters" in filters &&
|
|
228
|
+
filters.cidChapters &&
|
|
229
|
+
filters.cidChapters.length > 0) {
|
|
230
|
+
conditions.push(`cid_chapter IN (${filters.cidChapters.join(", ")})`);
|
|
231
|
+
}
|
|
232
|
+
if ("cidGroups" in filters && filters.cidGroups && filters.cidGroups.length > 0) {
|
|
233
|
+
const groups = filters.cidGroups.map((g) => `'${g}'`).join(", ");
|
|
234
|
+
conditions.push(`cid_group IN (${groups})`);
|
|
235
|
+
}
|
|
236
|
+
if (filters.sex) {
|
|
237
|
+
conditions.push(`sex = '${filters.sex}'`);
|
|
238
|
+
}
|
|
239
|
+
if (filters.ageMin !== undefined) {
|
|
240
|
+
conditions.push(`age >= ${filters.ageMin}`);
|
|
241
|
+
}
|
|
242
|
+
if (filters.ageMax !== undefined) {
|
|
243
|
+
conditions.push(`age <= ${filters.ageMax}`);
|
|
244
|
+
}
|
|
245
|
+
if (filters.races && filters.races.length > 0) {
|
|
246
|
+
const races = filters.races.map((r) => `'${r}'`).join(", ");
|
|
247
|
+
conditions.push(`race IN (${races})`);
|
|
248
|
+
}
|
|
249
|
+
if ("isCsap" in filters && filters.isCsap !== undefined) {
|
|
250
|
+
conditions.push(`is_csap = ${filters.isCsap}`);
|
|
251
|
+
}
|
|
252
|
+
if ("csapGroups" in filters && filters.csapGroups && filters.csapGroups.length > 0) {
|
|
253
|
+
const groups = filters.csapGroups.map((g) => `'${g}'`).join(", ");
|
|
254
|
+
conditions.push(`csap_group IN (${groups})`);
|
|
255
|
+
}
|
|
256
|
+
return conditions.length > 0 ? conditions.join(" AND ") : "1=1";
|
|
257
|
+
}
|
|
258
|
+
/**
|
|
259
|
+
* Expressão de uma coluna de agrupamento no SELECT e no GROUP BY. `cid_revision`
|
|
260
|
+
* só existe nos cubos do builder >= 2.5.0: os cubos são abertos por glob com
|
|
261
|
+
* `union_by_name`, então um cubo anterior (cache local misto) a traz NULA — e
|
|
262
|
+
* nula significa CID-10 (a coluna é constante 10 em 1998+). A defesa vale só
|
|
263
|
+
* para caches mistos; a série publicada é toda 2.5.0.
|
|
264
|
+
*/
|
|
265
|
+
/**
|
|
266
|
+
* Soma de `value` como DECIMAL(18,2) e não como DOUBLE: o SUM paralelo de
|
|
267
|
+
* doubles do DuckDB muda o último dígito conforme a ordem em que as threads
|
|
268
|
+
* terminam (golden de 07/09/2026: 9746572.610000005 numa execução,
|
|
269
|
+
* 9746572.61000001 na seguinte); em decimal a soma é exata e determinística.
|
|
270
|
+
* `value` é dinheiro com 2 casas — nada se perde.
|
|
271
|
+
*/
|
|
272
|
+
function groupSelect(column) {
|
|
273
|
+
return column === "cid_revision" ? "COALESCE(cid_revision, 10) AS cid_revision" : column;
|
|
274
|
+
}
|
|
275
|
+
function groupKey(column) {
|
|
276
|
+
return column === "cid_revision" ? "COALESCE(cid_revision, 10)" : column;
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* ORDER BY determinístico para consultas agrupadas.
|
|
280
|
+
*
|
|
281
|
+
* Sem isto, duas chamadas iguais devolvem respostas diferentes: GROUP BY sem
|
|
282
|
+
* ORDER BY sai na ordem em que as 4 threads do DuckDB terminam, e ORDER BY
|
|
283
|
+
* por métrica empata (dois grupos com o mesmo SUM) sem critério de desempate —
|
|
284
|
+
* medido em 05/09/2026 sobre a fixture: `compare_regions` dava posição 5 a GO
|
|
285
|
+
* numa chamada e a PB na seguinte, `rank_csap_groups` trocava g01 e g13, e
|
|
286
|
+
* qualquer `limit` em cima disso devolvia SUBCONJUNTOS diferentes. O golden
|
|
287
|
+
* das ferramentas (scripts/golden-tools.mjs) foi o que expôs; o conserto é
|
|
288
|
+
* aqui, no funil, e não tool a tool: toda coluna de agrupamento que o chamador
|
|
289
|
+
* não ordenou explicitamente entra como desempate, na ordem do GROUP BY.
|
|
290
|
+
*/
|
|
291
|
+
function buildOrderBy(orderBy, groupBy) {
|
|
292
|
+
const mentioned = new Set((orderBy ?? "")
|
|
293
|
+
.split(",")
|
|
294
|
+
.map((part) => part.trim().split(/\s+/)[0])
|
|
295
|
+
.filter(Boolean));
|
|
296
|
+
const tieBreak = groupBy.filter((column) => !mentioned.has(column));
|
|
297
|
+
const parts = [orderBy, ...tieBreak].filter((part) => !!part);
|
|
298
|
+
return parts.length > 0 ? ` ORDER BY ${parts.join(", ")}` : "";
|
|
299
|
+
}
|
|
300
|
+
/**
|
|
301
|
+
* Query no cubo de causas com agregação flexível
|
|
302
|
+
*/
|
|
303
|
+
export async function queryCausas(options) {
|
|
304
|
+
const pattern = getParquetPattern("causas");
|
|
305
|
+
const whereClause = buildWhereClause(options.filters || {});
|
|
306
|
+
// Métricas a calcular
|
|
307
|
+
const metricsMap = {
|
|
308
|
+
n: "SUM(n) as n_hospitalizations",
|
|
309
|
+
days: "SUM(days) as total_days",
|
|
310
|
+
value: "SUM(CAST(value AS DECIMAL(18,2))) as total_value",
|
|
311
|
+
deaths: "SUM(deaths) as deaths",
|
|
312
|
+
};
|
|
313
|
+
const metrics = options.metrics || ["n"];
|
|
314
|
+
const selectMetrics = metrics.map((m) => metricsMap[m]).join(", ");
|
|
315
|
+
// Group by
|
|
316
|
+
const groupBy = options.groupBy || [];
|
|
317
|
+
const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
|
|
318
|
+
const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
|
|
319
|
+
let sql = `
|
|
320
|
+
SELECT ${selectGroups}${selectMetrics}
|
|
321
|
+
FROM read_parquet('${pattern}', union_by_name = true)
|
|
322
|
+
WHERE ${whereClause}
|
|
323
|
+
${groupByClause}
|
|
324
|
+
`;
|
|
325
|
+
sql += buildOrderBy(options.orderBy, groupBy);
|
|
326
|
+
if (options.limit) {
|
|
327
|
+
sql += ` LIMIT ${options.limit}`;
|
|
328
|
+
}
|
|
329
|
+
return query(sql);
|
|
330
|
+
}
|
|
331
|
+
/**
|
|
332
|
+
* Query no cubo de ICSAP com agregação flexível
|
|
333
|
+
*/
|
|
334
|
+
export async function queryIcsap(options) {
|
|
335
|
+
// 0.9.0: toda agregação do cubo ICSAP passa por icsapAggregateSql (o
|
|
336
|
+
// denominador por estratos distintos); `metrics` fica só pela assinatura —
|
|
337
|
+
// as seis medidas saem sempre.
|
|
338
|
+
return query(icsapAggregateSql(options));
|
|
339
|
+
}
|
|
340
|
+
/** Chaves do ESTRATO do cubo ICSAP: `n_total` é o total do estrato, repetido em cada linha dele. */
|
|
341
|
+
// `exclusion` (builder >= 2.6.0) é chave do estrato: sem ela, dois estratos que
|
|
342
|
+
// só diferem no motivo de exclusão e têm o mesmo n_total colapsariam no
|
|
343
|
+
// DISTINCT (2023/RR com universe = "all" dava 48.122 em vez de 48.480).
|
|
344
|
+
const ICSAP_STRATUM_KEYS = ["year", "uf", "municipality_code", "cid_revision", "sex", "age", "race", "exclusion"];
|
|
345
|
+
function universeCondition(universe) {
|
|
346
|
+
return (universe ?? "csapaih") === "csapaih" ? "AND exclusion IS NULL" : "";
|
|
347
|
+
}
|
|
348
|
+
function icsapAggregateSql(options) {
|
|
349
|
+
const pattern = getParquetPattern("icsap");
|
|
350
|
+
const { csapGroups, ...strataFilters } = options.filters ?? {};
|
|
351
|
+
const whereStrata = `${buildWhereClause(strataFilters)} ${universeCondition(options.universe)}`;
|
|
352
|
+
const csapCond = csapGroups && csapGroups.length > 0 ? `AND csap_group IN (${csapGroups.map((g) => `'${g}'`).join(", ")})` : "";
|
|
353
|
+
const groupBy = options.groupBy ?? [];
|
|
354
|
+
const byGroup = groupBy.includes("csap_group");
|
|
355
|
+
const strataGroup = groupBy.filter((g) => g !== "csap_group");
|
|
356
|
+
const groupClause = (cols) => (cols.length > 0 ? `GROUP BY ${cols.join(", ")}` : "");
|
|
357
|
+
const joinOn = strataGroup.length > 0 ? strataGroup.map((k) => `icsap.${k} IS NOT DISTINCT FROM total.${k}`).join(" AND ") : "TRUE";
|
|
358
|
+
// Sem csap_group no agrupamento, a linha-mestre é a de total: um grupo com
|
|
359
|
+
// estratos mas sem ICSAP ainda aparece, com 0. Com csap_group, é a de ICSAP.
|
|
360
|
+
const fromClause = byGroup ? `icsap JOIN total ON ${joinOn}` : `total LEFT JOIN icsap ON ${joinOn}`;
|
|
361
|
+
const selectCols = groupBy
|
|
362
|
+
.map((g) => (byGroup || g === "csap_group" ? `icsap.${g} AS ${g}` : `total.${g} AS ${g}`))
|
|
363
|
+
.join(", ");
|
|
364
|
+
const strataKeys = ICSAP_STRATUM_KEYS.map(groupSelect);
|
|
365
|
+
let sql = `
|
|
366
|
+
WITH linhas AS (
|
|
367
|
+
SELECT * FROM read_parquet('${pattern}', union_by_name = true)
|
|
368
|
+
WHERE ${whereStrata}
|
|
369
|
+
),
|
|
370
|
+
estratos AS (SELECT DISTINCT ${strataKeys.join(", ")}, n_total FROM linhas),
|
|
371
|
+
total AS (
|
|
372
|
+
SELECT ${strataGroup.length > 0 ? strataGroup.join(", ") + ", " : ""}SUM(n_total) AS n_total
|
|
373
|
+
FROM estratos ${groupClause(strataGroup)}
|
|
374
|
+
),
|
|
375
|
+
icsap AS (
|
|
376
|
+
SELECT ${groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : ""}SUM(n) AS n_icsap, SUM(days) AS total_days, SUM(CAST(value AS DECIMAL(18,2))) AS total_value, SUM(deaths) AS deaths
|
|
377
|
+
FROM linhas
|
|
378
|
+
WHERE csap_group IS NOT NULL ${csapCond}
|
|
379
|
+
${groupClause(groupBy.map(groupKey))}
|
|
380
|
+
)
|
|
381
|
+
SELECT ${selectCols}${groupBy.length > 0 ? ", " : ""}
|
|
382
|
+
COALESCE(icsap.n_icsap, 0) AS n_icsap,
|
|
383
|
+
total.n_total AS n_total,
|
|
384
|
+
ROUND(COALESCE(icsap.n_icsap, 0) * 100.0 / NULLIF(total.n_total, 0), 2) AS icsap_percentage,
|
|
385
|
+
COALESCE(icsap.total_days, 0) AS total_days,
|
|
386
|
+
COALESCE(icsap.total_value, 0) AS total_value,
|
|
387
|
+
COALESCE(icsap.deaths, 0) AS deaths
|
|
388
|
+
FROM ${fromClause}
|
|
389
|
+
`;
|
|
390
|
+
sql += buildOrderBy(options.orderBy, groupBy);
|
|
391
|
+
if (options.limit)
|
|
392
|
+
sql += ` LIMIT ${options.limit}`;
|
|
393
|
+
return sql;
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* Query no cubo de séries temporais
|
|
397
|
+
*/
|
|
398
|
+
export async function querySeries(options) {
|
|
399
|
+
const pattern = getParquetPattern("series");
|
|
400
|
+
const conditions = [];
|
|
401
|
+
if (options.filters) {
|
|
402
|
+
if (options.filters.yearMonthStart) {
|
|
403
|
+
conditions.push(`year_month >= '${options.filters.yearMonthStart}'`);
|
|
404
|
+
}
|
|
405
|
+
if (options.filters.yearMonthEnd) {
|
|
406
|
+
conditions.push(`year_month <= '${options.filters.yearMonthEnd}'`);
|
|
407
|
+
}
|
|
408
|
+
if (options.filters.ufs && options.filters.ufs.length > 0) {
|
|
409
|
+
const ufs = options.filters.ufs.map((u) => `'${u}'`).join(", ");
|
|
410
|
+
conditions.push(`uf IN (${ufs})`);
|
|
411
|
+
}
|
|
412
|
+
if (options.filters.cidChapters && options.filters.cidChapters.length > 0) {
|
|
413
|
+
conditions.push(`cid_chapter IN (${options.filters.cidChapters.join(", ")})`);
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
const whereClause = conditions.length > 0 ? conditions.join(" AND ") : "1=1";
|
|
417
|
+
// Group by
|
|
418
|
+
const groupBy = options.groupBy || [];
|
|
419
|
+
const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
|
|
420
|
+
const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
|
|
421
|
+
let sql = `
|
|
422
|
+
SELECT ${selectGroups}SUM(n) as n, SUM(deaths) as deaths
|
|
423
|
+
FROM read_parquet('${pattern}', union_by_name = true)
|
|
424
|
+
WHERE ${whereClause}
|
|
425
|
+
${groupByClause}
|
|
426
|
+
`;
|
|
427
|
+
sql += buildOrderBy(options.orderBy, groupBy);
|
|
428
|
+
return query(sql);
|
|
429
|
+
}
|
|
430
|
+
/**
|
|
431
|
+
* Query para ranking de grupos CSAP
|
|
432
|
+
*/
|
|
433
|
+
export async function rankCsapGroups(options) {
|
|
434
|
+
const pattern = getParquetPattern("icsap");
|
|
435
|
+
const whereClause = `${buildWhereClause(options.filters || {})} ${universeCondition(options.universe)}`;
|
|
436
|
+
const metric = options.metric || "n";
|
|
437
|
+
const metricsMap = {
|
|
438
|
+
n: "SUM(n)",
|
|
439
|
+
days: "SUM(days)",
|
|
440
|
+
value: "SUM(CAST(value AS DECIMAL(18,2)))",
|
|
441
|
+
deaths: "SUM(deaths)",
|
|
442
|
+
};
|
|
443
|
+
const sql = `
|
|
444
|
+
SELECT
|
|
445
|
+
csap_group,
|
|
446
|
+
${metricsMap[metric]} as metric_value,
|
|
447
|
+
SUM(n) as n_hospitalizations,
|
|
448
|
+
SUM(days) as total_days,
|
|
449
|
+
SUM(CAST(value AS DECIMAL(18,2))) as total_value,
|
|
450
|
+
SUM(deaths) as deaths
|
|
451
|
+
FROM read_parquet('${pattern}', union_by_name = true)
|
|
452
|
+
WHERE ${whereClause} AND csap_group IS NOT NULL
|
|
453
|
+
GROUP BY csap_group
|
|
454
|
+
ORDER BY metric_value DESC, csap_group
|
|
455
|
+
${options.limit ? `LIMIT ${options.limit}` : ""}
|
|
456
|
+
`;
|
|
457
|
+
return query(sql);
|
|
458
|
+
}
|
|
459
|
+
/**
|
|
460
|
+
* Calcula indicadores de ICSAP (percentual)
|
|
461
|
+
*/
|
|
462
|
+
export async function calculateIcsapIndicators(options) {
|
|
463
|
+
// 0.9.0: denominador por estratos distintos (ver icsapAggregateSql)
|
|
464
|
+
return query(icsapAggregateSql({ filters: options.filters, groupBy: options.groupBy, universe: options.universe }));
|
|
465
|
+
}
|
|
466
|
+
// =============================================================================
|
|
467
|
+
// FUNÇÕES DE POPULAÇÃO
|
|
468
|
+
// =============================================================================
|
|
469
|
+
/**
|
|
470
|
+
* Onde os arquivos de população vivem (0.12.0, sih:populacao-no-canal): a
|
|
471
|
+
* pasta configurada (SIH_DATA_DIR ou data/ do projeto; nunca data/test/) tem
|
|
472
|
+
* precedência sempre que tiver pop_uf.parquet — é o caso da fixture do smoke
|
|
473
|
+
* e do golden e de quem gerou a população à mão; sem nada nela e com o cache
|
|
474
|
+
* ligado, o cache local, que `ensurePopulation()` (src/cache.ts) enche a
|
|
475
|
+
* partir do canal sih/cubos/ (pop_uf, pop_uf_agregado, pop_municipios +
|
|
476
|
+
* pop_provenance.json, assinados no bloco `population` do manifesto).
|
|
477
|
+
*/
|
|
478
|
+
export function getPopulationDir() {
|
|
479
|
+
if (existsSync(join(DATA_DIR, "pop_uf.parquet")))
|
|
480
|
+
return DATA_DIR;
|
|
481
|
+
if (CUBES_CACHE_ENABLED)
|
|
482
|
+
return cubesCacheDir();
|
|
483
|
+
return DATA_DIR;
|
|
484
|
+
}
|
|
485
|
+
/** Padrão de arquivo para dados populacionais (pasta de getPopulationDir()). */
|
|
486
|
+
export function getPopulationPattern(type) {
|
|
487
|
+
const filename = type === "municipios" ? "pop_municipios.parquet" :
|
|
488
|
+
type === "uf" ? "pop_uf.parquet" :
|
|
489
|
+
"pop_uf_agregado.parquet";
|
|
490
|
+
return join(getPopulationDir(), filename).replace(/\\/g, "/");
|
|
491
|
+
}
|
|
492
|
+
/** Verifica se algum arquivo de população está disponível em getPopulationDir(). */
|
|
493
|
+
export function hasPopulationData() {
|
|
494
|
+
try {
|
|
495
|
+
const dir = getPopulationDir();
|
|
496
|
+
if (!existsSync(dir))
|
|
497
|
+
return false;
|
|
498
|
+
const files = readdirSync(dir);
|
|
499
|
+
return files.some(f => f.startsWith("pop_") && f.endsWith(".parquet"));
|
|
500
|
+
}
|
|
501
|
+
catch {
|
|
502
|
+
return false;
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
/** Mensagem única para "sem denominador" — aponta o canal, não um script R. */
|
|
506
|
+
export const POPULATION_MISSING_MESSAGE = `Dados populacionais não disponíveis: nem pop_uf.parquet na pasta de dados nem no cache local. Com o cache ligado o servidor baixa pop_uf, pop_uf_agregado e pop_municipios de ${CUBES_BASE_URL} (bloco population do manifest.json); confira a rede ou a variável SIH_CUBES_CACHE.`;
|
|
507
|
+
/**
|
|
508
|
+
* Constrói cláusula WHERE para população UF
|
|
509
|
+
*/
|
|
510
|
+
function buildPopulationWhereClause(filters) {
|
|
511
|
+
const conditions = [];
|
|
512
|
+
if (filters.years && filters.years.length > 0) {
|
|
513
|
+
conditions.push(`year IN (${filters.years.join(", ")})`);
|
|
514
|
+
}
|
|
515
|
+
if (filters.ufs && filters.ufs.length > 0) {
|
|
516
|
+
const ufs = filters.ufs.map((u) => `'${u}'`).join(", ");
|
|
517
|
+
conditions.push(`uf IN (${ufs})`);
|
|
518
|
+
}
|
|
519
|
+
if (filters.sex) {
|
|
520
|
+
conditions.push(`sex = '${filters.sex}'`);
|
|
521
|
+
}
|
|
522
|
+
if (filters.ageMin !== undefined) {
|
|
523
|
+
conditions.push(`age >= ${filters.ageMin}`);
|
|
524
|
+
}
|
|
525
|
+
if (filters.ageMax !== undefined) {
|
|
526
|
+
conditions.push(`age <= ${filters.ageMax}`);
|
|
527
|
+
}
|
|
528
|
+
return conditions.length > 0 ? conditions.join(" AND ") : "1=1";
|
|
529
|
+
}
|
|
530
|
+
/**
|
|
531
|
+
* Query população por UF (dados com idade simples)
|
|
532
|
+
*/
|
|
533
|
+
export async function queryPopulationUf(options) {
|
|
534
|
+
const pattern = getPopulationPattern("uf");
|
|
535
|
+
const whereClause = buildPopulationWhereClause(options.filters || {});
|
|
536
|
+
const groupBy = options.groupBy || [];
|
|
537
|
+
const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
|
|
538
|
+
const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
|
|
539
|
+
const sql = `
|
|
540
|
+
SELECT
|
|
541
|
+
${selectGroups}
|
|
542
|
+
SUM(population) as population
|
|
543
|
+
FROM read_parquet('${pattern}', union_by_name = true)
|
|
544
|
+
WHERE ${whereClause}
|
|
545
|
+
${groupByClause}${buildOrderBy(undefined, groupBy)}
|
|
546
|
+
`;
|
|
547
|
+
return query(sql);
|
|
548
|
+
}
|
|
549
|
+
/**
|
|
550
|
+
* Query população agregada por UF (para anos 1991-1999, com faixa etária)
|
|
551
|
+
*/
|
|
552
|
+
export async function queryPopulationUfAgregado(options) {
|
|
553
|
+
const pattern = getPopulationPattern("uf_agregado");
|
|
554
|
+
const conditions = [];
|
|
555
|
+
if (options.years && options.years.length > 0) {
|
|
556
|
+
conditions.push(`year IN (${options.years.join(", ")})`);
|
|
557
|
+
}
|
|
558
|
+
if (options.ageGroups && options.ageGroups.length > 0) {
|
|
559
|
+
// As linhas de idade ignorada (age_group nulo; POPBR "I000") ficam fora de
|
|
560
|
+
// qualquer recorte etário — e dentro do total, que só bate com a
|
|
561
|
+
// estimativa do IBGE com elas (Brasil 1993: 151.556.521).
|
|
562
|
+
conditions.push(`age_group IN (${options.ageGroups.map((g) => `'${g}'`).join(", ")})`);
|
|
563
|
+
}
|
|
564
|
+
if (options.ufs && options.ufs.length > 0) {
|
|
565
|
+
const ufs = options.ufs.map((u) => `'${u}'`).join(", ");
|
|
566
|
+
conditions.push(`uf IN (${ufs})`);
|
|
567
|
+
}
|
|
568
|
+
if (options.sex) {
|
|
569
|
+
conditions.push(`sex = '${options.sex}'`);
|
|
570
|
+
}
|
|
571
|
+
const whereClause = conditions.length > 0 ? conditions.join(" AND ") : "1=1";
|
|
572
|
+
const groupBy = options.groupBy || [];
|
|
573
|
+
const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
|
|
574
|
+
const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
|
|
575
|
+
const sql = `
|
|
576
|
+
SELECT
|
|
577
|
+
${selectGroups}
|
|
578
|
+
SUM(population) as population
|
|
579
|
+
FROM read_parquet('${pattern}', union_by_name = true)
|
|
580
|
+
WHERE ${whereClause}
|
|
581
|
+
${groupByClause}${buildOrderBy(undefined, groupBy)}
|
|
582
|
+
`;
|
|
583
|
+
return query(sql);
|
|
584
|
+
}
|
|
585
|
+
/**
|
|
586
|
+
* Verifica se pop_uf.parquet existe
|
|
587
|
+
*/
|
|
588
|
+
function hasPopUf() {
|
|
589
|
+
const path = join(getPopulationDir(), "pop_uf.parquet");
|
|
590
|
+
return existsSync(path);
|
|
591
|
+
}
|
|
592
|
+
/** Intervalo de anos de um arquivo de população, lido do próprio arquivo. */
|
|
593
|
+
async function yearRangeOf(file) {
|
|
594
|
+
const path = join(getPopulationDir(), file);
|
|
595
|
+
if (!existsSync(path))
|
|
596
|
+
return null;
|
|
597
|
+
const rows = await query(`SELECT CAST(min(year) AS INTEGER) AS first_year, CAST(max(year) AS INTEGER) AS last_year FROM read_parquet('${path.replace(/\\/g, "/")}')`);
|
|
598
|
+
const r = rows[0];
|
|
599
|
+
return r && r.first_year != null && r.last_year != null
|
|
600
|
+
? { first_year: Number(r.first_year), last_year: Number(r.last_year) }
|
|
601
|
+
: null;
|
|
602
|
+
}
|
|
603
|
+
/** Faixas etárias quinquenais de pop_uf_agregado.parquet (1991–1999), na ordem. */
|
|
604
|
+
export const AGGREGATED_AGE_GROUPS = [
|
|
605
|
+
"0-4", "5-9", "10-14", "15-19", "20-24", "25-29", "30-34", "35-39", "40-44",
|
|
606
|
+
"45-49", "50-54", "55-59", "60-64", "65-69", "70-74", "75-79", "80 e +",
|
|
607
|
+
];
|
|
608
|
+
/**
|
|
609
|
+
* Faixas de pop_uf_agregado.parquet que cobrem EXATAMENTE [ageMin, ageMax].
|
|
610
|
+
* Devolve null quando o intervalo não cai nos limites das faixas (ex.: 0–14
|
|
611
|
+
* serve, 0–17 não): antes de 2000 só há população por faixa quinquenal, e uma
|
|
612
|
+
* taxa com denominador aproximado seria número errado com cara de certo.
|
|
613
|
+
* Sem filtro de idade devolve undefined (todas as faixas).
|
|
614
|
+
*/
|
|
615
|
+
export function aggregatedAgeGroupsFor(ageMin, ageMax) {
|
|
616
|
+
if (ageMin === undefined && ageMax === undefined)
|
|
617
|
+
return undefined;
|
|
618
|
+
const lo = ageMin ?? 0;
|
|
619
|
+
const hi = ageMax ?? Infinity;
|
|
620
|
+
if (lo % 5 !== 0)
|
|
621
|
+
return null;
|
|
622
|
+
if (hi !== Infinity && hi < 79 && (hi + 1) % 5 !== 0)
|
|
623
|
+
return null;
|
|
624
|
+
const out = [];
|
|
625
|
+
for (const g of AGGREGATED_AGE_GROUPS) {
|
|
626
|
+
const start = parseInt(g, 10);
|
|
627
|
+
const end = g === "80 e +" ? Infinity : start + 4;
|
|
628
|
+
if (start >= lo && end <= hi)
|
|
629
|
+
out.push(g);
|
|
630
|
+
else if (start >= lo && g === "80 e +" && hi >= 80)
|
|
631
|
+
out.push(g);
|
|
632
|
+
}
|
|
633
|
+
if (hi !== Infinity && hi >= 80 && hi < Infinity && !out.includes("80 e +"))
|
|
634
|
+
return null;
|
|
635
|
+
return out.length > 0 ? out : null;
|
|
636
|
+
}
|
|
637
|
+
let popCoverageCache;
|
|
638
|
+
/**
|
|
639
|
+
* Esquece a cobertura memoizada — chamada depois que ensurePopulation() baixa
|
|
640
|
+
* algum arquivo: uma chamada anterior pode ter memoizado `null` (sem
|
|
641
|
+
* população) antes do download.
|
|
642
|
+
*/
|
|
643
|
+
export function resetPopulationCoverageCache() {
|
|
644
|
+
popCoverageCache = undefined;
|
|
645
|
+
}
|
|
646
|
+
/**
|
|
647
|
+
* Cobertura populacional, lida dos próprios arquivos — nunca fixada no código.
|
|
648
|
+
* `first_year`/`last_year` é a união de pop_uf.parquet (idade simples, 2000+;
|
|
649
|
+
* a regra do CONTEXT.md diz que vai até o último ano de cubo FECHADO do SIH,
|
|
650
|
+
* e quem a cumpre é o build-population.R do healthbr-data) e de pop_uf_agregado.parquet
|
|
651
|
+
* (faixa etária, 1991–1999; sih:taxas-1992-1999). Antes de 2000 a taxa por
|
|
652
|
+
* idade só sai nas faixas do arquivo (aggregatedAgeGroupsFor). Devolve null
|
|
653
|
+
* sem nenhum dos dois arquivos.
|
|
654
|
+
*/
|
|
655
|
+
export async function getPopulationCoverage() {
|
|
656
|
+
if (popCoverageCache !== undefined)
|
|
657
|
+
return popCoverageCache;
|
|
658
|
+
const detailed = await yearRangeOf("pop_uf.parquet");
|
|
659
|
+
const aggregated = await yearRangeOf("pop_uf_agregado.parquet");
|
|
660
|
+
if (!detailed && !aggregated) {
|
|
661
|
+
popCoverageCache = null;
|
|
662
|
+
return null;
|
|
663
|
+
}
|
|
664
|
+
const ranges = [detailed, aggregated].filter((r) => !!r);
|
|
665
|
+
popCoverageCache = {
|
|
666
|
+
first_year: Math.min(...ranges.map((r) => r.first_year)),
|
|
667
|
+
last_year: Math.max(...ranges.map((r) => r.last_year)),
|
|
668
|
+
detailed: detailed ? { ...detailed, source: "pop_uf.parquet", age: "idade simples" } : null,
|
|
669
|
+
aggregated: aggregated
|
|
670
|
+
? { ...aggregated, source: "pop_uf_agregado.parquet", age: "faixa etária quinquenal", age_groups: AGGREGATED_AGE_GROUPS }
|
|
671
|
+
: null,
|
|
672
|
+
};
|
|
673
|
+
return popCoverageCache;
|
|
674
|
+
}
|
|
675
|
+
/** Qual arquivo de população serve a um ano (null = nenhum). */
|
|
676
|
+
export async function populationSourceFor(year) {
|
|
677
|
+
const c = await getPopulationCoverage();
|
|
678
|
+
if (!c)
|
|
679
|
+
return null;
|
|
680
|
+
if (c.detailed && year >= c.detailed.first_year && year <= c.detailed.last_year)
|
|
681
|
+
return "detailed";
|
|
682
|
+
if (c.aggregated && year >= c.aggregated.first_year && year <= c.aggregated.last_year)
|
|
683
|
+
return "aggregated";
|
|
684
|
+
return null;
|
|
685
|
+
}
|
|
686
|
+
/**
|
|
687
|
+
* Compatibilidade: o intervalo aceito pelas taxas (união dos dois arquivos).
|
|
688
|
+
* Devolve null sem população.
|
|
689
|
+
*/
|
|
690
|
+
export async function getPopulationYearRange() {
|
|
691
|
+
const c = await getPopulationCoverage();
|
|
692
|
+
return c ? { first_year: c.first_year, last_year: c.last_year } : null;
|
|
693
|
+
}
|
|
694
|
+
/**
|
|
695
|
+
* Query população de pop_municipios.parquet agregando por UF
|
|
696
|
+
*/
|
|
697
|
+
async function queryPopulationFromMunicipios(options) {
|
|
698
|
+
const pattern = getPopulationPattern("municipios");
|
|
699
|
+
const conditions = [];
|
|
700
|
+
if (options.years && options.years.length > 0) {
|
|
701
|
+
conditions.push(`year IN (${options.years.join(", ")})`);
|
|
702
|
+
}
|
|
703
|
+
if (options.ufs && options.ufs.length > 0) {
|
|
704
|
+
const ufs = options.ufs.map((u) => `'${u}'`).join(", ");
|
|
705
|
+
conditions.push(`uf IN (${ufs})`);
|
|
706
|
+
}
|
|
707
|
+
if (options.sex) {
|
|
708
|
+
conditions.push(`sex = '${options.sex}'`);
|
|
709
|
+
}
|
|
710
|
+
const whereClause = conditions.length > 0 ? conditions.join(" AND ") : "1=1";
|
|
711
|
+
const groupBy = options.groupBy || [];
|
|
712
|
+
const selectGroups = groupBy.length > 0 ? groupBy.map(groupSelect).join(", ") + ", " : "";
|
|
713
|
+
const groupByClause = groupBy.length > 0 ? `GROUP BY ${groupBy.map(groupKey).join(", ")}` : "";
|
|
714
|
+
const sql = `
|
|
715
|
+
SELECT
|
|
716
|
+
${selectGroups}
|
|
717
|
+
SUM(population) as population
|
|
718
|
+
FROM read_parquet('${pattern}', union_by_name = true)
|
|
719
|
+
WHERE ${whereClause}
|
|
720
|
+
${groupByClause}${buildOrderBy(undefined, groupBy)}
|
|
721
|
+
`;
|
|
722
|
+
return query(sql);
|
|
723
|
+
}
|
|
724
|
+
/**
|
|
725
|
+
* Obtém população total para um filtro específico.
|
|
726
|
+
* Hierarquia de fontes:
|
|
727
|
+
* 1. pop_uf.parquet (idade simples, anos >= 2000)
|
|
728
|
+
* 2. pop_uf_agregado.parquet (faixa etária, anos 1991-1999)
|
|
729
|
+
* 3. pop_municipios.parquet (fallback, agregando por UF)
|
|
730
|
+
*/
|
|
731
|
+
export async function getPopulation(options) {
|
|
732
|
+
if (!hasPopulationData()) {
|
|
733
|
+
throw new Error(POPULATION_MISSING_MESSAGE);
|
|
734
|
+
}
|
|
735
|
+
// Tenta pop_uf.parquet para anos >= 2000 (idade simples)
|
|
736
|
+
if (options.year >= 2000 && hasPopUf()) {
|
|
737
|
+
const result = await queryPopulationUf({
|
|
738
|
+
filters: {
|
|
739
|
+
years: [options.year],
|
|
740
|
+
ufs: options.uf,
|
|
741
|
+
sex: options.sex,
|
|
742
|
+
ageMin: options.ageMin,
|
|
743
|
+
ageMax: options.ageMax,
|
|
744
|
+
},
|
|
745
|
+
});
|
|
746
|
+
if (result[0]?.population)
|
|
747
|
+
return result[0].population;
|
|
748
|
+
}
|
|
749
|
+
// pop_uf_agregado.parquet para anos < 2000 (sih:taxas-1992-1999): idade só
|
|
750
|
+
// nas faixas do arquivo — intervalo fora dos limites das faixas é erro, não
|
|
751
|
+
// aproximação; e com filtro de idade não há fallback para os municípios,
|
|
752
|
+
// que não têm idade.
|
|
753
|
+
if (options.year < 2000) {
|
|
754
|
+
const ageGroups = aggregatedAgeGroupsFor(options.ageMin, options.ageMax);
|
|
755
|
+
if (ageGroups === null) {
|
|
756
|
+
throw new Error(`Antes de 2000 a população por UF só existe em faixas etárias quinquenais (${AGGREGATED_AGE_GROUPS.join(", ")}): ` +
|
|
757
|
+
`o intervalo de idade ${options.ageMin ?? 0}–${options.ageMax ?? "+"} não cai nos limites das faixas. Use age_min múltiplo de 5 e age_max terminado em 4 ou 9 (ou 80+).`);
|
|
758
|
+
}
|
|
759
|
+
const result = await queryPopulationUfAgregado({
|
|
760
|
+
years: [options.year],
|
|
761
|
+
ufs: options.uf,
|
|
762
|
+
sex: options.sex,
|
|
763
|
+
ageGroups,
|
|
764
|
+
});
|
|
765
|
+
if (result[0]?.population)
|
|
766
|
+
return result[0].population;
|
|
767
|
+
if (ageGroups)
|
|
768
|
+
return 0;
|
|
769
|
+
}
|
|
770
|
+
// Fallback: pop_municipios.parquet (agregando por UF)
|
|
771
|
+
const result = await queryPopulationFromMunicipios({
|
|
772
|
+
years: [options.year],
|
|
773
|
+
ufs: options.uf,
|
|
774
|
+
sex: options.sex,
|
|
775
|
+
});
|
|
776
|
+
return result[0]?.population || 0;
|
|
777
|
+
}
|
|
778
|
+
/**
|
|
779
|
+
* Obtém população por UF para um ano.
|
|
780
|
+
* Usa pop_uf.parquet se disponível, senão agrega pop_municipios.parquet.
|
|
781
|
+
*/
|
|
782
|
+
export async function getPopulationByUf(options) {
|
|
783
|
+
if (!hasPopulationData()) {
|
|
784
|
+
throw new Error(POPULATION_MISSING_MESSAGE);
|
|
785
|
+
}
|
|
786
|
+
let result;
|
|
787
|
+
if (options.year >= 2000 && hasPopUf()) {
|
|
788
|
+
result = await queryPopulationUf({
|
|
789
|
+
filters: {
|
|
790
|
+
years: [options.year],
|
|
791
|
+
sex: options.sex,
|
|
792
|
+
ageMin: options.ageMin,
|
|
793
|
+
ageMax: options.ageMax,
|
|
794
|
+
},
|
|
795
|
+
groupBy: ["uf"],
|
|
796
|
+
});
|
|
797
|
+
}
|
|
798
|
+
else if (options.year < 2000) {
|
|
799
|
+
const ageGroups = aggregatedAgeGroupsFor(options.ageMin, options.ageMax);
|
|
800
|
+
if (ageGroups === null) {
|
|
801
|
+
throw new Error(`Antes de 2000 a população por UF só existe em faixas etárias quinquenais (${AGGREGATED_AGE_GROUPS.join(", ")}): ` +
|
|
802
|
+
`o intervalo de idade ${options.ageMin ?? 0}–${options.ageMax ?? "+"} não cai nos limites das faixas.`);
|
|
803
|
+
}
|
|
804
|
+
try {
|
|
805
|
+
result = await queryPopulationUfAgregado({
|
|
806
|
+
years: [options.year],
|
|
807
|
+
sex: options.sex,
|
|
808
|
+
ageGroups,
|
|
809
|
+
groupBy: ["uf"],
|
|
810
|
+
});
|
|
811
|
+
}
|
|
812
|
+
catch {
|
|
813
|
+
if (ageGroups)
|
|
814
|
+
throw new Error("pop_uf_agregado.parquet indisponível: sem população por faixa etária antes de 2000.");
|
|
815
|
+
// Fallback (sem idade)
|
|
816
|
+
result = await queryPopulationFromMunicipios({
|
|
817
|
+
years: [options.year],
|
|
818
|
+
sex: options.sex,
|
|
819
|
+
groupBy: ["uf"],
|
|
820
|
+
});
|
|
821
|
+
}
|
|
822
|
+
}
|
|
823
|
+
else {
|
|
824
|
+
// Fallback: pop_municipios.parquet agregado por UF
|
|
825
|
+
result = await queryPopulationFromMunicipios({
|
|
826
|
+
years: [options.year],
|
|
827
|
+
sex: options.sex,
|
|
828
|
+
groupBy: ["uf"],
|
|
829
|
+
});
|
|
830
|
+
}
|
|
831
|
+
const byUf = {};
|
|
832
|
+
for (const row of result) {
|
|
833
|
+
byUf[row.uf] = row.population;
|
|
834
|
+
}
|
|
835
|
+
return byUf;
|
|
836
|
+
}
|
|
837
|
+
//# sourceMappingURL=duckdb.js.map
|