datera 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +132 -0
- package/dist/api/clean.d.ts +138 -0
- package/dist/api/clean.js +138 -0
- package/dist/api/formats.d.ts +19 -0
- package/dist/api/formats.js +106 -0
- package/dist/api/index.d.ts +8 -0
- package/dist/api/index.js +27 -0
- package/dist/api/inspect.d.ts +13 -0
- package/dist/api/inspect.js +44 -0
- package/dist/api/parse.d.ts +2 -0
- package/dist/api/parse.js +19 -0
- package/dist/api/rows.d.ts +23 -0
- package/dist/api/rows.js +63 -0
- package/dist/cli.d.ts +62 -0
- package/dist/cli.js +225 -0
- package/dist/commands.d.ts +3 -0
- package/dist/commands.js +169 -0
- package/dist/config/check.d.ts +11 -0
- package/dist/config/check.js +99 -0
- package/dist/config/index.d.ts +7 -0
- package/dist/config/index.js +16 -0
- package/dist/config/ioSchema.d.ts +97 -0
- package/dist/config/ioSchema.js +111 -0
- package/dist/config/load.d.ts +5 -0
- package/dist/config/load.js +68 -0
- package/dist/config/mergeSchema.d.ts +36 -0
- package/dist/config/mergeSchema.js +46 -0
- package/dist/config/messages.d.ts +7 -0
- package/dist/config/messages.js +33 -0
- package/dist/config/prepareSchema.d.ts +12 -0
- package/dist/config/prepareSchema.js +15 -0
- package/dist/config/schema.d.ts +195 -0
- package/dist/config/schema.js +44 -0
- package/dist/config/validationSchema.d.ts +27 -0
- package/dist/config/validationSchema.js +50 -0
- package/dist/env.d.ts +2 -0
- package/dist/env.js +15 -0
- package/dist/filters/combine.d.ts +10 -0
- package/dist/filters/combine.js +60 -0
- package/dist/filters/combineTypes.d.ts +6 -0
- package/dist/filters/combineTypes.js +2 -0
- package/dist/filters/dedupe.d.ts +12 -0
- package/dist/filters/dedupe.js +45 -0
- package/dist/filters/fillEmpty.d.ts +9 -0
- package/dist/filters/fillEmpty.js +32 -0
- package/dist/filters/fillEmptyTypes.d.ts +5 -0
- package/dist/filters/fillEmptyTypes.js +2 -0
- package/dist/filters/merge.d.ts +15 -0
- package/dist/filters/merge.js +113 -0
- package/dist/filters/mergeStrategies.d.ts +10 -0
- package/dist/filters/mergeStrategies.js +69 -0
- package/dist/filters/mergeTypes.d.ts +32 -0
- package/dist/filters/mergeTypes.js +2 -0
- package/dist/filters/mergeUnkeyed.d.ts +3 -0
- package/dist/filters/mergeUnkeyed.js +21 -0
- package/dist/filters/types.d.ts +3 -0
- package/dist/filters/types.js +2 -0
- package/dist/filters/validate.d.ts +16 -0
- package/dist/filters/validate.js +79 -0
- package/dist/filters/validateTypes.d.ts +28 -0
- package/dist/filters/validateTypes.js +2 -0
- package/dist/index.d.ts +13 -0
- package/dist/index.js +40 -0
- package/dist/io/csv/csvFormat.d.ts +4 -0
- package/dist/io/csv/csvFormat.js +94 -0
- package/dist/io/csv/csvSink.d.ts +14 -0
- package/dist/io/csv/csvSink.js +32 -0
- package/dist/io/csv/csvSource.d.ts +13 -0
- package/dist/io/csv/csvSource.js +29 -0
- package/dist/io/custom/loader.d.ts +1 -0
- package/dist/io/custom/loader.js +36 -0
- package/dist/io/custom/registry.d.ts +13 -0
- package/dist/io/custom/registry.js +50 -0
- package/dist/io/excel/excelCell.d.ts +4 -0
- package/dist/io/excel/excelCell.js +29 -0
- package/dist/io/excel/excelSink.d.ts +14 -0
- package/dist/io/excel/excelSink.js +57 -0
- package/dist/io/excel/excelSource.d.ts +11 -0
- package/dist/io/excel/excelSource.js +47 -0
- package/dist/io/excel/workbook.d.ts +4 -0
- package/dist/io/excel/workbook.js +20 -0
- package/dist/io/factory.d.ts +5 -0
- package/dist/io/factory.js +106 -0
- package/dist/io/files.d.ts +7 -0
- package/dist/io/files.js +48 -0
- package/dist/io/header.d.ts +2 -0
- package/dist/io/header.js +42 -0
- package/dist/io/json/jsonSink.d.ts +12 -0
- package/dist/io/json/jsonSink.js +19 -0
- package/dist/io/json/jsonSource.d.ts +11 -0
- package/dist/io/json/jsonSource.js +66 -0
- package/dist/io/mysql/client.d.ts +13 -0
- package/dist/io/mysql/client.js +73 -0
- package/dist/io/mysql/mysqlSource.d.ts +13 -0
- package/dist/io/mysql/mysqlSource.js +25 -0
- package/dist/io/optional.d.ts +2 -0
- package/dist/io/optional.js +36 -0
- package/dist/io/parquet/parquetLibrary.d.ts +11 -0
- package/dist/io/parquet/parquetLibrary.js +20 -0
- package/dist/io/parquet/parquetSink.d.ts +15 -0
- package/dist/io/parquet/parquetSink.js +58 -0
- package/dist/io/parquet/parquetSource.d.ts +13 -0
- package/dist/io/parquet/parquetSource.js +68 -0
- package/dist/io/postgres/postgresSource.d.ts +26 -0
- package/dist/io/postgres/postgresSource.js +42 -0
- package/dist/io/safety.d.ts +4 -0
- package/dist/io/safety.js +83 -0
- package/dist/io/sheets/client.d.ts +10 -0
- package/dist/io/sheets/client.js +141 -0
- package/dist/io/sheets/sheetsSink.d.ts +12 -0
- package/dist/io/sheets/sheetsSink.js +24 -0
- package/dist/io/sheets/sheetsSource.d.ts +10 -0
- package/dist/io/sheets/sheetsSource.js +47 -0
- package/dist/io/sql/cell.d.ts +5 -0
- package/dist/io/sql/cell.js +38 -0
- package/dist/io/sql/driver.d.ts +1 -0
- package/dist/io/sql/driver.js +15 -0
- package/dist/io/sql/names.d.ts +1 -0
- package/dist/io/sql/names.js +12 -0
- package/dist/io/sqlite/sqliteSource.d.ts +11 -0
- package/dist/io/sqlite/sqliteSource.js +52 -0
- package/dist/io/sqlserver/sqlServerSource.d.ts +24 -0
- package/dist/io/sqlserver/sqlServerSource.js +49 -0
- package/dist/io/types.d.ts +11 -0
- package/dist/io/types.js +2 -0
- package/dist/io/xml/xmlParse.d.ts +9 -0
- package/dist/io/xml/xmlParse.js +178 -0
- package/dist/io/xml/xmlSink.d.ts +18 -0
- package/dist/io/xml/xmlSink.js +79 -0
- package/dist/io/xml/xmlSource.d.ts +11 -0
- package/dist/io/xml/xmlSource.js +93 -0
- package/dist/logger.d.ts +11 -0
- package/dist/logger.js +29 -0
- package/dist/main.d.ts +2 -0
- package/dist/main.js +18 -0
- package/dist/normalizers/builtin.d.ts +5 -0
- package/dist/normalizers/builtin.js +23 -0
- package/dist/normalizers/index.d.ts +2 -0
- package/dist/normalizers/index.js +14 -0
- package/dist/normalizers/loader.d.ts +1 -0
- package/dist/normalizers/loader.js +30 -0
- package/dist/normalizers/registry.d.ts +9 -0
- package/dist/normalizers/registry.js +29 -0
- package/dist/pipeline/index.d.ts +7 -0
- package/dist/pipeline/index.js +17 -0
- package/dist/pipeline/modes.d.ts +6 -0
- package/dist/pipeline/modes.js +48 -0
- package/dist/pipeline/report.d.ts +3 -0
- package/dist/pipeline/report.js +32 -0
- package/dist/pipeline/runEtl.d.ts +24 -0
- package/dist/pipeline/runEtl.js +98 -0
- package/dist/pipeline/steps.d.ts +4 -0
- package/dist/pipeline/steps.js +36 -0
- package/dist/pipeline/types.d.ts +34 -0
- package/dist/pipeline/types.js +2 -0
- package/dist/types.d.ts +3 -0
- package/dist/types.js +2 -0
- package/package.json +128 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alessandra Vieira
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="docs/assets/datera-banner.png" alt="Datera: dados bagunçados entrando de um lado e saindo organizados do outro">
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
<p align="center">Ferramenta pra limpar e juntar dados bagunçados de planilha, banco e arquivo.</p>
|
|
6
|
+
|
|
7
|
+
<p align="center">
|
|
8
|
+
<a href="https://github.com/alessandravieiradev-blip/datera/actions/workflows/ci.yml"><img src="https://github.com/alessandravieiradev-blip/datera/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
9
|
+
</p>
|
|
10
|
+
|
|
11
|
+
<p align="center">
|
|
12
|
+
<a href="docs/gestores.md"><img src="https://img.shields.io/badge/Para%20gestores-2563EB?style=for-the-badge" alt="Guia para gestores"></a>
|
|
13
|
+
<a href="docs/devs.md"><img src="https://img.shields.io/badge/Para%20devs-16181D?style=for-the-badge" alt="Guia para devs"></a>
|
|
14
|
+
</p>
|
|
15
|
+
|
|
16
|
+
## O que é
|
|
17
|
+
|
|
18
|
+
O Datera é um ETL que eu fiz em TypeScript. Ele lê dados de banco, planilha ou arquivo, arruma o que dá e escreve o resultado organizado em outro lugar:
|
|
19
|
+
|
|
20
|
+
```mermaid
|
|
21
|
+
flowchart LR
|
|
22
|
+
subgraph fontes["Lê de"]
|
|
23
|
+
direction TB
|
|
24
|
+
B["Bancos<br/>MySQL · PostgreSQL<br/>SQL Server · SQLite"]
|
|
25
|
+
P1["Planilhas<br/>Google Sheets · Excel"]
|
|
26
|
+
A1["Arquivos<br/>CSV · JSON<br/>XML · Parquet"]
|
|
27
|
+
end
|
|
28
|
+
subgraph destinos["Escreve em"]
|
|
29
|
+
direction TB
|
|
30
|
+
P2["Planilhas<br/>Google Sheets · Excel"]
|
|
31
|
+
A2["Arquivos<br/>CSV · JSON<br/>XML · Parquet"]
|
|
32
|
+
end
|
|
33
|
+
B --> D((Datera))
|
|
34
|
+
P1 --> D
|
|
35
|
+
A1 --> D
|
|
36
|
+
D --> P2
|
|
37
|
+
D --> A2
|
|
38
|
+
classDef datera fill:#2563EB,stroke:#2563EB,color:#ffffff
|
|
39
|
+
class D datera
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Ele começou bem simples, era só pra copiar uma tabela do MySQL pra uma planilha. Só que aí eu fui vendo que dado de verdade vem uma bagunça: a mesma pessoa cadastrada duas vezes, e-mail com maiúscula num lugar e minúscula no outro, campo vazio, informação espalhada em várias colunas. Então fui colocando coisa nova até conseguir resolver quase tudo mexendo só no `config.json`.
|
|
43
|
+
|
|
44
|
+
## A ideia
|
|
45
|
+
|
|
46
|
+
Quem mais sofre com planilha bagunçada normalmente não é quem programa. É quem precisa da lista certinha pra mandar um e-mail, fechar um relatório ou ligar pra alguém. E essa pessoa não deveria ter que pedir ajuda toda vez que precisa dos dados limpos.
|
|
47
|
+
|
|
48
|
+
Então o Datera funciona assim: as regras ficam num arquivo de configuração (o `config.json`), que pode ser montado com calma uma vez só. Depois disso qualquer pessoa roda com um clique e recebe duas coisas: o resultado organizado e uma lista separada do que precisa de alguém dar uma olhada, com o motivo escrito do lado. Nada some sem explicação.
|
|
49
|
+
|
|
50
|
+
Por isso ele tem três jeitos de usar, todos com o mesmo motor e a mesma configuração:
|
|
51
|
+
|
|
52
|
+
| Jeito | Pra quem | O que tem |
|
|
53
|
+
| -------------- | --------------------------------------------- | ------------------------------------------------------------------------- |
|
|
54
|
+
| **Datera** | quem cuida dos dados mas não programa | app com tela, passo a passo, regras em frases, gráficos e histórico |
|
|
55
|
+
| **Datera Dev** | quem programa | app escuro com editor da config, prévia na hora, log e paleta de comandos |
|
|
56
|
+
| **Terminal** | quem quer automatizar ou usar em outro código | o comando `datera`, `--dry-run` e a função `runEtl` |
|
|
57
|
+
|
|
58
|
+
## Como funciona
|
|
59
|
+
|
|
60
|
+
```text
|
|
61
|
+
fonte ──► prepara ──► separa o que tem problema ──► tira ou junta repetidos ──► destino
|
|
62
|
+
│
|
|
63
|
+
└──► pendências (com o motivo de cada uma)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
1. **Fonte:** lê de onde os dados estão.
|
|
67
|
+
2. **Prepara:** preenche célula vazia e junta colunas, se você pedir.
|
|
68
|
+
3. **Pendências:** as linhas que quebram alguma regra (e-mail inválido, campo vazio, valor fora da lista...) vão pra uma aba ou arquivo à parte.
|
|
69
|
+
4. **Repetidos:** deixa como está, tira as linhas repetidas ou junta as linhas da mesma pessoa sem perder nada.
|
|
70
|
+
5. **Destino:** escreve o resultado onde você escolheu.
|
|
71
|
+
|
|
72
|
+
Hoje ele consegue:
|
|
73
|
+
|
|
74
|
+
- tirar linhas repetidas (`dedupe`) ou juntar as linhas da mesma pessoa sem perder nada (`merge`)
|
|
75
|
+
- usar normalizadores pra `Lia@Email.com` e ` lia@email.com` contarem como a mesma pessoa
|
|
76
|
+
- ter regras diferentes dependendo do tipo da chave
|
|
77
|
+
- decidir o que fazer com as linhas que não têm chave, sem elas sumirem nem se misturarem
|
|
78
|
+
- espalhar valores em várias colunas sem jogar nenhum fora
|
|
79
|
+
- ler e escrever em formatos diferentes sem mudar nada das regras
|
|
80
|
+
|
|
81
|
+
## Escolha o seu caminho
|
|
82
|
+
|
|
83
|
+
<table>
|
|
84
|
+
<tr>
|
|
85
|
+
<td width="50%" valign="top">
|
|
86
|
+
|
|
87
|
+
### Para gestores
|
|
88
|
+
|
|
89
|
+
Você quer usar o app com tela: instalar, escolher de onde vêm os dados, montar as regras em frases e exportar. Não precisa saber programar.
|
|
90
|
+
|
|
91
|
+
<a href="https://github.com/alessandravieiradev-blip/datera/releases/latest"><img src="https://img.shields.io/badge/Baixar%20o%20Datera-2563EB?style=for-the-badge" alt="Baixar o Datera"></a>
|
|
92
|
+
<a href="docs/gestores.md"><img src="https://img.shields.io/badge/Abrir%20o%20guia-475569?style=for-the-badge" alt="Abrir o guia para gestores"></a>
|
|
93
|
+
|
|
94
|
+
</td>
|
|
95
|
+
<td width="50%" valign="top">
|
|
96
|
+
|
|
97
|
+
### Para devs
|
|
98
|
+
|
|
99
|
+
Você quer rodar pelo terminal, usar o Datera Dev, escrever a config na mão, criar normalizadores ou chamar o Datera de dentro de outro código.
|
|
100
|
+
|
|
101
|
+
<a href="docs/devs.md"><img src="https://img.shields.io/badge/Abrir%20o%20guia%20para%20devs-16181D?style=for-the-badge" alt="Abrir o guia para devs"></a>
|
|
102
|
+
<a href="docs/api.md"><img src="https://img.shields.io/badge/Comandos%20e%20API-475569?style=for-the-badge" alt="Comandos e API"></a>
|
|
103
|
+
|
|
104
|
+
</td>
|
|
105
|
+
</tr>
|
|
106
|
+
</table>
|
|
107
|
+
|
|
108
|
+
## Todos os guias
|
|
109
|
+
|
|
110
|
+
| Guia | O que tem |
|
|
111
|
+
| ---------------------------------------------------------- | -------------------------------------------------------------------- |
|
|
112
|
+
| [Para gestores](docs/gestores.md) | instalar e usar o app com tela |
|
|
113
|
+
| [Para devs](docs/devs.md) | teste em 1 minuto, terminal, Datera Dev, uso em código e testes |
|
|
114
|
+
| [Comandos e API](docs/api.md) | os comandos do terminal e as funções pra usar no seu código |
|
|
115
|
+
| [Configuração](docs/configuracao.md) | todos os campos do `config.json`, os modos e as estratégias do merge |
|
|
116
|
+
| [Fontes e destinos](docs/fontes-e-destinos.md) | cada formato que ele lê e escreve, e como criar o seu |
|
|
117
|
+
| [Exemplos](docs/exemplos.md) | dez configs, do mais simples ao mais completo |
|
|
118
|
+
| [Preparação e pendências](docs/preparacao-e-pendencias.md) | `fillEmpty`, `combineColumns`, `distribute` e `validation` |
|
|
119
|
+
| [Normalizadores](docs/normalizadores.md) | os prontos e como criar o seu |
|
|
120
|
+
| [Estrutura do projeto](docs/estrutura.md) | o que tem em cada pasta |
|
|
121
|
+
|
|
122
|
+
## Contribuir e licença
|
|
123
|
+
|
|
124
|
+
Se quiser ajudar, o jeito de rodar, testar e mandar mudanças está no [CONTRIBUTING.md](CONTRIBUTING.md). O código é aberto, sob a [licença MIT](LICENSE).
|
|
125
|
+
|
|
126
|
+
## Próximos passos
|
|
127
|
+
|
|
128
|
+
A ideia é chegar num instalador de um clique pra quem não programa e numa biblioteca no npm pra quem programa.
|
|
129
|
+
|
|
130
|
+
O que eu quero fazer fica nas [issues](https://github.com/alessandravieiradev-blip/datera/issues), agrupadas em etapas nos [milestones](https://github.com/alessandravieiradev-blip/datera/milestones). Lá dá pra ver o que já foi feito e o que falta, e tudo se atualiza sozinho conforme eu vou fazendo.
|
|
131
|
+
|
|
132
|
+
Se quiser saber no que eu tô trabalhando agora, me chama no Instagram ou aqui no GitHub. Os dois estão no [meu perfil](https://github.com/alessandravieiradev-blip).
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { etlConfigSchema } from "../config";
|
|
3
|
+
import { mergeColumnSchema } from "../config/mergeSchema";
|
|
4
|
+
import { combineColumnsSchema, fillEmptySchema } from "../config/prepareSchema";
|
|
5
|
+
import { validationSchema } from "../config/validationSchema";
|
|
6
|
+
import { PendingReason, StepReport } from "../pipeline";
|
|
7
|
+
import { TableRow } from "../types";
|
|
8
|
+
export declare const rulesSchema: z.ZodObject<{
|
|
9
|
+
fillEmpty: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
10
|
+
column: z.ZodString;
|
|
11
|
+
fallbackColumns: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
12
|
+
default: z.ZodOptional<z.ZodString>;
|
|
13
|
+
}, z.core.$strip>>>;
|
|
14
|
+
combineColumns: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
15
|
+
into: z.ZodString;
|
|
16
|
+
columns: z.ZodArray<z.ZodString>;
|
|
17
|
+
separator: z.ZodOptional<z.ZodString>;
|
|
18
|
+
keepSources: z.ZodOptional<z.ZodBoolean>;
|
|
19
|
+
}, z.core.$strip>>>;
|
|
20
|
+
validation: z.ZodOptional<z.ZodObject<{
|
|
21
|
+
rules: z.ZodArray<z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
22
|
+
message: z.ZodOptional<z.ZodString>;
|
|
23
|
+
column: z.ZodString;
|
|
24
|
+
rule: z.ZodLiteral<"required">;
|
|
25
|
+
}, z.core.$strip>, z.ZodObject<{
|
|
26
|
+
message: z.ZodOptional<z.ZodString>;
|
|
27
|
+
column: z.ZodString;
|
|
28
|
+
rule: z.ZodLiteral<"pattern">;
|
|
29
|
+
pattern: z.ZodString;
|
|
30
|
+
flags: z.ZodOptional<z.ZodString>;
|
|
31
|
+
}, z.core.$strip>, z.ZodObject<{
|
|
32
|
+
message: z.ZodOptional<z.ZodString>;
|
|
33
|
+
column: z.ZodString;
|
|
34
|
+
rule: z.ZodLiteral<"oneOf">;
|
|
35
|
+
values: z.ZodArray<z.ZodString>;
|
|
36
|
+
ignoreCase: z.ZodOptional<z.ZodBoolean>;
|
|
37
|
+
}, z.core.$strip>, z.ZodObject<{
|
|
38
|
+
message: z.ZodOptional<z.ZodString>;
|
|
39
|
+
column: z.ZodString;
|
|
40
|
+
rule: z.ZodLiteral<"normalizer">;
|
|
41
|
+
normalizer: z.ZodString;
|
|
42
|
+
}, z.core.$strip>], "rule">>;
|
|
43
|
+
pendingSheet: z.ZodOptional<z.ZodString>;
|
|
44
|
+
reasonColumn: z.ZodOptional<z.ZodString>;
|
|
45
|
+
}, z.core.$strip>>;
|
|
46
|
+
dedupe: z.ZodOptional<z.ZodObject<{
|
|
47
|
+
column: z.ZodString;
|
|
48
|
+
keep: z.ZodOptional<z.ZodEnum<{
|
|
49
|
+
first: "first";
|
|
50
|
+
last: "last";
|
|
51
|
+
}>>;
|
|
52
|
+
normalizer: z.ZodOptional<z.ZodString>;
|
|
53
|
+
}, z.core.$strip>>;
|
|
54
|
+
merge: z.ZodOptional<z.ZodObject<{
|
|
55
|
+
key: z.ZodString;
|
|
56
|
+
columns: z.ZodArray<z.ZodObject<{
|
|
57
|
+
column: z.ZodString;
|
|
58
|
+
strategy: z.ZodEnum<{
|
|
59
|
+
concat: "concat";
|
|
60
|
+
"extra-column": "extra-column";
|
|
61
|
+
overwrite: "overwrite";
|
|
62
|
+
}>;
|
|
63
|
+
separator: z.ZodOptional<z.ZodString>;
|
|
64
|
+
distribute: z.ZodOptional<z.ZodObject<{
|
|
65
|
+
columns: z.ZodArray<z.ZodString>;
|
|
66
|
+
overflowInto: z.ZodOptional<z.ZodString>;
|
|
67
|
+
sources: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
68
|
+
}, z.core.$strip>>;
|
|
69
|
+
byGroup: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodObject<{
|
|
70
|
+
strategy: z.ZodEnum<{
|
|
71
|
+
concat: "concat";
|
|
72
|
+
"extra-column": "extra-column";
|
|
73
|
+
overwrite: "overwrite";
|
|
74
|
+
}>;
|
|
75
|
+
separator: z.ZodOptional<z.ZodString>;
|
|
76
|
+
into: z.ZodOptional<z.ZodString>;
|
|
77
|
+
distribute: z.ZodOptional<z.ZodObject<{
|
|
78
|
+
columns: z.ZodArray<z.ZodString>;
|
|
79
|
+
overflowInto: z.ZodOptional<z.ZodString>;
|
|
80
|
+
sources: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
81
|
+
}, z.core.$strip>>;
|
|
82
|
+
}, z.core.$strip>>>;
|
|
83
|
+
unkeyed: z.ZodOptional<z.ZodObject<{
|
|
84
|
+
strategy: z.ZodEnum<{
|
|
85
|
+
"collapse-column": "collapse-column";
|
|
86
|
+
}>;
|
|
87
|
+
into: z.ZodOptional<z.ZodString>;
|
|
88
|
+
separator: z.ZodOptional<z.ZodString>;
|
|
89
|
+
}, z.core.$strip>>;
|
|
90
|
+
}, z.core.$strip>>;
|
|
91
|
+
normalizer: z.ZodOptional<z.ZodString>;
|
|
92
|
+
emptyKeyLabel: z.ZodOptional<z.ZodString>;
|
|
93
|
+
rejectedKeyLabel: z.ZodOptional<z.ZodString>;
|
|
94
|
+
}, z.core.$strip>>;
|
|
95
|
+
}, z.core.$strip>;
|
|
96
|
+
export type Rules = z.input<typeof rulesSchema>;
|
|
97
|
+
export type ConfigInput = z.input<typeof etlConfigSchema>;
|
|
98
|
+
export interface CleanOptions {
|
|
99
|
+
onStep?: ((report: StepReport) => void) | undefined;
|
|
100
|
+
}
|
|
101
|
+
export interface CleanResult {
|
|
102
|
+
rows: TableRow[];
|
|
103
|
+
pending: TableRow[];
|
|
104
|
+
pendingByReason: PendingReason[];
|
|
105
|
+
steps: StepReport[];
|
|
106
|
+
durationMs: number;
|
|
107
|
+
}
|
|
108
|
+
export declare function defineConfig(config: ConfigInput): ConfigInput;
|
|
109
|
+
export declare function defineRules(rules: Rules): Rules;
|
|
110
|
+
export declare function clean(rows: TableRow[], rules?: Rules, options?: CleanOptions): CleanResult;
|
|
111
|
+
type FillEmptyRule = z.input<typeof fillEmptySchema>;
|
|
112
|
+
type CombineRule = z.input<typeof combineColumnsSchema>;
|
|
113
|
+
type Validation = z.input<typeof validationSchema>;
|
|
114
|
+
type ValidationRule = Validation["rules"][number];
|
|
115
|
+
type MergeColumn = z.input<typeof mergeColumnSchema>;
|
|
116
|
+
export declare function fillEmpty(rows: TableRow[], rules: FillEmptyRule[]): TableRow[];
|
|
117
|
+
export declare function combineColumns(rows: TableRow[], rules: CombineRule[]): TableRow[];
|
|
118
|
+
export declare function validate(rows: TableRow[], rules: ValidationRule[] | Validation): {
|
|
119
|
+
valid: TableRow[];
|
|
120
|
+
pending: TableRow[];
|
|
121
|
+
};
|
|
122
|
+
export interface DedupeOptions {
|
|
123
|
+
keep?: "first" | "last" | undefined;
|
|
124
|
+
normalizer?: string | undefined;
|
|
125
|
+
}
|
|
126
|
+
export declare function dedupe(rows: TableRow[], column: string, options?: DedupeOptions): TableRow[];
|
|
127
|
+
export interface MergeOptions {
|
|
128
|
+
columns?: MergeColumn[] | undefined;
|
|
129
|
+
normalizer?: string | undefined;
|
|
130
|
+
separator?: string | undefined;
|
|
131
|
+
overwrite?: string[] | undefined;
|
|
132
|
+
extraColumn?: string[] | undefined;
|
|
133
|
+
emptyKeyLabel?: string | undefined;
|
|
134
|
+
rejectedKeyLabel?: string | undefined;
|
|
135
|
+
}
|
|
136
|
+
export declare function mergeColumnsFor(rows: TableRow[], key: string, options?: MergeOptions): MergeColumn[];
|
|
137
|
+
export declare function merge(rows: TableRow[], key: string, options?: MergeOptions): TableRow[];
|
|
138
|
+
export {};
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.rulesSchema = void 0;
|
|
4
|
+
exports.defineConfig = defineConfig;
|
|
5
|
+
exports.defineRules = defineRules;
|
|
6
|
+
exports.clean = clean;
|
|
7
|
+
exports.fillEmpty = fillEmpty;
|
|
8
|
+
exports.combineColumns = combineColumns;
|
|
9
|
+
exports.validate = validate;
|
|
10
|
+
exports.dedupe = dedupe;
|
|
11
|
+
exports.mergeColumnsFor = mergeColumnsFor;
|
|
12
|
+
exports.merge = merge;
|
|
13
|
+
const zod_1 = require("zod");
|
|
14
|
+
const mergeSchema_1 = require("../config/mergeSchema");
|
|
15
|
+
const prepareSchema_1 = require("../config/prepareSchema");
|
|
16
|
+
const validationSchema_1 = require("../config/validationSchema");
|
|
17
|
+
const normalizers_1 = require("../normalizers");
|
|
18
|
+
const pipeline_1 = require("../pipeline");
|
|
19
|
+
const parse_1 = require("./parse");
|
|
20
|
+
exports.rulesSchema = zod_1.z
|
|
21
|
+
.object({
|
|
22
|
+
fillEmpty: zod_1.z.array(prepareSchema_1.fillEmptySchema).optional(),
|
|
23
|
+
combineColumns: zod_1.z.array(prepareSchema_1.combineColumnsSchema).optional(),
|
|
24
|
+
validation: validationSchema_1.validationSchema.optional(),
|
|
25
|
+
dedupe: zod_1.z
|
|
26
|
+
.object({
|
|
27
|
+
column: zod_1.z.string().min(1),
|
|
28
|
+
keep: zod_1.z.enum(["first", "last"]).optional(),
|
|
29
|
+
normalizer: zod_1.z.string().optional(),
|
|
30
|
+
})
|
|
31
|
+
.optional(),
|
|
32
|
+
merge: zod_1.z
|
|
33
|
+
.object({
|
|
34
|
+
key: zod_1.z.string().min(1),
|
|
35
|
+
columns: zod_1.z.array(mergeSchema_1.mergeColumnSchema).min(1),
|
|
36
|
+
normalizer: zod_1.z.string().optional(),
|
|
37
|
+
emptyKeyLabel: zod_1.z.string().optional(),
|
|
38
|
+
rejectedKeyLabel: zod_1.z.string().optional(),
|
|
39
|
+
})
|
|
40
|
+
.optional(),
|
|
41
|
+
})
|
|
42
|
+
.refine((rules) => !(rules.dedupe && rules.merge), {
|
|
43
|
+
message: "Escolha dedupe ou merge, os dois juntos não dá.",
|
|
44
|
+
path: ["merge"],
|
|
45
|
+
});
|
|
46
|
+
function defineConfig(config) {
|
|
47
|
+
return config;
|
|
48
|
+
}
|
|
49
|
+
function defineRules(rules) {
|
|
50
|
+
return rules;
|
|
51
|
+
}
|
|
52
|
+
function configOf(rules) {
|
|
53
|
+
const mode = rules.merge ? "merge" : rules.dedupe ? "dedupe" : "raw";
|
|
54
|
+
const config = {
|
|
55
|
+
mode,
|
|
56
|
+
fillEmpty: rules.fillEmpty,
|
|
57
|
+
combineColumns: rules.combineColumns,
|
|
58
|
+
validation: rules.validation,
|
|
59
|
+
dedupeColumn: rules.dedupe?.column,
|
|
60
|
+
dedupeStrategy: rules.dedupe?.keep === "last" ? "keep-last" : "keep-first",
|
|
61
|
+
dedupeKeyNormalizer: rules.dedupe?.normalizer,
|
|
62
|
+
mergeKeyColumn: rules.merge?.key,
|
|
63
|
+
mergeColumns: rules.merge?.columns,
|
|
64
|
+
mergeKeyNormalizer: rules.merge?.normalizer,
|
|
65
|
+
mergeEmptyKeyLabel: rules.merge?.emptyKeyLabel,
|
|
66
|
+
mergeRejectedKeyLabel: rules.merge?.rejectedKeyLabel,
|
|
67
|
+
};
|
|
68
|
+
return { config, mode };
|
|
69
|
+
}
|
|
70
|
+
function clean(rows, rules = {}, options = {}) {
|
|
71
|
+
const startedAt = Date.now();
|
|
72
|
+
const parsed = (0, parse_1.parseWith)(exports.rulesSchema, rules, "As regras");
|
|
73
|
+
(0, normalizers_1.registerBuiltinKeyNormalizers)();
|
|
74
|
+
const { config, mode } = configOf(parsed);
|
|
75
|
+
const result = (0, pipeline_1.applySteps)((0, pipeline_1.buildSteps)(config, mode), rows, options.onStep);
|
|
76
|
+
return {
|
|
77
|
+
rows: result.rows,
|
|
78
|
+
pending: result.pending,
|
|
79
|
+
pendingByReason: (0, pipeline_1.countPendingReasons)(result.pending, parsed.validation?.reasonColumn),
|
|
80
|
+
steps: result.steps,
|
|
81
|
+
durationMs: Date.now() - startedAt,
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
function fillEmpty(rows, rules) {
|
|
85
|
+
return clean(rows, { fillEmpty: rules }).rows;
|
|
86
|
+
}
|
|
87
|
+
function combineColumns(rows, rules) {
|
|
88
|
+
return clean(rows, { combineColumns: rules }).rows;
|
|
89
|
+
}
|
|
90
|
+
function validate(rows, rules) {
|
|
91
|
+
const validation = Array.isArray(rules) ? { rules } : rules;
|
|
92
|
+
const result = clean(rows, { validation });
|
|
93
|
+
return { valid: result.rows, pending: result.pending };
|
|
94
|
+
}
|
|
95
|
+
function dedupe(rows, column, options = {}) {
|
|
96
|
+
return clean(rows, {
|
|
97
|
+
dedupe: {
|
|
98
|
+
column,
|
|
99
|
+
...(options.keep ? { keep: options.keep } : {}),
|
|
100
|
+
...(options.normalizer ? { normalizer: options.normalizer } : {}),
|
|
101
|
+
},
|
|
102
|
+
}).rows;
|
|
103
|
+
}
|
|
104
|
+
function mergeColumnsFor(rows, key, options = {}) {
|
|
105
|
+
const names = new Set();
|
|
106
|
+
for (const row of rows) {
|
|
107
|
+
for (const name of Object.keys(row))
|
|
108
|
+
names.add(name);
|
|
109
|
+
}
|
|
110
|
+
names.delete(key);
|
|
111
|
+
return [...names].map((column) => options.overwrite?.includes(column)
|
|
112
|
+
? { column, strategy: "overwrite" }
|
|
113
|
+
: options.extraColumn?.includes(column)
|
|
114
|
+
? { column, strategy: "extra-column" }
|
|
115
|
+
: {
|
|
116
|
+
column,
|
|
117
|
+
strategy: "concat",
|
|
118
|
+
separator: options.separator ?? " | ",
|
|
119
|
+
});
|
|
120
|
+
}
|
|
121
|
+
function merge(rows, key, options = {}) {
|
|
122
|
+
const columns = options.columns ?? mergeColumnsFor(rows, key, options);
|
|
123
|
+
if (columns.length === 0)
|
|
124
|
+
return rows;
|
|
125
|
+
return clean(rows, {
|
|
126
|
+
merge: {
|
|
127
|
+
key,
|
|
128
|
+
columns,
|
|
129
|
+
...(options.normalizer ? { normalizer: options.normalizer } : {}),
|
|
130
|
+
...(options.emptyKeyLabel
|
|
131
|
+
? { emptyKeyLabel: options.emptyKeyLabel }
|
|
132
|
+
: {}),
|
|
133
|
+
...(options.rejectedKeyLabel
|
|
134
|
+
? { rejectedKeyLabel: options.rejectedKeyLabel }
|
|
135
|
+
: {}),
|
|
136
|
+
},
|
|
137
|
+
}).rows;
|
|
138
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { DestinationConfig, SourceConfig } from "../config/ioSchema";
|
|
2
|
+
export type FileFormat = "csv" | "json" | "xml" | "excel" | "parquet" | "sqlite";
|
|
3
|
+
export interface ReadOptions {
|
|
4
|
+
table?: string | undefined;
|
|
5
|
+
sheet?: string | undefined;
|
|
6
|
+
recordsPath?: string | undefined;
|
|
7
|
+
delimiter?: string | undefined;
|
|
8
|
+
encoding?: "utf-8" | "latin1" | undefined;
|
|
9
|
+
}
|
|
10
|
+
export interface WriteOptions {
|
|
11
|
+
sheet?: string | undefined;
|
|
12
|
+
delimiter?: string | undefined;
|
|
13
|
+
bom?: boolean | undefined;
|
|
14
|
+
root?: string | undefined;
|
|
15
|
+
record?: string | undefined;
|
|
16
|
+
}
|
|
17
|
+
export declare function detectFormat(filePath: string): FileFormat;
|
|
18
|
+
export declare function sourceFromPath(filePath: string, options?: ReadOptions): SourceConfig;
|
|
19
|
+
export declare function destinationFromPath(filePath: string, options?: WriteOptions): DestinationConfig;
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
+
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
+
};
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.detectFormat = detectFormat;
|
|
7
|
+
exports.sourceFromPath = sourceFromPath;
|
|
8
|
+
exports.destinationFromPath = destinationFromPath;
|
|
9
|
+
const path_1 = __importDefault(require("path"));
|
|
10
|
+
const EXTENSIONS = {
|
|
11
|
+
".csv": "csv",
|
|
12
|
+
".tsv": "csv",
|
|
13
|
+
".txt": "csv",
|
|
14
|
+
".json": "json",
|
|
15
|
+
".xml": "xml",
|
|
16
|
+
".xlsx": "excel",
|
|
17
|
+
".parquet": "parquet",
|
|
18
|
+
".db": "sqlite",
|
|
19
|
+
".sqlite": "sqlite",
|
|
20
|
+
".sqlite3": "sqlite",
|
|
21
|
+
};
|
|
22
|
+
function defined(value) {
|
|
23
|
+
return Object.fromEntries(Object.entries(value).filter(([, inner]) => inner !== undefined));
|
|
24
|
+
}
|
|
25
|
+
function detectFormat(filePath) {
|
|
26
|
+
const extension = path_1.default.extname(filePath).toLowerCase();
|
|
27
|
+
const format = EXTENSIONS[extension];
|
|
28
|
+
if (!format) {
|
|
29
|
+
throw new Error(`Não sei qual é o formato de "${filePath}" pela extensão. As que eu conheço: ${Object.keys(EXTENSIONS).join(", ")}.`);
|
|
30
|
+
}
|
|
31
|
+
return format;
|
|
32
|
+
}
|
|
33
|
+
function sourceFromPath(filePath, options = {}) {
|
|
34
|
+
const format = detectFormat(filePath);
|
|
35
|
+
switch (format) {
|
|
36
|
+
case "csv":
|
|
37
|
+
return defined({
|
|
38
|
+
type: "csv",
|
|
39
|
+
path: filePath,
|
|
40
|
+
delimiter: options.delimiter ??
|
|
41
|
+
(path_1.default.extname(filePath).toLowerCase() === ".tsv"
|
|
42
|
+
? "\t"
|
|
43
|
+
: undefined),
|
|
44
|
+
encoding: options.encoding,
|
|
45
|
+
});
|
|
46
|
+
case "json":
|
|
47
|
+
return defined({
|
|
48
|
+
type: "json",
|
|
49
|
+
path: filePath,
|
|
50
|
+
recordsPath: options.recordsPath,
|
|
51
|
+
});
|
|
52
|
+
case "xml":
|
|
53
|
+
return defined({
|
|
54
|
+
type: "xml",
|
|
55
|
+
path: filePath,
|
|
56
|
+
recordsPath: options.recordsPath,
|
|
57
|
+
});
|
|
58
|
+
case "excel":
|
|
59
|
+
return defined({
|
|
60
|
+
type: "excel",
|
|
61
|
+
path: filePath,
|
|
62
|
+
sheet: options.sheet,
|
|
63
|
+
});
|
|
64
|
+
case "parquet":
|
|
65
|
+
return { type: "parquet", path: filePath };
|
|
66
|
+
case "sqlite":
|
|
67
|
+
if (!options.table) {
|
|
68
|
+
throw new Error(`Pra ler o SQLite "${filePath}", diga qual tabela (table).`);
|
|
69
|
+
}
|
|
70
|
+
return { type: "sqlite", path: filePath, table: options.table };
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
function destinationFromPath(filePath, options = {}) {
|
|
74
|
+
const format = detectFormat(filePath);
|
|
75
|
+
switch (format) {
|
|
76
|
+
case "csv":
|
|
77
|
+
return defined({
|
|
78
|
+
type: "csv",
|
|
79
|
+
path: filePath,
|
|
80
|
+
delimiter: options.delimiter ??
|
|
81
|
+
(path_1.default.extname(filePath).toLowerCase() === ".tsv"
|
|
82
|
+
? "\t"
|
|
83
|
+
: undefined),
|
|
84
|
+
bom: options.bom,
|
|
85
|
+
});
|
|
86
|
+
case "json":
|
|
87
|
+
return { type: "json", path: filePath };
|
|
88
|
+
case "xml":
|
|
89
|
+
return defined({
|
|
90
|
+
type: "xml",
|
|
91
|
+
path: filePath,
|
|
92
|
+
root: options.root,
|
|
93
|
+
record: options.record,
|
|
94
|
+
});
|
|
95
|
+
case "excel":
|
|
96
|
+
return defined({
|
|
97
|
+
type: "excel",
|
|
98
|
+
path: filePath,
|
|
99
|
+
sheet: options.sheet,
|
|
100
|
+
});
|
|
101
|
+
case "parquet":
|
|
102
|
+
return { type: "parquet", path: filePath };
|
|
103
|
+
case "sqlite":
|
|
104
|
+
throw new Error("Ainda não dá pra gravar em SQLite. Escolha outro formato pra saída.");
|
|
105
|
+
}
|
|
106
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
export { detectFormat, sourceFromPath, destinationFromPath } from "./formats";
|
|
2
|
+
export type { FileFormat, ReadOptions, WriteOptions } from "./formats";
|
|
3
|
+
export { readRows, writeRows, convert, DEFAULT_PENDING_NAME } from "./rows";
|
|
4
|
+
export type { Input, Output, ReadRowsOptions, WriteRowsOptions, ConvertOptions, } from "./rows";
|
|
5
|
+
export { clean, fillEmpty, combineColumns, validate, dedupe, merge, mergeColumnsFor, defineConfig, defineRules, rulesSchema, } from "./clean";
|
|
6
|
+
export type { Rules, ConfigInput, CleanOptions, CleanResult, DedupeOptions, MergeOptions, } from "./clean";
|
|
7
|
+
export { describeColumns, listNormalizers, normalize } from "./inspect";
|
|
8
|
+
export type { ColumnInfo, ColumnKind } from "./inspect";
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.normalize = exports.listNormalizers = exports.describeColumns = exports.rulesSchema = exports.defineRules = exports.defineConfig = exports.mergeColumnsFor = exports.merge = exports.dedupe = exports.validate = exports.combineColumns = exports.fillEmpty = exports.clean = exports.DEFAULT_PENDING_NAME = exports.convert = exports.writeRows = exports.readRows = exports.destinationFromPath = exports.sourceFromPath = exports.detectFormat = void 0;
|
|
4
|
+
var formats_1 = require("./formats");
|
|
5
|
+
Object.defineProperty(exports, "detectFormat", { enumerable: true, get: function () { return formats_1.detectFormat; } });
|
|
6
|
+
Object.defineProperty(exports, "sourceFromPath", { enumerable: true, get: function () { return formats_1.sourceFromPath; } });
|
|
7
|
+
Object.defineProperty(exports, "destinationFromPath", { enumerable: true, get: function () { return formats_1.destinationFromPath; } });
|
|
8
|
+
var rows_1 = require("./rows");
|
|
9
|
+
Object.defineProperty(exports, "readRows", { enumerable: true, get: function () { return rows_1.readRows; } });
|
|
10
|
+
Object.defineProperty(exports, "writeRows", { enumerable: true, get: function () { return rows_1.writeRows; } });
|
|
11
|
+
Object.defineProperty(exports, "convert", { enumerable: true, get: function () { return rows_1.convert; } });
|
|
12
|
+
Object.defineProperty(exports, "DEFAULT_PENDING_NAME", { enumerable: true, get: function () { return rows_1.DEFAULT_PENDING_NAME; } });
|
|
13
|
+
var clean_1 = require("./clean");
|
|
14
|
+
Object.defineProperty(exports, "clean", { enumerable: true, get: function () { return clean_1.clean; } });
|
|
15
|
+
Object.defineProperty(exports, "fillEmpty", { enumerable: true, get: function () { return clean_1.fillEmpty; } });
|
|
16
|
+
Object.defineProperty(exports, "combineColumns", { enumerable: true, get: function () { return clean_1.combineColumns; } });
|
|
17
|
+
Object.defineProperty(exports, "validate", { enumerable: true, get: function () { return clean_1.validate; } });
|
|
18
|
+
Object.defineProperty(exports, "dedupe", { enumerable: true, get: function () { return clean_1.dedupe; } });
|
|
19
|
+
Object.defineProperty(exports, "merge", { enumerable: true, get: function () { return clean_1.merge; } });
|
|
20
|
+
Object.defineProperty(exports, "mergeColumnsFor", { enumerable: true, get: function () { return clean_1.mergeColumnsFor; } });
|
|
21
|
+
Object.defineProperty(exports, "defineConfig", { enumerable: true, get: function () { return clean_1.defineConfig; } });
|
|
22
|
+
Object.defineProperty(exports, "defineRules", { enumerable: true, get: function () { return clean_1.defineRules; } });
|
|
23
|
+
Object.defineProperty(exports, "rulesSchema", { enumerable: true, get: function () { return clean_1.rulesSchema; } });
|
|
24
|
+
var inspect_1 = require("./inspect");
|
|
25
|
+
Object.defineProperty(exports, "describeColumns", { enumerable: true, get: function () { return inspect_1.describeColumns; } });
|
|
26
|
+
Object.defineProperty(exports, "listNormalizers", { enumerable: true, get: function () { return inspect_1.listNormalizers; } });
|
|
27
|
+
Object.defineProperty(exports, "normalize", { enumerable: true, get: function () { return inspect_1.normalize; } });
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import { TableRow } from "../types";
|
|
2
|
+
export type ColumnKind = "número" | "texto" | "misto" | "vazio";
|
|
3
|
+
export interface ColumnInfo {
|
|
4
|
+
name: string;
|
|
5
|
+
filled: number;
|
|
6
|
+
empty: number;
|
|
7
|
+
distinct: number;
|
|
8
|
+
kind: ColumnKind;
|
|
9
|
+
examples: string[];
|
|
10
|
+
}
|
|
11
|
+
export declare function describeColumns(rows: TableRow[]): ColumnInfo[];
|
|
12
|
+
export declare function listNormalizers(): string[];
|
|
13
|
+
export declare function normalize(name: string, value: string | number): string | number | null;
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.describeColumns = describeColumns;
|
|
4
|
+
exports.listNormalizers = listNormalizers;
|
|
5
|
+
exports.normalize = normalize;
|
|
6
|
+
const header_1 = require("../io/header");
|
|
7
|
+
const normalizers_1 = require("../normalizers");
|
|
8
|
+
const registry_1 = require("../normalizers/registry");
|
|
9
|
+
const EXAMPLES = 3;
|
|
10
|
+
function isEmpty(value) {
|
|
11
|
+
return value === null || value === undefined || value === "";
|
|
12
|
+
}
|
|
13
|
+
function kindOf(values) {
|
|
14
|
+
if (values.length === 0)
|
|
15
|
+
return "vazio";
|
|
16
|
+
const numbers = values.filter((value) => typeof value === "number").length;
|
|
17
|
+
if (numbers === values.length)
|
|
18
|
+
return "número";
|
|
19
|
+
return numbers === 0 ? "texto" : "misto";
|
|
20
|
+
}
|
|
21
|
+
function describeColumns(rows) {
|
|
22
|
+
return (0, header_1.buildHeader)(rows).map((name) => {
|
|
23
|
+
const present = rows
|
|
24
|
+
.map((row) => row[name])
|
|
25
|
+
.filter((value) => !isEmpty(value));
|
|
26
|
+
const distinct = new Set(present.map((value) => String(value)));
|
|
27
|
+
return {
|
|
28
|
+
name,
|
|
29
|
+
filled: present.length,
|
|
30
|
+
empty: rows.length - present.length,
|
|
31
|
+
distinct: distinct.size,
|
|
32
|
+
kind: kindOf(present),
|
|
33
|
+
examples: [...distinct].slice(0, EXAMPLES),
|
|
34
|
+
};
|
|
35
|
+
});
|
|
36
|
+
}
|
|
37
|
+
function listNormalizers() {
|
|
38
|
+
(0, normalizers_1.registerBuiltinKeyNormalizers)();
|
|
39
|
+
return (0, registry_1.listKeyNormalizers)();
|
|
40
|
+
}
|
|
41
|
+
function normalize(name, value) {
|
|
42
|
+
(0, normalizers_1.registerBuiltinKeyNormalizers)();
|
|
43
|
+
return (0, registry_1.getKeyNormalizer)(name)(value)?.key ?? null;
|
|
44
|
+
}
|