datera 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (159) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +132 -0
  3. package/dist/api/clean.d.ts +138 -0
  4. package/dist/api/clean.js +138 -0
  5. package/dist/api/formats.d.ts +19 -0
  6. package/dist/api/formats.js +106 -0
  7. package/dist/api/index.d.ts +8 -0
  8. package/dist/api/index.js +27 -0
  9. package/dist/api/inspect.d.ts +13 -0
  10. package/dist/api/inspect.js +44 -0
  11. package/dist/api/parse.d.ts +2 -0
  12. package/dist/api/parse.js +19 -0
  13. package/dist/api/rows.d.ts +23 -0
  14. package/dist/api/rows.js +63 -0
  15. package/dist/cli.d.ts +62 -0
  16. package/dist/cli.js +225 -0
  17. package/dist/commands.d.ts +3 -0
  18. package/dist/commands.js +169 -0
  19. package/dist/config/check.d.ts +11 -0
  20. package/dist/config/check.js +99 -0
  21. package/dist/config/index.d.ts +7 -0
  22. package/dist/config/index.js +16 -0
  23. package/dist/config/ioSchema.d.ts +97 -0
  24. package/dist/config/ioSchema.js +111 -0
  25. package/dist/config/load.d.ts +5 -0
  26. package/dist/config/load.js +68 -0
  27. package/dist/config/mergeSchema.d.ts +36 -0
  28. package/dist/config/mergeSchema.js +46 -0
  29. package/dist/config/messages.d.ts +7 -0
  30. package/dist/config/messages.js +33 -0
  31. package/dist/config/prepareSchema.d.ts +12 -0
  32. package/dist/config/prepareSchema.js +15 -0
  33. package/dist/config/schema.d.ts +195 -0
  34. package/dist/config/schema.js +44 -0
  35. package/dist/config/validationSchema.d.ts +27 -0
  36. package/dist/config/validationSchema.js +50 -0
  37. package/dist/env.d.ts +2 -0
  38. package/dist/env.js +15 -0
  39. package/dist/filters/combine.d.ts +10 -0
  40. package/dist/filters/combine.js +60 -0
  41. package/dist/filters/combineTypes.d.ts +6 -0
  42. package/dist/filters/combineTypes.js +2 -0
  43. package/dist/filters/dedupe.d.ts +12 -0
  44. package/dist/filters/dedupe.js +45 -0
  45. package/dist/filters/fillEmpty.d.ts +9 -0
  46. package/dist/filters/fillEmpty.js +32 -0
  47. package/dist/filters/fillEmptyTypes.d.ts +5 -0
  48. package/dist/filters/fillEmptyTypes.js +2 -0
  49. package/dist/filters/merge.d.ts +15 -0
  50. package/dist/filters/merge.js +113 -0
  51. package/dist/filters/mergeStrategies.d.ts +10 -0
  52. package/dist/filters/mergeStrategies.js +69 -0
  53. package/dist/filters/mergeTypes.d.ts +32 -0
  54. package/dist/filters/mergeTypes.js +2 -0
  55. package/dist/filters/mergeUnkeyed.d.ts +3 -0
  56. package/dist/filters/mergeUnkeyed.js +21 -0
  57. package/dist/filters/types.d.ts +3 -0
  58. package/dist/filters/types.js +2 -0
  59. package/dist/filters/validate.d.ts +16 -0
  60. package/dist/filters/validate.js +79 -0
  61. package/dist/filters/validateTypes.d.ts +28 -0
  62. package/dist/filters/validateTypes.js +2 -0
  63. package/dist/index.d.ts +13 -0
  64. package/dist/index.js +40 -0
  65. package/dist/io/csv/csvFormat.d.ts +4 -0
  66. package/dist/io/csv/csvFormat.js +94 -0
  67. package/dist/io/csv/csvSink.d.ts +14 -0
  68. package/dist/io/csv/csvSink.js +32 -0
  69. package/dist/io/csv/csvSource.d.ts +13 -0
  70. package/dist/io/csv/csvSource.js +29 -0
  71. package/dist/io/custom/loader.d.ts +1 -0
  72. package/dist/io/custom/loader.js +36 -0
  73. package/dist/io/custom/registry.d.ts +13 -0
  74. package/dist/io/custom/registry.js +50 -0
  75. package/dist/io/excel/excelCell.d.ts +4 -0
  76. package/dist/io/excel/excelCell.js +29 -0
  77. package/dist/io/excel/excelSink.d.ts +14 -0
  78. package/dist/io/excel/excelSink.js +57 -0
  79. package/dist/io/excel/excelSource.d.ts +11 -0
  80. package/dist/io/excel/excelSource.js +47 -0
  81. package/dist/io/excel/workbook.d.ts +4 -0
  82. package/dist/io/excel/workbook.js +20 -0
  83. package/dist/io/factory.d.ts +5 -0
  84. package/dist/io/factory.js +106 -0
  85. package/dist/io/files.d.ts +7 -0
  86. package/dist/io/files.js +48 -0
  87. package/dist/io/header.d.ts +2 -0
  88. package/dist/io/header.js +42 -0
  89. package/dist/io/json/jsonSink.d.ts +12 -0
  90. package/dist/io/json/jsonSink.js +19 -0
  91. package/dist/io/json/jsonSource.d.ts +11 -0
  92. package/dist/io/json/jsonSource.js +66 -0
  93. package/dist/io/mysql/client.d.ts +13 -0
  94. package/dist/io/mysql/client.js +73 -0
  95. package/dist/io/mysql/mysqlSource.d.ts +13 -0
  96. package/dist/io/mysql/mysqlSource.js +25 -0
  97. package/dist/io/optional.d.ts +2 -0
  98. package/dist/io/optional.js +36 -0
  99. package/dist/io/parquet/parquetLibrary.d.ts +11 -0
  100. package/dist/io/parquet/parquetLibrary.js +20 -0
  101. package/dist/io/parquet/parquetSink.d.ts +15 -0
  102. package/dist/io/parquet/parquetSink.js +58 -0
  103. package/dist/io/parquet/parquetSource.d.ts +13 -0
  104. package/dist/io/parquet/parquetSource.js +68 -0
  105. package/dist/io/postgres/postgresSource.d.ts +26 -0
  106. package/dist/io/postgres/postgresSource.js +42 -0
  107. package/dist/io/safety.d.ts +4 -0
  108. package/dist/io/safety.js +83 -0
  109. package/dist/io/sheets/client.d.ts +10 -0
  110. package/dist/io/sheets/client.js +141 -0
  111. package/dist/io/sheets/sheetsSink.d.ts +12 -0
  112. package/dist/io/sheets/sheetsSink.js +24 -0
  113. package/dist/io/sheets/sheetsSource.d.ts +10 -0
  114. package/dist/io/sheets/sheetsSource.js +47 -0
  115. package/dist/io/sql/cell.d.ts +5 -0
  116. package/dist/io/sql/cell.js +38 -0
  117. package/dist/io/sql/driver.d.ts +1 -0
  118. package/dist/io/sql/driver.js +15 -0
  119. package/dist/io/sql/names.d.ts +1 -0
  120. package/dist/io/sql/names.js +12 -0
  121. package/dist/io/sqlite/sqliteSource.d.ts +11 -0
  122. package/dist/io/sqlite/sqliteSource.js +52 -0
  123. package/dist/io/sqlserver/sqlServerSource.d.ts +24 -0
  124. package/dist/io/sqlserver/sqlServerSource.js +49 -0
  125. package/dist/io/types.d.ts +11 -0
  126. package/dist/io/types.js +2 -0
  127. package/dist/io/xml/xmlParse.d.ts +9 -0
  128. package/dist/io/xml/xmlParse.js +178 -0
  129. package/dist/io/xml/xmlSink.d.ts +18 -0
  130. package/dist/io/xml/xmlSink.js +79 -0
  131. package/dist/io/xml/xmlSource.d.ts +11 -0
  132. package/dist/io/xml/xmlSource.js +93 -0
  133. package/dist/logger.d.ts +11 -0
  134. package/dist/logger.js +29 -0
  135. package/dist/main.d.ts +2 -0
  136. package/dist/main.js +18 -0
  137. package/dist/normalizers/builtin.d.ts +5 -0
  138. package/dist/normalizers/builtin.js +23 -0
  139. package/dist/normalizers/index.d.ts +2 -0
  140. package/dist/normalizers/index.js +14 -0
  141. package/dist/normalizers/loader.d.ts +1 -0
  142. package/dist/normalizers/loader.js +30 -0
  143. package/dist/normalizers/registry.d.ts +9 -0
  144. package/dist/normalizers/registry.js +29 -0
  145. package/dist/pipeline/index.d.ts +7 -0
  146. package/dist/pipeline/index.js +17 -0
  147. package/dist/pipeline/modes.d.ts +6 -0
  148. package/dist/pipeline/modes.js +48 -0
  149. package/dist/pipeline/report.d.ts +3 -0
  150. package/dist/pipeline/report.js +32 -0
  151. package/dist/pipeline/runEtl.d.ts +24 -0
  152. package/dist/pipeline/runEtl.js +98 -0
  153. package/dist/pipeline/steps.d.ts +4 -0
  154. package/dist/pipeline/steps.js +36 -0
  155. package/dist/pipeline/types.d.ts +34 -0
  156. package/dist/pipeline/types.js +2 -0
  157. package/dist/types.d.ts +3 -0
  158. package/dist/types.js +2 -0
  159. package/package.json +128 -0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Alessandra Vieira
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,132 @@
1
+ <p align="center">
2
+ <img src="docs/assets/datera-banner.png" alt="Datera: dados bagunçados entrando de um lado e saindo organizados do outro">
3
+ </p>
4
+
5
+ <p align="center">Ferramenta pra limpar e juntar dados bagunçados de planilha, banco e arquivo.</p>
6
+
7
+ <p align="center">
8
+ <a href="https://github.com/alessandravieiradev-blip/datera/actions/workflows/ci.yml"><img src="https://github.com/alessandravieiradev-blip/datera/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
9
+ </p>
10
+
11
+ <p align="center">
12
+ <a href="docs/gestores.md"><img src="https://img.shields.io/badge/Para%20gestores-2563EB?style=for-the-badge" alt="Guia para gestores"></a>
13
+ <a href="docs/devs.md"><img src="https://img.shields.io/badge/Para%20devs-16181D?style=for-the-badge" alt="Guia para devs"></a>
14
+ </p>
15
+
16
+ ## O que é
17
+
18
+ O Datera é um ETL que eu fiz em TypeScript. Ele lê dados de banco, planilha ou arquivo, arruma o que dá e escreve o resultado organizado em outro lugar:
19
+
20
+ ```mermaid
21
+ flowchart LR
22
+ subgraph fontes["Lê de"]
23
+ direction TB
24
+ B["Bancos<br/>MySQL · PostgreSQL<br/>SQL Server · SQLite"]
25
+ P1["Planilhas<br/>Google Sheets · Excel"]
26
+ A1["Arquivos<br/>CSV · JSON<br/>XML · Parquet"]
27
+ end
28
+ subgraph destinos["Escreve em"]
29
+ direction TB
30
+ P2["Planilhas<br/>Google Sheets · Excel"]
31
+ A2["Arquivos<br/>CSV · JSON<br/>XML · Parquet"]
32
+ end
33
+ B --> D((Datera))
34
+ P1 --> D
35
+ A1 --> D
36
+ D --> P2
37
+ D --> A2
38
+ classDef datera fill:#2563EB,stroke:#2563EB,color:#ffffff
39
+ class D datera
40
+ ```
41
+
42
+ Ele começou bem simples, era só pra copiar uma tabela do MySQL pra uma planilha. Só que aí eu fui vendo que dado de verdade vem uma bagunça: a mesma pessoa cadastrada duas vezes, e-mail com maiúscula num lugar e minúscula no outro, campo vazio, informação espalhada em várias colunas. Então fui colocando coisa nova até conseguir resolver quase tudo mexendo só no `config.json`.
43
+
44
+ ## A ideia
45
+
46
+ Quem mais sofre com planilha bagunçada normalmente não é quem programa. É quem precisa da lista certinha pra mandar um e-mail, fechar um relatório ou ligar pra alguém. E essa pessoa não deveria ter que pedir ajuda toda vez que precisa dos dados limpos.
47
+
48
+ Então o Datera funciona assim: as regras ficam num arquivo de configuração (o `config.json`), que pode ser montado com calma uma vez só. Depois disso qualquer pessoa roda com um clique e recebe duas coisas: o resultado organizado e uma lista separada do que precisa de alguém dar uma olhada, com o motivo escrito do lado. Nada some sem explicação.
49
+
50
+ Por isso ele tem três jeitos de usar, todos com o mesmo motor e a mesma configuração:
51
+
52
+ | Jeito | Pra quem | O que tem |
53
+ | -------------- | --------------------------------------------- | ------------------------------------------------------------------------- |
54
+ | **Datera** | quem cuida dos dados mas não programa | app com tela, passo a passo, regras em frases, gráficos e histórico |
55
+ | **Datera Dev** | quem programa | app escuro com editor da config, prévia na hora, log e paleta de comandos |
56
+ | **Terminal** | quem quer automatizar ou usar em outro código | o comando `datera`, `--dry-run` e a função `runEtl` |
57
+
58
+ ## Como funciona
59
+
60
+ ```text
61
+ fonte ──► prepara ──► separa o que tem problema ──► tira ou junta repetidos ──► destino
62
+ │
63
+ └──► pendências (com o motivo de cada uma)
64
+ ```
65
+
66
+ 1. **Fonte:** lê de onde os dados estão.
67
+ 2. **Prepara:** preenche célula vazia e junta colunas, se você pedir.
68
+ 3. **Pendências:** as linhas que quebram alguma regra (e-mail inválido, campo vazio, valor fora da lista...) vão pra uma aba ou arquivo à parte.
69
+ 4. **Repetidos:** deixa como está, tira as linhas repetidas ou junta as linhas da mesma pessoa sem perder nada.
70
+ 5. **Destino:** escreve o resultado onde você escolheu.
71
+
72
+ Hoje ele consegue:
73
+
74
+ - tirar linhas repetidas (`dedupe`) ou juntar as linhas da mesma pessoa sem perder nada (`merge`)
75
+ - usar normalizadores pra `Lia@Email.com` e ` lia@email.com` contarem como a mesma pessoa
76
+ - ter regras diferentes dependendo do tipo da chave
77
+ - decidir o que fazer com as linhas que não têm chave, sem elas sumirem nem se misturarem
78
+ - espalhar valores em várias colunas sem jogar nenhum fora
79
+ - ler e escrever em formatos diferentes sem mudar nada das regras
80
+
81
+ ## Escolha o seu caminho
82
+
83
+ <table>
84
+ <tr>
85
+ <td width="50%" valign="top">
86
+
87
+ ### Para gestores
88
+
89
+ Você quer usar o app com tela: instalar, escolher de onde vêm os dados, montar as regras em frases e exportar. Não precisa saber programar.
90
+
91
+ <a href="https://github.com/alessandravieiradev-blip/datera/releases/latest"><img src="https://img.shields.io/badge/Baixar%20o%20Datera-2563EB?style=for-the-badge" alt="Baixar o Datera"></a>
92
+ <a href="docs/gestores.md"><img src="https://img.shields.io/badge/Abrir%20o%20guia-475569?style=for-the-badge" alt="Abrir o guia para gestores"></a>
93
+
94
+ </td>
95
+ <td width="50%" valign="top">
96
+
97
+ ### Para devs
98
+
99
+ Você quer rodar pelo terminal, usar o Datera Dev, escrever a config na mão, criar normalizadores ou chamar o Datera de dentro de outro código.
100
+
101
+ <a href="docs/devs.md"><img src="https://img.shields.io/badge/Abrir%20o%20guia%20para%20devs-16181D?style=for-the-badge" alt="Abrir o guia para devs"></a>
102
+ <a href="docs/api.md"><img src="https://img.shields.io/badge/Comandos%20e%20API-475569?style=for-the-badge" alt="Comandos e API"></a>
103
+
104
+ </td>
105
+ </tr>
106
+ </table>
107
+
108
+ ## Todos os guias
109
+
110
+ | Guia | O que tem |
111
+ | ---------------------------------------------------------- | -------------------------------------------------------------------- |
112
+ | [Para gestores](docs/gestores.md) | instalar e usar o app com tela |
113
+ | [Para devs](docs/devs.md) | teste em 1 minuto, terminal, Datera Dev, uso em código e testes |
114
+ | [Comandos e API](docs/api.md) | os comandos do terminal e as funções pra usar no seu código |
115
+ | [Configuração](docs/configuracao.md) | todos os campos do `config.json`, os modos e as estratégias do merge |
116
+ | [Fontes e destinos](docs/fontes-e-destinos.md) | cada formato que ele lê e escreve, e como criar o seu |
117
+ | [Exemplos](docs/exemplos.md) | dez configs, do mais simples ao mais completo |
118
+ | [Preparação e pendências](docs/preparacao-e-pendencias.md) | `fillEmpty`, `combineColumns`, `distribute` e `validation` |
119
+ | [Normalizadores](docs/normalizadores.md) | os prontos e como criar o seu |
120
+ | [Estrutura do projeto](docs/estrutura.md) | o que tem em cada pasta |
121
+
122
+ ## Contribuir e licença
123
+
124
+ Se quiser ajudar, o jeito de rodar, testar e mandar mudanças está no [CONTRIBUTING.md](CONTRIBUTING.md). O código é aberto, sob a [licença MIT](LICENSE).
125
+
126
+ ## Próximos passos
127
+
128
+ A ideia é chegar num instalador de um clique pra quem não programa e numa biblioteca no npm pra quem programa.
129
+
130
+ O que eu quero fazer fica nas [issues](https://github.com/alessandravieiradev-blip/datera/issues), agrupadas em etapas nos [milestones](https://github.com/alessandravieiradev-blip/datera/milestones). Lá dá pra ver o que já foi feito e o que falta, e tudo se atualiza sozinho conforme eu vou fazendo.
131
+
132
+ Se quiser saber no que eu tô trabalhando agora, me chama no Instagram ou aqui no GitHub. Os dois estão no [meu perfil](https://github.com/alessandravieiradev-blip).
@@ -0,0 +1,138 @@
1
+ import { z } from "zod";
2
+ import { etlConfigSchema } from "../config";
3
+ import { mergeColumnSchema } from "../config/mergeSchema";
4
+ import { combineColumnsSchema, fillEmptySchema } from "../config/prepareSchema";
5
+ import { validationSchema } from "../config/validationSchema";
6
+ import { PendingReason, StepReport } from "../pipeline";
7
+ import { TableRow } from "../types";
8
+ export declare const rulesSchema: z.ZodObject<{
9
+ fillEmpty: z.ZodOptional<z.ZodArray<z.ZodObject<{
10
+ column: z.ZodString;
11
+ fallbackColumns: z.ZodOptional<z.ZodArray<z.ZodString>>;
12
+ default: z.ZodOptional<z.ZodString>;
13
+ }, z.core.$strip>>>;
14
+ combineColumns: z.ZodOptional<z.ZodArray<z.ZodObject<{
15
+ into: z.ZodString;
16
+ columns: z.ZodArray<z.ZodString>;
17
+ separator: z.ZodOptional<z.ZodString>;
18
+ keepSources: z.ZodOptional<z.ZodBoolean>;
19
+ }, z.core.$strip>>>;
20
+ validation: z.ZodOptional<z.ZodObject<{
21
+ rules: z.ZodArray<z.ZodDiscriminatedUnion<[z.ZodObject<{
22
+ message: z.ZodOptional<z.ZodString>;
23
+ column: z.ZodString;
24
+ rule: z.ZodLiteral<"required">;
25
+ }, z.core.$strip>, z.ZodObject<{
26
+ message: z.ZodOptional<z.ZodString>;
27
+ column: z.ZodString;
28
+ rule: z.ZodLiteral<"pattern">;
29
+ pattern: z.ZodString;
30
+ flags: z.ZodOptional<z.ZodString>;
31
+ }, z.core.$strip>, z.ZodObject<{
32
+ message: z.ZodOptional<z.ZodString>;
33
+ column: z.ZodString;
34
+ rule: z.ZodLiteral<"oneOf">;
35
+ values: z.ZodArray<z.ZodString>;
36
+ ignoreCase: z.ZodOptional<z.ZodBoolean>;
37
+ }, z.core.$strip>, z.ZodObject<{
38
+ message: z.ZodOptional<z.ZodString>;
39
+ column: z.ZodString;
40
+ rule: z.ZodLiteral<"normalizer">;
41
+ normalizer: z.ZodString;
42
+ }, z.core.$strip>], "rule">>;
43
+ pendingSheet: z.ZodOptional<z.ZodString>;
44
+ reasonColumn: z.ZodOptional<z.ZodString>;
45
+ }, z.core.$strip>>;
46
+ dedupe: z.ZodOptional<z.ZodObject<{
47
+ column: z.ZodString;
48
+ keep: z.ZodOptional<z.ZodEnum<{
49
+ first: "first";
50
+ last: "last";
51
+ }>>;
52
+ normalizer: z.ZodOptional<z.ZodString>;
53
+ }, z.core.$strip>>;
54
+ merge: z.ZodOptional<z.ZodObject<{
55
+ key: z.ZodString;
56
+ columns: z.ZodArray<z.ZodObject<{
57
+ column: z.ZodString;
58
+ strategy: z.ZodEnum<{
59
+ concat: "concat";
60
+ "extra-column": "extra-column";
61
+ overwrite: "overwrite";
62
+ }>;
63
+ separator: z.ZodOptional<z.ZodString>;
64
+ distribute: z.ZodOptional<z.ZodObject<{
65
+ columns: z.ZodArray<z.ZodString>;
66
+ overflowInto: z.ZodOptional<z.ZodString>;
67
+ sources: z.ZodOptional<z.ZodArray<z.ZodString>>;
68
+ }, z.core.$strip>>;
69
+ byGroup: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodObject<{
70
+ strategy: z.ZodEnum<{
71
+ concat: "concat";
72
+ "extra-column": "extra-column";
73
+ overwrite: "overwrite";
74
+ }>;
75
+ separator: z.ZodOptional<z.ZodString>;
76
+ into: z.ZodOptional<z.ZodString>;
77
+ distribute: z.ZodOptional<z.ZodObject<{
78
+ columns: z.ZodArray<z.ZodString>;
79
+ overflowInto: z.ZodOptional<z.ZodString>;
80
+ sources: z.ZodOptional<z.ZodArray<z.ZodString>>;
81
+ }, z.core.$strip>>;
82
+ }, z.core.$strip>>>;
83
+ unkeyed: z.ZodOptional<z.ZodObject<{
84
+ strategy: z.ZodEnum<{
85
+ "collapse-column": "collapse-column";
86
+ }>;
87
+ into: z.ZodOptional<z.ZodString>;
88
+ separator: z.ZodOptional<z.ZodString>;
89
+ }, z.core.$strip>>;
90
+ }, z.core.$strip>>;
91
+ normalizer: z.ZodOptional<z.ZodString>;
92
+ emptyKeyLabel: z.ZodOptional<z.ZodString>;
93
+ rejectedKeyLabel: z.ZodOptional<z.ZodString>;
94
+ }, z.core.$strip>>;
95
+ }, z.core.$strip>;
96
+ export type Rules = z.input<typeof rulesSchema>;
97
+ export type ConfigInput = z.input<typeof etlConfigSchema>;
98
+ export interface CleanOptions {
99
+ onStep?: ((report: StepReport) => void) | undefined;
100
+ }
101
+ export interface CleanResult {
102
+ rows: TableRow[];
103
+ pending: TableRow[];
104
+ pendingByReason: PendingReason[];
105
+ steps: StepReport[];
106
+ durationMs: number;
107
+ }
108
+ export declare function defineConfig(config: ConfigInput): ConfigInput;
109
+ export declare function defineRules(rules: Rules): Rules;
110
+ export declare function clean(rows: TableRow[], rules?: Rules, options?: CleanOptions): CleanResult;
111
+ type FillEmptyRule = z.input<typeof fillEmptySchema>;
112
+ type CombineRule = z.input<typeof combineColumnsSchema>;
113
+ type Validation = z.input<typeof validationSchema>;
114
+ type ValidationRule = Validation["rules"][number];
115
+ type MergeColumn = z.input<typeof mergeColumnSchema>;
116
+ export declare function fillEmpty(rows: TableRow[], rules: FillEmptyRule[]): TableRow[];
117
+ export declare function combineColumns(rows: TableRow[], rules: CombineRule[]): TableRow[];
118
+ export declare function validate(rows: TableRow[], rules: ValidationRule[] | Validation): {
119
+ valid: TableRow[];
120
+ pending: TableRow[];
121
+ };
122
+ export interface DedupeOptions {
123
+ keep?: "first" | "last" | undefined;
124
+ normalizer?: string | undefined;
125
+ }
126
+ export declare function dedupe(rows: TableRow[], column: string, options?: DedupeOptions): TableRow[];
127
+ export interface MergeOptions {
128
+ columns?: MergeColumn[] | undefined;
129
+ normalizer?: string | undefined;
130
+ separator?: string | undefined;
131
+ overwrite?: string[] | undefined;
132
+ extraColumn?: string[] | undefined;
133
+ emptyKeyLabel?: string | undefined;
134
+ rejectedKeyLabel?: string | undefined;
135
+ }
136
+ export declare function mergeColumnsFor(rows: TableRow[], key: string, options?: MergeOptions): MergeColumn[];
137
+ export declare function merge(rows: TableRow[], key: string, options?: MergeOptions): TableRow[];
138
+ export {};
@@ -0,0 +1,138 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.rulesSchema = void 0;
4
+ exports.defineConfig = defineConfig;
5
+ exports.defineRules = defineRules;
6
+ exports.clean = clean;
7
+ exports.fillEmpty = fillEmpty;
8
+ exports.combineColumns = combineColumns;
9
+ exports.validate = validate;
10
+ exports.dedupe = dedupe;
11
+ exports.mergeColumnsFor = mergeColumnsFor;
12
+ exports.merge = merge;
13
+ const zod_1 = require("zod");
14
+ const mergeSchema_1 = require("../config/mergeSchema");
15
+ const prepareSchema_1 = require("../config/prepareSchema");
16
+ const validationSchema_1 = require("../config/validationSchema");
17
+ const normalizers_1 = require("../normalizers");
18
+ const pipeline_1 = require("../pipeline");
19
+ const parse_1 = require("./parse");
20
+ exports.rulesSchema = zod_1.z
21
+ .object({
22
+ fillEmpty: zod_1.z.array(prepareSchema_1.fillEmptySchema).optional(),
23
+ combineColumns: zod_1.z.array(prepareSchema_1.combineColumnsSchema).optional(),
24
+ validation: validationSchema_1.validationSchema.optional(),
25
+ dedupe: zod_1.z
26
+ .object({
27
+ column: zod_1.z.string().min(1),
28
+ keep: zod_1.z.enum(["first", "last"]).optional(),
29
+ normalizer: zod_1.z.string().optional(),
30
+ })
31
+ .optional(),
32
+ merge: zod_1.z
33
+ .object({
34
+ key: zod_1.z.string().min(1),
35
+ columns: zod_1.z.array(mergeSchema_1.mergeColumnSchema).min(1),
36
+ normalizer: zod_1.z.string().optional(),
37
+ emptyKeyLabel: zod_1.z.string().optional(),
38
+ rejectedKeyLabel: zod_1.z.string().optional(),
39
+ })
40
+ .optional(),
41
+ })
42
+ .refine((rules) => !(rules.dedupe && rules.merge), {
43
+ message: "Escolha dedupe ou merge, os dois juntos não dá.",
44
+ path: ["merge"],
45
+ });
46
+ function defineConfig(config) {
47
+ return config;
48
+ }
49
+ function defineRules(rules) {
50
+ return rules;
51
+ }
52
+ function configOf(rules) {
53
+ const mode = rules.merge ? "merge" : rules.dedupe ? "dedupe" : "raw";
54
+ const config = {
55
+ mode,
56
+ fillEmpty: rules.fillEmpty,
57
+ combineColumns: rules.combineColumns,
58
+ validation: rules.validation,
59
+ dedupeColumn: rules.dedupe?.column,
60
+ dedupeStrategy: rules.dedupe?.keep === "last" ? "keep-last" : "keep-first",
61
+ dedupeKeyNormalizer: rules.dedupe?.normalizer,
62
+ mergeKeyColumn: rules.merge?.key,
63
+ mergeColumns: rules.merge?.columns,
64
+ mergeKeyNormalizer: rules.merge?.normalizer,
65
+ mergeEmptyKeyLabel: rules.merge?.emptyKeyLabel,
66
+ mergeRejectedKeyLabel: rules.merge?.rejectedKeyLabel,
67
+ };
68
+ return { config, mode };
69
+ }
70
+ function clean(rows, rules = {}, options = {}) {
71
+ const startedAt = Date.now();
72
+ const parsed = (0, parse_1.parseWith)(exports.rulesSchema, rules, "As regras");
73
+ (0, normalizers_1.registerBuiltinKeyNormalizers)();
74
+ const { config, mode } = configOf(parsed);
75
+ const result = (0, pipeline_1.applySteps)((0, pipeline_1.buildSteps)(config, mode), rows, options.onStep);
76
+ return {
77
+ rows: result.rows,
78
+ pending: result.pending,
79
+ pendingByReason: (0, pipeline_1.countPendingReasons)(result.pending, parsed.validation?.reasonColumn),
80
+ steps: result.steps,
81
+ durationMs: Date.now() - startedAt,
82
+ };
83
+ }
84
+ function fillEmpty(rows, rules) {
85
+ return clean(rows, { fillEmpty: rules }).rows;
86
+ }
87
+ function combineColumns(rows, rules) {
88
+ return clean(rows, { combineColumns: rules }).rows;
89
+ }
90
+ function validate(rows, rules) {
91
+ const validation = Array.isArray(rules) ? { rules } : rules;
92
+ const result = clean(rows, { validation });
93
+ return { valid: result.rows, pending: result.pending };
94
+ }
95
+ function dedupe(rows, column, options = {}) {
96
+ return clean(rows, {
97
+ dedupe: {
98
+ column,
99
+ ...(options.keep ? { keep: options.keep } : {}),
100
+ ...(options.normalizer ? { normalizer: options.normalizer } : {}),
101
+ },
102
+ }).rows;
103
+ }
104
+ function mergeColumnsFor(rows, key, options = {}) {
105
+ const names = new Set();
106
+ for (const row of rows) {
107
+ for (const name of Object.keys(row))
108
+ names.add(name);
109
+ }
110
+ names.delete(key);
111
+ return [...names].map((column) => options.overwrite?.includes(column)
112
+ ? { column, strategy: "overwrite" }
113
+ : options.extraColumn?.includes(column)
114
+ ? { column, strategy: "extra-column" }
115
+ : {
116
+ column,
117
+ strategy: "concat",
118
+ separator: options.separator ?? " | ",
119
+ });
120
+ }
121
+ function merge(rows, key, options = {}) {
122
+ const columns = options.columns ?? mergeColumnsFor(rows, key, options);
123
+ if (columns.length === 0)
124
+ return rows;
125
+ return clean(rows, {
126
+ merge: {
127
+ key,
128
+ columns,
129
+ ...(options.normalizer ? { normalizer: options.normalizer } : {}),
130
+ ...(options.emptyKeyLabel
131
+ ? { emptyKeyLabel: options.emptyKeyLabel }
132
+ : {}),
133
+ ...(options.rejectedKeyLabel
134
+ ? { rejectedKeyLabel: options.rejectedKeyLabel }
135
+ : {}),
136
+ },
137
+ }).rows;
138
+ }
@@ -0,0 +1,19 @@
1
+ import { DestinationConfig, SourceConfig } from "../config/ioSchema";
2
+ export type FileFormat = "csv" | "json" | "xml" | "excel" | "parquet" | "sqlite";
3
+ export interface ReadOptions {
4
+ table?: string | undefined;
5
+ sheet?: string | undefined;
6
+ recordsPath?: string | undefined;
7
+ delimiter?: string | undefined;
8
+ encoding?: "utf-8" | "latin1" | undefined;
9
+ }
10
+ export interface WriteOptions {
11
+ sheet?: string | undefined;
12
+ delimiter?: string | undefined;
13
+ bom?: boolean | undefined;
14
+ root?: string | undefined;
15
+ record?: string | undefined;
16
+ }
17
+ export declare function detectFormat(filePath: string): FileFormat;
18
+ export declare function sourceFromPath(filePath: string, options?: ReadOptions): SourceConfig;
19
+ export declare function destinationFromPath(filePath: string, options?: WriteOptions): DestinationConfig;
@@ -0,0 +1,106 @@
1
+ "use strict";
2
+ var __importDefault = (this && this.__importDefault) || function (mod) {
3
+ return (mod && mod.__esModule) ? mod : { "default": mod };
4
+ };
5
+ Object.defineProperty(exports, "__esModule", { value: true });
6
+ exports.detectFormat = detectFormat;
7
+ exports.sourceFromPath = sourceFromPath;
8
+ exports.destinationFromPath = destinationFromPath;
9
+ const path_1 = __importDefault(require("path"));
10
+ const EXTENSIONS = {
11
+ ".csv": "csv",
12
+ ".tsv": "csv",
13
+ ".txt": "csv",
14
+ ".json": "json",
15
+ ".xml": "xml",
16
+ ".xlsx": "excel",
17
+ ".parquet": "parquet",
18
+ ".db": "sqlite",
19
+ ".sqlite": "sqlite",
20
+ ".sqlite3": "sqlite",
21
+ };
22
+ function defined(value) {
23
+ return Object.fromEntries(Object.entries(value).filter(([, inner]) => inner !== undefined));
24
+ }
25
+ function detectFormat(filePath) {
26
+ const extension = path_1.default.extname(filePath).toLowerCase();
27
+ const format = EXTENSIONS[extension];
28
+ if (!format) {
29
+ throw new Error(`Não sei qual é o formato de "${filePath}" pela extensão. As que eu conheço: ${Object.keys(EXTENSIONS).join(", ")}.`);
30
+ }
31
+ return format;
32
+ }
33
+ function sourceFromPath(filePath, options = {}) {
34
+ const format = detectFormat(filePath);
35
+ switch (format) {
36
+ case "csv":
37
+ return defined({
38
+ type: "csv",
39
+ path: filePath,
40
+ delimiter: options.delimiter ??
41
+ (path_1.default.extname(filePath).toLowerCase() === ".tsv"
42
+ ? "\t"
43
+ : undefined),
44
+ encoding: options.encoding,
45
+ });
46
+ case "json":
47
+ return defined({
48
+ type: "json",
49
+ path: filePath,
50
+ recordsPath: options.recordsPath,
51
+ });
52
+ case "xml":
53
+ return defined({
54
+ type: "xml",
55
+ path: filePath,
56
+ recordsPath: options.recordsPath,
57
+ });
58
+ case "excel":
59
+ return defined({
60
+ type: "excel",
61
+ path: filePath,
62
+ sheet: options.sheet,
63
+ });
64
+ case "parquet":
65
+ return { type: "parquet", path: filePath };
66
+ case "sqlite":
67
+ if (!options.table) {
68
+ throw new Error(`Pra ler o SQLite "${filePath}", diga qual tabela (table).`);
69
+ }
70
+ return { type: "sqlite", path: filePath, table: options.table };
71
+ }
72
+ }
73
+ function destinationFromPath(filePath, options = {}) {
74
+ const format = detectFormat(filePath);
75
+ switch (format) {
76
+ case "csv":
77
+ return defined({
78
+ type: "csv",
79
+ path: filePath,
80
+ delimiter: options.delimiter ??
81
+ (path_1.default.extname(filePath).toLowerCase() === ".tsv"
82
+ ? "\t"
83
+ : undefined),
84
+ bom: options.bom,
85
+ });
86
+ case "json":
87
+ return { type: "json", path: filePath };
88
+ case "xml":
89
+ return defined({
90
+ type: "xml",
91
+ path: filePath,
92
+ root: options.root,
93
+ record: options.record,
94
+ });
95
+ case "excel":
96
+ return defined({
97
+ type: "excel",
98
+ path: filePath,
99
+ sheet: options.sheet,
100
+ });
101
+ case "parquet":
102
+ return { type: "parquet", path: filePath };
103
+ case "sqlite":
104
+ throw new Error("Ainda não dá pra gravar em SQLite. Escolha outro formato pra saída.");
105
+ }
106
+ }
@@ -0,0 +1,8 @@
1
+ export { detectFormat, sourceFromPath, destinationFromPath } from "./formats";
2
+ export type { FileFormat, ReadOptions, WriteOptions } from "./formats";
3
+ export { readRows, writeRows, convert, DEFAULT_PENDING_NAME } from "./rows";
4
+ export type { Input, Output, ReadRowsOptions, WriteRowsOptions, ConvertOptions, } from "./rows";
5
+ export { clean, fillEmpty, combineColumns, validate, dedupe, merge, mergeColumnsFor, defineConfig, defineRules, rulesSchema, } from "./clean";
6
+ export type { Rules, ConfigInput, CleanOptions, CleanResult, DedupeOptions, MergeOptions, } from "./clean";
7
+ export { describeColumns, listNormalizers, normalize } from "./inspect";
8
+ export type { ColumnInfo, ColumnKind } from "./inspect";
@@ -0,0 +1,27 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.normalize = exports.listNormalizers = exports.describeColumns = exports.rulesSchema = exports.defineRules = exports.defineConfig = exports.mergeColumnsFor = exports.merge = exports.dedupe = exports.validate = exports.combineColumns = exports.fillEmpty = exports.clean = exports.DEFAULT_PENDING_NAME = exports.convert = exports.writeRows = exports.readRows = exports.destinationFromPath = exports.sourceFromPath = exports.detectFormat = void 0;
4
+ var formats_1 = require("./formats");
5
+ Object.defineProperty(exports, "detectFormat", { enumerable: true, get: function () { return formats_1.detectFormat; } });
6
+ Object.defineProperty(exports, "sourceFromPath", { enumerable: true, get: function () { return formats_1.sourceFromPath; } });
7
+ Object.defineProperty(exports, "destinationFromPath", { enumerable: true, get: function () { return formats_1.destinationFromPath; } });
8
+ var rows_1 = require("./rows");
9
+ Object.defineProperty(exports, "readRows", { enumerable: true, get: function () { return rows_1.readRows; } });
10
+ Object.defineProperty(exports, "writeRows", { enumerable: true, get: function () { return rows_1.writeRows; } });
11
+ Object.defineProperty(exports, "convert", { enumerable: true, get: function () { return rows_1.convert; } });
12
+ Object.defineProperty(exports, "DEFAULT_PENDING_NAME", { enumerable: true, get: function () { return rows_1.DEFAULT_PENDING_NAME; } });
13
+ var clean_1 = require("./clean");
14
+ Object.defineProperty(exports, "clean", { enumerable: true, get: function () { return clean_1.clean; } });
15
+ Object.defineProperty(exports, "fillEmpty", { enumerable: true, get: function () { return clean_1.fillEmpty; } });
16
+ Object.defineProperty(exports, "combineColumns", { enumerable: true, get: function () { return clean_1.combineColumns; } });
17
+ Object.defineProperty(exports, "validate", { enumerable: true, get: function () { return clean_1.validate; } });
18
+ Object.defineProperty(exports, "dedupe", { enumerable: true, get: function () { return clean_1.dedupe; } });
19
+ Object.defineProperty(exports, "merge", { enumerable: true, get: function () { return clean_1.merge; } });
20
+ Object.defineProperty(exports, "mergeColumnsFor", { enumerable: true, get: function () { return clean_1.mergeColumnsFor; } });
21
+ Object.defineProperty(exports, "defineConfig", { enumerable: true, get: function () { return clean_1.defineConfig; } });
22
+ Object.defineProperty(exports, "defineRules", { enumerable: true, get: function () { return clean_1.defineRules; } });
23
+ Object.defineProperty(exports, "rulesSchema", { enumerable: true, get: function () { return clean_1.rulesSchema; } });
24
+ var inspect_1 = require("./inspect");
25
+ Object.defineProperty(exports, "describeColumns", { enumerable: true, get: function () { return inspect_1.describeColumns; } });
26
+ Object.defineProperty(exports, "listNormalizers", { enumerable: true, get: function () { return inspect_1.listNormalizers; } });
27
+ Object.defineProperty(exports, "normalize", { enumerable: true, get: function () { return inspect_1.normalize; } });
@@ -0,0 +1,13 @@
1
+ import { TableRow } from "../types";
2
+ export type ColumnKind = "número" | "texto" | "misto" | "vazio";
3
+ export interface ColumnInfo {
4
+ name: string;
5
+ filled: number;
6
+ empty: number;
7
+ distinct: number;
8
+ kind: ColumnKind;
9
+ examples: string[];
10
+ }
11
+ export declare function describeColumns(rows: TableRow[]): ColumnInfo[];
12
+ export declare function listNormalizers(): string[];
13
+ export declare function normalize(name: string, value: string | number): string | number | null;
@@ -0,0 +1,44 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.describeColumns = describeColumns;
4
+ exports.listNormalizers = listNormalizers;
5
+ exports.normalize = normalize;
6
+ const header_1 = require("../io/header");
7
+ const normalizers_1 = require("../normalizers");
8
+ const registry_1 = require("../normalizers/registry");
9
+ const EXAMPLES = 3;
10
+ function isEmpty(value) {
11
+ return value === null || value === undefined || value === "";
12
+ }
13
+ function kindOf(values) {
14
+ if (values.length === 0)
15
+ return "vazio";
16
+ const numbers = values.filter((value) => typeof value === "number").length;
17
+ if (numbers === values.length)
18
+ return "número";
19
+ return numbers === 0 ? "texto" : "misto";
20
+ }
21
+ function describeColumns(rows) {
22
+ return (0, header_1.buildHeader)(rows).map((name) => {
23
+ const present = rows
24
+ .map((row) => row[name])
25
+ .filter((value) => !isEmpty(value));
26
+ const distinct = new Set(present.map((value) => String(value)));
27
+ return {
28
+ name,
29
+ filled: present.length,
30
+ empty: rows.length - present.length,
31
+ distinct: distinct.size,
32
+ kind: kindOf(present),
33
+ examples: [...distinct].slice(0, EXAMPLES),
34
+ };
35
+ });
36
+ }
37
+ function listNormalizers() {
38
+ (0, normalizers_1.registerBuiltinKeyNormalizers)();
39
+ return (0, registry_1.listKeyNormalizers)();
40
+ }
41
+ function normalize(name, value) {
42
+ (0, normalizers_1.registerBuiltinKeyNormalizers)();
43
+ return (0, registry_1.getKeyNormalizer)(name)(value)?.key ?? null;
44
+ }
@@ -0,0 +1,2 @@
1
+ import { z } from "zod";
2
+ export declare function parseWith<T extends z.ZodType>(schema: T, value: unknown, label: string): z.infer<T>;