uydatos 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- uydatos/__init__.py +24 -0
- uydatos/ancap.py +200 -0
- uydatos/bcu.py +233 -0
- uydatos/catalogo.py +81 -0
- uydatos/ibge.py +97 -0
- uydatos/ideuy.py +306 -0
- uydatos/indec.py +155 -0
- uydatos/inia.py +276 -0
- uydatos/py.typed +0 -0
- uydatos/rut.py +114 -0
- uydatos/salto_grande.py +189 -0
- uydatos/semanas.py +90 -0
- uydatos-0.2.1.dist-info/METADATA +150 -0
- uydatos-0.2.1.dist-info/RECORD +16 -0
- uydatos-0.2.1.dist-info/WHEEL +4 -0
- uydatos-0.2.1.dist-info/licenses/LICENSE +21 -0
uydatos/__init__.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""uydatos — lectores estrictos de datos públicos de Uruguay y la región.
|
|
2
|
+
|
|
3
|
+
Cada módulo lee una fuente y **rechaza en vez de arreglar**: una respuesta mala, una columna
|
|
4
|
+
cambiada o un hueco en la serie levantan una excepción con el motivo, en lugar de devolver una
|
|
5
|
+
serie creíble pero falsa. La única función de cada módulo que sale a internet se llama
|
|
6
|
+
`descargar*()`; todo lo demás trabaja sobre archivos en disco.
|
|
7
|
+
|
|
8
|
+
Módulos:
|
|
9
|
+
|
|
10
|
+
- `bcu` — cotizaciones del Banco Central del Uruguay (dólar, UI, peso argentino, real).
|
|
11
|
+
- `ancap` — precios máximos de combustibles fijados por el Poder Ejecutivo.
|
|
12
|
+
- `inia` — clima diario de INIA Salto Grande, con normales y anomalías.
|
|
13
|
+
- `salto_grande` — estaciones hidrometeorológicas de la CTM de Salto Grande.
|
|
14
|
+
- `indec` — IPC nacional de Argentina (API de series de datos.gob.ar).
|
|
15
|
+
- `ibge` — IPCA de Brasil (API SIDRA del IBGE).
|
|
16
|
+
- `ideuy` — geocodificación de direcciones uruguayas con IDE.uy.
|
|
17
|
+
- `rut` — dígito verificador de RUT y cédula uruguaya.
|
|
18
|
+
- `semanas` — semanas ISO 8601 (W1, W2...) con los bordes de año resueltos.
|
|
19
|
+
- `catalogo` — recursos del catálogo de datos abiertos de Uruguay (CKAN).
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
__version__ = "0.2.1"
|
|
23
|
+
|
|
24
|
+
USER_AGENT = f"uydatos/{__version__} (lectura de datos abiertos)"
|
uydatos/ancap.py
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""Precios máximos de combustibles de ANCAP, aprobados por decreto del Poder Ejecutivo.
|
|
2
|
+
|
|
3
|
+
Dataset `ancap-precio-combustible-pe` del catálogo de datos abiertos: **precio máximo al
|
|
4
|
+
público, IVA incluido**. **No es** `ancap-precio-combustible-ex-planta`, que es el precio
|
|
5
|
+
deducidos flete, tasas y margen. Los dos se citan como "los precios de ANCAP" y no son el mismo
|
|
6
|
+
número.
|
|
7
|
+
|
|
8
|
+
Las trampas (medidas en octubre de 2026)
|
|
9
|
+
---------------------------------------
|
|
10
|
+
1. **Las columnas vienen invertidas.** El encabezado dice `Unidad` y `Valor`, pero `Unidad`
|
|
11
|
+
trae el **número** y `Valor` la **unidad de medida**. Quien sume "Valor" suma `'$/lt'`.
|
|
12
|
+
Si la fuente las vuelve a dar vuelta, el lector lo rechaza.
|
|
13
|
+
2. **El archivo no es una serie: es un log de publicaciones acumulativas.** Cada mes ANCAP
|
|
14
|
+
republica el año entero hasta ese mes y el archivo **apila** las publicaciones: un mismo mes
|
|
15
|
+
aparece decenas de veces. Contar o promediar filas mide publicaciones, no precios: usar
|
|
16
|
+
`serie_mensual()`.
|
|
17
|
+
3. **El BOM viene doble**: abrir con `utf-8-sig` no alcanza, queda un `\\ufeff` pegado al
|
|
18
|
+
nombre de la primera columna.
|
|
19
|
+
4. **`S/C`** es *sin cotización*: un nulo, no un cero.
|
|
20
|
+
5. **Coma decimal y punto de miles** (`1.234,50`): un `float()` pelado no sirve.
|
|
21
|
+
|
|
22
|
+
Uso:
|
|
23
|
+
python -m uydatos.ancap bajar precios.csv
|
|
24
|
+
python -m uydatos.ancap serie precios.csv "GASOIL 50-S *"
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import csv
|
|
31
|
+
import sys
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Optional, Sequence
|
|
34
|
+
|
|
35
|
+
DATASET = "ancap-precio-combustible-pe"
|
|
36
|
+
URL_CSV = (
|
|
37
|
+
"https://catalogodatos.gub.uy/dataset/556beb30-02fc-4d06-909a-f42083c89ef7/resource/"
|
|
38
|
+
"1a37fc4a-44fd-4842-bcb4-b5845dc8eced/download/datos-de-precios-de-combustibles.csv"
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
COLUMNAS_ESPERADAS = ["Año", "Mes", "Producto", "Unidad", "Valor"]
|
|
42
|
+
SIN_COTIZACION = "S/C"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AncapInvalido(ValueError):
|
|
46
|
+
"""La serie no cumple el contrato de la fuente."""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def numero_europeo(texto: str) -> Optional[float]:
|
|
50
|
+
"""`'1.234,50'` -> 1234.5. `None` si no se puede leer: nunca un cero inventado."""
|
|
51
|
+
t = texto.replace("\xa0", " ").strip().replace(".", "").replace(",", ".")
|
|
52
|
+
if not t:
|
|
53
|
+
return None
|
|
54
|
+
try:
|
|
55
|
+
return float(t)
|
|
56
|
+
except ValueError:
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _limpiar_encabezado(campos: list[str]) -> list[str]:
|
|
61
|
+
return [c.replace("", "").strip() for c in campos]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def leer_publicaciones(ruta: Path) -> list[dict]:
|
|
65
|
+
"""Lee el archivo **crudo**: una fila por publicación, sin deduplicar.
|
|
66
|
+
|
|
67
|
+
Devuelve, por fila, `anio`, `mes`, `producto`, `precio` (float | None) y `unidad`.
|
|
68
|
+
Para el precio de un mes usar `serie_mensual()`, no esto.
|
|
69
|
+
"""
|
|
70
|
+
ruta = Path(ruta)
|
|
71
|
+
if not ruta.is_file():
|
|
72
|
+
raise AncapInvalido(f"{ruta}: el archivo no está en el disco")
|
|
73
|
+
with ruta.open(encoding="utf-8-sig", newline="") as f:
|
|
74
|
+
lector = csv.reader(f, delimiter=";")
|
|
75
|
+
try:
|
|
76
|
+
encabezado = _limpiar_encabezado(next(lector))
|
|
77
|
+
except StopIteration:
|
|
78
|
+
raise AncapInvalido(f"{ruta}: archivo vacío") from None
|
|
79
|
+
if encabezado != COLUMNAS_ESPERADAS:
|
|
80
|
+
raise AncapInvalido(f"{ruta}: columnas inesperadas {encabezado}, se esperaban {COLUMNAS_ESPERADAS}")
|
|
81
|
+
publicaciones = []
|
|
82
|
+
for n, campos in enumerate(lector, start=2):
|
|
83
|
+
if not any(c.strip() for c in campos):
|
|
84
|
+
continue
|
|
85
|
+
if len(campos) != len(COLUMNAS_ESPERADAS):
|
|
86
|
+
raise AncapInvalido(f"{ruta} fila {n}: {len(campos)} columnas")
|
|
87
|
+
anio_crudo, mes_crudo, producto, precio_crudo, unidad = (c.strip() for c in campos)
|
|
88
|
+
# `Valor` tiene que traer la UNIDAD. Si trae un número, la fuente invirtió las columnas.
|
|
89
|
+
if unidad and unidad != SIN_COTIZACION and not unidad.startswith("$"):
|
|
90
|
+
raise AncapInvalido(
|
|
91
|
+
f"{ruta} fila {n}: la columna `Valor` debería traer la unidad y trae {unidad!r}. "
|
|
92
|
+
f"¿La fuente invirtió las columnas?"
|
|
93
|
+
)
|
|
94
|
+
try:
|
|
95
|
+
anio, mes = int(anio_crudo), int(mes_crudo)
|
|
96
|
+
except ValueError:
|
|
97
|
+
raise AncapInvalido(f"{ruta} fila {n}: año/mes ilegible {campos[:2]!r}") from None
|
|
98
|
+
if precio_crudo == SIN_COTIZACION:
|
|
99
|
+
precio = None
|
|
100
|
+
else:
|
|
101
|
+
precio = numero_europeo(precio_crudo)
|
|
102
|
+
if precio is None:
|
|
103
|
+
raise AncapInvalido(f"{ruta} fila {n}: precio ilegible {precio_crudo!r}")
|
|
104
|
+
publicaciones.append(
|
|
105
|
+
{
|
|
106
|
+
"anio": anio,
|
|
107
|
+
"mes": mes,
|
|
108
|
+
"producto": producto,
|
|
109
|
+
"precio": precio,
|
|
110
|
+
"unidad": "" if unidad == SIN_COTIZACION else unidad,
|
|
111
|
+
}
|
|
112
|
+
)
|
|
113
|
+
if not publicaciones:
|
|
114
|
+
raise AncapInvalido(f"{ruta}: archivo sin filas de datos")
|
|
115
|
+
return publicaciones
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def serie_mensual(publicaciones: list[dict]) -> list[dict]:
|
|
119
|
+
"""Deduplica el log apilado: **una fila por (producto, año, mes)**.
|
|
120
|
+
|
|
121
|
+
Desempate declarado: **el valor manda sobre el `S/C`**. Un mes que alguna vez tuvo precio no
|
|
122
|
+
se pierde porque una publicación posterior diga `S/C`; entre dos precios distintos gana el
|
|
123
|
+
de la publicación más reciente (la que aparece después en el archivo).
|
|
124
|
+
"""
|
|
125
|
+
orden: list[tuple] = []
|
|
126
|
+
estado: dict[tuple, dict] = {}
|
|
127
|
+
for p in publicaciones:
|
|
128
|
+
clave = (p["producto"], p["anio"], p["mes"])
|
|
129
|
+
if clave not in estado:
|
|
130
|
+
orden.append(clave)
|
|
131
|
+
estado[clave] = {
|
|
132
|
+
"anio": p["anio"],
|
|
133
|
+
"mes": p["mes"],
|
|
134
|
+
"producto": p["producto"],
|
|
135
|
+
"precio": None,
|
|
136
|
+
"unidad": "",
|
|
137
|
+
}
|
|
138
|
+
if p["precio"] is not None:
|
|
139
|
+
estado[clave]["precio"] = p["precio"]
|
|
140
|
+
estado[clave]["unidad"] = p["unidad"]
|
|
141
|
+
return [estado[k] for k in orden]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def precio_del_mes(serie: list[dict], producto: str, anio: int, mes: int) -> Optional[float]:
|
|
145
|
+
"""El precio de ese producto en ese mes, o `None` si nunca tuvo cotización.
|
|
146
|
+
|
|
147
|
+
Espera la salida de `serie_mensual()`, no la de `leer_publicaciones()`.
|
|
148
|
+
"""
|
|
149
|
+
for s in serie:
|
|
150
|
+
if s["producto"] == producto and s["anio"] == anio and s["mes"] == mes:
|
|
151
|
+
return s["precio"]
|
|
152
|
+
return None
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def descargar(destino: Path, url: str = URL_CSV) -> Path:
|
|
156
|
+
"""Baja el CSV del catálogo a `destino`. **Única función que sale a la red.**
|
|
157
|
+
|
|
158
|
+
Si la URL del recurso cambia, `uydatos.catalogo.recursos(DATASET)` da la vigente.
|
|
159
|
+
"""
|
|
160
|
+
import urllib.request
|
|
161
|
+
|
|
162
|
+
from uydatos import USER_AGENT
|
|
163
|
+
|
|
164
|
+
destino = Path(destino)
|
|
165
|
+
destino.parent.mkdir(parents=True, exist_ok=True)
|
|
166
|
+
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
167
|
+
with urllib.request.urlopen(req, timeout=90) as r: # noqa: S310 — host fijo
|
|
168
|
+
destino.write_bytes(r.read())
|
|
169
|
+
return destino
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
173
|
+
ap = argparse.ArgumentParser(prog="python -m uydatos.ancap", description=__doc__.splitlines()[0])
|
|
174
|
+
sub = ap.add_subparsers(dest="orden", required=True)
|
|
175
|
+
b = sub.add_parser("bajar")
|
|
176
|
+
b.add_argument("archivo", type=Path)
|
|
177
|
+
s = sub.add_parser("serie", help="precio por mes de un producto (sin argumento: lista los productos)")
|
|
178
|
+
s.add_argument("archivo", type=Path)
|
|
179
|
+
s.add_argument("producto", nargs="?")
|
|
180
|
+
args = ap.parse_args(argv)
|
|
181
|
+
|
|
182
|
+
if args.orden == "bajar":
|
|
183
|
+
print(f"bajado {descargar(args.archivo)}")
|
|
184
|
+
return 0
|
|
185
|
+
publicaciones = leer_publicaciones(args.archivo)
|
|
186
|
+
serie = serie_mensual(publicaciones)
|
|
187
|
+
print(f"{len(publicaciones)} publicaciones, {len(serie)} meses distintos (producto x año x mes)")
|
|
188
|
+
if not args.producto:
|
|
189
|
+
for p in sorted({s["producto"] for s in serie}):
|
|
190
|
+
print(f" {p}")
|
|
191
|
+
return 0
|
|
192
|
+
print("anio;mes;precio;unidad")
|
|
193
|
+
for s in serie:
|
|
194
|
+
if s["producto"] == args.producto:
|
|
195
|
+
print(f"{s['anio']};{s['mes']:02d};{'' if s['precio'] is None else s['precio']};{s['unidad']}")
|
|
196
|
+
return 0
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
if __name__ == "__main__":
|
|
200
|
+
sys.exit(main())
|
uydatos/bcu.py
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Cotizaciones del Banco Central del Uruguay, desde su web service SOAP público.
|
|
2
|
+
|
|
3
|
+
Monedas declaradas (códigos y nombres tal como los devuelve `awsbcumonedas`):
|
|
4
|
+
|
|
5
|
+
- **2225, DLS. USA BILLETE**.
|
|
6
|
+
- **9800, UNIDAD INDEXADA**: sigue al IPC de Uruguay día a día. Sirve de deflactor para comparar
|
|
7
|
+
pesos de dos fechas sin bajar el IPC del INE aparte (`a_pesos_de`).
|
|
8
|
+
- **501, PESO ARG.BILLETE** y **1001, REAL BILLETE**: las monedas de las dos fronteras. Con el
|
|
9
|
+
IPC de cada país (`uydatos.indec`, `uydatos.ibge`) arman el tipo de cambio real bilateral.
|
|
10
|
+
|
|
11
|
+
Las trampas, y por qué el lector las corta
|
|
12
|
+
-----------------------------------------
|
|
13
|
+
1. **Una respuesta mala igual se guarda.** `status 0` o `codigoerror` distinto de 0 se rechaza:
|
|
14
|
+
no es "un mes sin cotización".
|
|
15
|
+
2. **Un archivo, una moneda.** La UI (~6,3) leída como dólar (~40) no da error, da un cálculo
|
|
16
|
+
absurdo. Se rechaza el archivo que trae otra moneda o dos.
|
|
17
|
+
3. **Fines de semana y feriados no vienen.** `cotizacion_del_dia` devuelve `None` salvo que se
|
|
18
|
+
pida `hacia_atras=True`, y aun así no salta más de `MAX_DIAS_ATRAS` días: un hueco de un mes
|
|
19
|
+
es un archivo que falta, no un fin de semana.
|
|
20
|
+
4. **El mes en curso viene parcial.** `promedio_mensual` declara los días de cada mes.
|
|
21
|
+
5. **El servicio rechaza rangos largos** (`codigoerror 104`): `descargar` baja un mes por archivo.
|
|
22
|
+
|
|
23
|
+
Qué cotización corresponde para un fin fiscal (la del día, la del día hábil anterior, compra o
|
|
24
|
+
venta) **no lo decide este lector**. `TCC` y `TCV` vienen iguales en todo lo observado; se usa
|
|
25
|
+
`TCV` y se rechaza un día donde difieran, para que nadie elija en silencio.
|
|
26
|
+
|
|
27
|
+
Uso:
|
|
28
|
+
python -m uydatos.bcu bajar DIRECTORIO 9800 2024-01 2026-09
|
|
29
|
+
python -m uydatos.bcu serie DIRECTORIO 9800
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import argparse
|
|
35
|
+
import re
|
|
36
|
+
import sys
|
|
37
|
+
import xml.etree.ElementTree as ET
|
|
38
|
+
from collections import defaultdict
|
|
39
|
+
from datetime import date, timedelta
|
|
40
|
+
from pathlib import Path
|
|
41
|
+
from typing import Optional, Sequence
|
|
42
|
+
|
|
43
|
+
URL = "https://cotizaciones.bcu.gub.uy/wscotizaciones/servlet/awsbcucotizaciones"
|
|
44
|
+
|
|
45
|
+
MONEDAS = {
|
|
46
|
+
2225: "DLS. USA BILLETE",
|
|
47
|
+
9800: "UNIDAD INDEXADA",
|
|
48
|
+
501: "PESO ARG.BILLETE",
|
|
49
|
+
1001: "REAL BILLETE",
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
MAX_DIAS_ATRAS = 7
|
|
53
|
+
|
|
54
|
+
_NS = "{Cotiza}"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class BcuInvalido(ValueError):
|
|
58
|
+
"""El archivo no cumple el contrato de la fuente."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _texto(nodo: ET.Element, etiqueta: str) -> str:
|
|
62
|
+
hijo = nodo.find(f"{_NS}{etiqueta}")
|
|
63
|
+
return (hijo.text or "").strip() if hijo is not None else ""
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def leer_crudo(ruta: Path, moneda: int) -> dict[date, float]:
|
|
67
|
+
"""Lee una respuesta SOAP guardada y devuelve {fecha: TCV}. Rechaza en vez de arreglar."""
|
|
68
|
+
ruta = Path(ruta)
|
|
69
|
+
try:
|
|
70
|
+
raiz = ET.parse(ruta).getroot()
|
|
71
|
+
except ET.ParseError as e:
|
|
72
|
+
raise BcuInvalido(f"{ruta.name}: no es XML ({e})") from e
|
|
73
|
+
|
|
74
|
+
status = raiz.find(f".//{_NS}respuestastatus")
|
|
75
|
+
if status is None:
|
|
76
|
+
raise BcuInvalido(f"{ruta.name}: sin respuestastatus (no es una respuesta de cotizaciones)")
|
|
77
|
+
st, cod = _texto(status, "status"), _texto(status, "codigoerror")
|
|
78
|
+
if st != "1" or cod != "0":
|
|
79
|
+
raise BcuInvalido(
|
|
80
|
+
f"{ruta.name}: respuesta mala del BCU (status {st}, codigoerror {cod}: "
|
|
81
|
+
f"{_texto(status, 'mensaje')!r}). No se lee como mes sin cotización."
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
serie: dict[date, float] = {}
|
|
85
|
+
for dato in raiz.iter(f"{_NS}datoscotizaciones.dato"):
|
|
86
|
+
m = _texto(dato, "Moneda")
|
|
87
|
+
if m != str(moneda):
|
|
88
|
+
raise BcuInvalido(f"{ruta.name}: trae la moneda {m}, se pidió {moneda} ({MONEDAS.get(moneda, '?')})")
|
|
89
|
+
fecha = date.fromisoformat(_texto(dato, "Fecha"))
|
|
90
|
+
tcc, tcv = float(_texto(dato, "TCC")), float(_texto(dato, "TCV"))
|
|
91
|
+
if tcc != tcv:
|
|
92
|
+
raise BcuInvalido(f"{ruta.name}: {fecha} trae TCC {tcc} != TCV {tcv}. Elegir una no es tarea del lector.")
|
|
93
|
+
if fecha in serie:
|
|
94
|
+
raise BcuInvalido(f"{ruta.name}: {fecha} repetido")
|
|
95
|
+
serie[fecha] = tcv
|
|
96
|
+
if not serie:
|
|
97
|
+
raise BcuInvalido(f"{ruta.name}: respuesta buena sin datos")
|
|
98
|
+
return serie
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def leer_serie(directorio: Path, moneda: int) -> dict[date, float]:
|
|
102
|
+
"""Une los archivos `<moneda>_AAAA-MM.xml` de un directorio. Un día en dos archivos es error."""
|
|
103
|
+
serie: dict[date, float] = {}
|
|
104
|
+
patron = re.compile(rf"^{moneda}_\d{{4}}-\d{{2}}\.xml$")
|
|
105
|
+
archivos = sorted(p for p in Path(directorio).iterdir() if patron.match(p.name))
|
|
106
|
+
if not archivos:
|
|
107
|
+
raise BcuInvalido(f"{directorio}: no hay archivos de la moneda {moneda}")
|
|
108
|
+
for p in archivos:
|
|
109
|
+
parte = leer_crudo(p, moneda)
|
|
110
|
+
repetidos = serie.keys() & parte.keys()
|
|
111
|
+
if repetidos:
|
|
112
|
+
raise BcuInvalido(f"{p.name}: días ya leídos en otro archivo: {sorted(repetidos)[:3]}")
|
|
113
|
+
serie.update(parte)
|
|
114
|
+
return dict(sorted(serie.items()))
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def cotizacion_del_dia(serie: dict[date, float], dia: date, hacia_atras: bool = False) -> Optional[float]:
|
|
118
|
+
"""La cotización de `dia`. Sin publicación: None, o la del último día previo si se pide."""
|
|
119
|
+
if dia in serie:
|
|
120
|
+
return serie[dia]
|
|
121
|
+
if not hacia_atras:
|
|
122
|
+
return None
|
|
123
|
+
for atras in range(1, MAX_DIAS_ATRAS + 1):
|
|
124
|
+
previo = dia - timedelta(days=atras)
|
|
125
|
+
if previo in serie:
|
|
126
|
+
return serie[previo]
|
|
127
|
+
return None
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def promedio_mensual(serie: dict[date, float]) -> list[dict]:
|
|
131
|
+
"""Promedio simple por mes calendario, con la cantidad de días publicados que lo forman."""
|
|
132
|
+
por_mes: dict[str, list[float]] = defaultdict(list)
|
|
133
|
+
for f, v in serie.items():
|
|
134
|
+
por_mes[f"{f.year:04d}-{f.month:02d}"].append(v)
|
|
135
|
+
return [
|
|
136
|
+
{"mes": mes, "promedio": round(sum(vs) / len(vs), 6), "dias": len(vs)} for mes, vs in sorted(por_mes.items())
|
|
137
|
+
]
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def a_pesos_de(montos: dict[str, float], base: str, ui: dict[str, float]) -> dict[str, float]:
|
|
141
|
+
"""Lleva montos mensuales ('AAAA-MM' -> pesos corrientes) a pesos del mes `base`.
|
|
142
|
+
|
|
143
|
+
`ui` es {'AAAA-MM': UI promedio del mes}, por ejemplo
|
|
144
|
+
`{f["mes"]: f["promedio"] for f in promedio_mensual(leer_serie(dir, 9800))}`.
|
|
145
|
+
Un mes sin UI **se rechaza**: dejarlo nominal mezclaría pesos de dos fechas sin avisar.
|
|
146
|
+
"""
|
|
147
|
+
faltan = sorted(m for m in [*montos, base] if m not in ui)
|
|
148
|
+
if faltan:
|
|
149
|
+
raise BcuInvalido(f"sin UI para {faltan}: bajar la serie 9800 de esos meses")
|
|
150
|
+
return {m: v * ui[base] / ui[m] for m, v in montos.items()}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _sobre(moneda: int, desde: date, hasta: date) -> bytes:
|
|
154
|
+
return (
|
|
155
|
+
'<soapenv:Envelope xmlns:soapenv="http://schemas.xmlsoap.org/soap/envelope/" xmlns:cot="Cotiza">'
|
|
156
|
+
"<soapenv:Header/><soapenv:Body><cot:wsbcucotizaciones.Execute><cot:Entrada>"
|
|
157
|
+
f"<cot:Moneda><cot:item>{moneda}</cot:item></cot:Moneda>"
|
|
158
|
+
f"<cot:FechaDesde>{desde.isoformat()}</cot:FechaDesde><cot:FechaHasta>{hasta.isoformat()}</cot:FechaHasta>"
|
|
159
|
+
"<cot:Grupo>0</cot:Grupo></cot:Entrada></cot:wsbcucotizaciones.Execute></soapenv:Body></soapenv:Envelope>"
|
|
160
|
+
).encode("utf-8")
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _meses(desde: str, hasta: str) -> list[tuple[date, date]]:
|
|
164
|
+
a, m = map(int, desde.split("-"))
|
|
165
|
+
fa, fm = map(int, hasta.split("-"))
|
|
166
|
+
out = []
|
|
167
|
+
while (a, m) <= (fa, fm):
|
|
168
|
+
ini = date(a, m, 1)
|
|
169
|
+
sig = date(a + (m == 12), m % 12 + 1, 1)
|
|
170
|
+
out.append((ini, sig - timedelta(days=1)))
|
|
171
|
+
a, m = sig.year, sig.month
|
|
172
|
+
return out
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def descargar(moneda: int, desde: str, hasta: str, destino: Path) -> list[Path]:
|
|
176
|
+
"""Baja un archivo por mes (`desde` y `hasta` en 'AAAA-MM') a `destino`.
|
|
177
|
+
|
|
178
|
+
**Única función que sale a la red.** No pisa un archivo que ya existe: un mes cerrado no
|
|
179
|
+
debería cambiar, y si cambia conviene verlo comparando, no sobreescribiendo. El mes en
|
|
180
|
+
curso queda parcial: para completarlo, borrar ese archivo y volver a bajarlo.
|
|
181
|
+
"""
|
|
182
|
+
import urllib.request
|
|
183
|
+
|
|
184
|
+
if moneda not in MONEDAS:
|
|
185
|
+
raise BcuInvalido(f"moneda {moneda} no declarada en MONEDAS")
|
|
186
|
+
destino = Path(destino)
|
|
187
|
+
destino.mkdir(parents=True, exist_ok=True)
|
|
188
|
+
escritos = []
|
|
189
|
+
for ini, fin in _meses(desde, hasta):
|
|
190
|
+
p = destino / f"{moneda}_{ini:%Y-%m}.xml"
|
|
191
|
+
if p.exists():
|
|
192
|
+
continue
|
|
193
|
+
pedido = urllib.request.Request(
|
|
194
|
+
URL,
|
|
195
|
+
data=_sobre(moneda, ini, fin),
|
|
196
|
+
headers={
|
|
197
|
+
"Content-Type": "text/xml; charset=utf-8",
|
|
198
|
+
"SOAPAction": "Cotizaaction/AWSBCUCOTIZACIONES.Execute",
|
|
199
|
+
},
|
|
200
|
+
)
|
|
201
|
+
with urllib.request.urlopen(pedido, timeout=60) as r: # noqa: S310 — host fijo
|
|
202
|
+
p.write_bytes(r.read())
|
|
203
|
+
escritos.append(p)
|
|
204
|
+
return escritos
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
208
|
+
ap = argparse.ArgumentParser(prog="python -m uydatos.bcu", description=__doc__.splitlines()[0])
|
|
209
|
+
sub = ap.add_subparsers(dest="orden", required=True)
|
|
210
|
+
b = sub.add_parser("bajar", help="baja un archivo por mes")
|
|
211
|
+
b.add_argument("directorio", type=Path)
|
|
212
|
+
b.add_argument("moneda", type=int, choices=sorted(MONEDAS))
|
|
213
|
+
b.add_argument("desde", help="AAAA-MM")
|
|
214
|
+
b.add_argument("hasta", help="AAAA-MM")
|
|
215
|
+
s = sub.add_parser("serie", help="promedio mensual de lo bajado")
|
|
216
|
+
s.add_argument("directorio", type=Path)
|
|
217
|
+
s.add_argument("moneda", type=int, choices=sorted(MONEDAS))
|
|
218
|
+
args = ap.parse_args(argv)
|
|
219
|
+
|
|
220
|
+
if args.orden == "bajar":
|
|
221
|
+
for p in descargar(args.moneda, args.desde, args.hasta, args.directorio):
|
|
222
|
+
print(f"bajado {p.name}")
|
|
223
|
+
return 0
|
|
224
|
+
serie = leer_serie(args.directorio, args.moneda)
|
|
225
|
+
print(f"# {MONEDAS[args.moneda]} ({args.moneda}): {len(serie)} días, {min(serie)} a {max(serie)}")
|
|
226
|
+
print("mes;promedio;dias")
|
|
227
|
+
for fila in promedio_mensual(serie):
|
|
228
|
+
print(f"{fila['mes']};{fila['promedio']};{fila['dias']}")
|
|
229
|
+
return 0
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
if __name__ == "__main__":
|
|
233
|
+
sys.exit(main())
|
uydatos/catalogo.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Recursos del catálogo de datos abiertos de Uruguay (catalogodatos.gub.uy, CKAN).
|
|
2
|
+
|
|
3
|
+
Las URLs de descarga de un dataset cambian cuando el organismo vuelve a subir un archivo. En vez
|
|
4
|
+
de fijarlas, se le pregunta al catálogo con `package_show` cuáles son las vigentes.
|
|
5
|
+
|
|
6
|
+
Uso:
|
|
7
|
+
python -m uydatos.catalogo inia-precipitacion-temps-extremas-sg
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Optional, Sequence
|
|
17
|
+
|
|
18
|
+
API = "https://catalogodatos.gub.uy/api/3/action/package_show"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class CatalogoInvalido(ValueError):
|
|
22
|
+
"""La respuesta del catálogo no cumple el contrato de CKAN."""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def parsear_package_show(crudo: dict) -> list[dict]:
|
|
26
|
+
"""De una respuesta de `package_show`, la lista de recursos con `nombre`, `url`, `formato`
|
|
27
|
+
y `modificado`. Rechaza una respuesta con `success` falso o sin recursos."""
|
|
28
|
+
if not crudo.get("success"):
|
|
29
|
+
raise CatalogoInvalido(f"el catálogo respondió success={crudo.get('success')!r}: {crudo.get('error')}")
|
|
30
|
+
recursos = (crudo.get("result") or {}).get("resources") or []
|
|
31
|
+
if not recursos:
|
|
32
|
+
raise CatalogoInvalido("el dataset no tiene recursos")
|
|
33
|
+
return [
|
|
34
|
+
{
|
|
35
|
+
"nombre": r.get("name") or "",
|
|
36
|
+
"url": r.get("url") or "",
|
|
37
|
+
"formato": (r.get("format") or "").upper(),
|
|
38
|
+
"modificado": r.get("last_modified") or r.get("metadata_modified") or "",
|
|
39
|
+
}
|
|
40
|
+
for r in recursos
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def recursos(dataset: str) -> list[dict]:
|
|
45
|
+
"""Los recursos vigentes de un dataset. **Sale a la red.**"""
|
|
46
|
+
import urllib.parse
|
|
47
|
+
import urllib.request
|
|
48
|
+
|
|
49
|
+
from uydatos import USER_AGENT
|
|
50
|
+
|
|
51
|
+
url = f"{API}?{urllib.parse.urlencode({'id': dataset})}"
|
|
52
|
+
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
53
|
+
with urllib.request.urlopen(req, timeout=60) as r: # noqa: S310 — host fijo
|
|
54
|
+
return parsear_package_show(json.loads(r.read().decode("utf-8")))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def bajar_recurso(url: str, destino: Path) -> Path:
|
|
58
|
+
"""Baja un recurso a `destino`. **Sale a la red.**"""
|
|
59
|
+
import urllib.request
|
|
60
|
+
|
|
61
|
+
from uydatos import USER_AGENT
|
|
62
|
+
|
|
63
|
+
destino = Path(destino)
|
|
64
|
+
destino.parent.mkdir(parents=True, exist_ok=True)
|
|
65
|
+
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
66
|
+
with urllib.request.urlopen(req, timeout=90) as r: # noqa: S310 — URL del catálogo
|
|
67
|
+
destino.write_bytes(r.read())
|
|
68
|
+
return destino
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
72
|
+
ap = argparse.ArgumentParser(prog="python -m uydatos.catalogo", description=__doc__.splitlines()[0])
|
|
73
|
+
ap.add_argument("dataset")
|
|
74
|
+
args = ap.parse_args(argv)
|
|
75
|
+
for r in recursos(args.dataset):
|
|
76
|
+
print(f"{r['formato']:5} {r['modificado'][:10]} {r['nombre']}\n {r['url']}")
|
|
77
|
+
return 0
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
if __name__ == "__main__":
|
|
81
|
+
sys.exit(main())
|
uydatos/ibge.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""IPCA de Brasil (IBGE), número-índice, desde la API SIDRA (tabla 1737, variable 2266).
|
|
2
|
+
|
|
3
|
+
Base diciembre 1993 = 100. Sin clave.
|
|
4
|
+
|
|
5
|
+
Las trampas, y por qué el lector las corta
|
|
6
|
+
-----------------------------------------
|
|
7
|
+
1. **La tabla 1737 trae varias variables** (variación mensual 63, acumulada en 12 meses 2265,
|
|
8
|
+
número-índice 2266...). Una variación leída como índice no da error. Se exige 2266 y la
|
|
9
|
+
unidad `Número-índice`.
|
|
10
|
+
2. **El dato faltante es `...` o `-`**: un nulo, no un cero.
|
|
11
|
+
3. **La primera fila es el diccionario de columnas**, no un dato.
|
|
12
|
+
4. Un **índice no es una inflación**: el lector devuelve el índice.
|
|
13
|
+
|
|
14
|
+
Uso:
|
|
15
|
+
python -m uydatos.ibge bajar ipca.json
|
|
16
|
+
python -m uydatos.ibge leer ipca.json
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import sys
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Optional, Sequence
|
|
26
|
+
|
|
27
|
+
from uydatos import USER_AGENT
|
|
28
|
+
|
|
29
|
+
URL = "https://apisidra.ibge.gov.br/values/t/1737/n1/all/v/2266/p/all"
|
|
30
|
+
VARIABLE = "2266"
|
|
31
|
+
UNIDAD = "Número-índice"
|
|
32
|
+
SIN_DATO = ("...", "-", "..", "X", "")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class IpcaInvalido(ValueError):
|
|
36
|
+
"""La respuesta de SIDRA no cumple el contrato de la fuente."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def leer(ruta: Path) -> list[dict]:
|
|
40
|
+
"""[{periodo: 'AAAA-MM', valor: float}] del número-índice. Rechaza en vez de arreglar."""
|
|
41
|
+
ruta = Path(ruta)
|
|
42
|
+
try:
|
|
43
|
+
filas = json.loads(ruta.read_text(encoding="utf-8"))
|
|
44
|
+
except (OSError, json.JSONDecodeError) as e:
|
|
45
|
+
raise IpcaInvalido(f"{ruta}: ilegible ({e})") from None
|
|
46
|
+
if not isinstance(filas, list) or len(filas) < 2:
|
|
47
|
+
raise IpcaInvalido(f"{ruta}: sin filas de datos (la primera es el diccionario de columnas)")
|
|
48
|
+
serie = []
|
|
49
|
+
for f in filas[1:]:
|
|
50
|
+
if f.get("D2C") != VARIABLE or f.get("MN") != UNIDAD:
|
|
51
|
+
raise IpcaInvalido(
|
|
52
|
+
f"{ruta}: fila de la variable {f.get('D2C')} ({f.get('MN')}); se exige {VARIABLE} ({UNIDAD})"
|
|
53
|
+
)
|
|
54
|
+
valor = (f.get("V") or "").strip()
|
|
55
|
+
if valor in SIN_DATO:
|
|
56
|
+
continue
|
|
57
|
+
p = f["D3C"]
|
|
58
|
+
serie.append({"periodo": f"{p[:4]}-{p[4:6]}", "valor": float(valor)})
|
|
59
|
+
if not serie:
|
|
60
|
+
raise IpcaInvalido(f"{ruta}: ninguna fila con valor")
|
|
61
|
+
return serie
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _pedido():
|
|
65
|
+
import urllib.request
|
|
66
|
+
|
|
67
|
+
return urllib.request.Request(URL, headers={"User-Agent": USER_AGENT})
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def descargar(destino: Path) -> Path:
|
|
71
|
+
"""Baja la serie completa a `destino`. **Única función que sale a la red.**"""
|
|
72
|
+
import urllib.request
|
|
73
|
+
|
|
74
|
+
destino = Path(destino)
|
|
75
|
+
destino.parent.mkdir(parents=True, exist_ok=True)
|
|
76
|
+
with urllib.request.urlopen(_pedido(), timeout=90) as r: # noqa: S310 — host fijo
|
|
77
|
+
destino.write_bytes(r.read())
|
|
78
|
+
return destino
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
82
|
+
ap = argparse.ArgumentParser(prog="python -m uydatos.ibge", description=__doc__.splitlines()[0])
|
|
83
|
+
ap.add_argument("orden", choices=["bajar", "leer"])
|
|
84
|
+
ap.add_argument("archivo", type=Path)
|
|
85
|
+
args = ap.parse_args(argv)
|
|
86
|
+
if args.orden == "bajar":
|
|
87
|
+
print(f"bajado {descargar(args.archivo)}")
|
|
88
|
+
return 0
|
|
89
|
+
serie = leer(args.archivo)
|
|
90
|
+
print(f"IPCA número-índice: {len(serie)} meses, {serie[0]['periodo']} a {serie[-1]['periodo']}")
|
|
91
|
+
for s in serie[-3:]:
|
|
92
|
+
print(f" {s['periodo']}: {s['valor']}")
|
|
93
|
+
return 0
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
if __name__ == "__main__":
|
|
97
|
+
sys.exit(main())
|