pluto-enem 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ Metadata-Version: 2.4
2
+ Name: pluto-enem
3
+ Version: 0.1.0
4
+ Summary: Biblioteca Python para questões, alternativas e imagens do ENEM
5
+ Requires-Python: >=3.10
@@ -0,0 +1,5 @@
1
+ from .library import PlutoLibrary
2
+ from .models import Alternative, ImageAsset, Question
3
+
4
+ __all__ = ["Alternative", "ImageAsset", "PlutoLibrary", "Question"]
5
+ __version__ = "0.1.0"
@@ -0,0 +1,251 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import random
5
+ import sqlite3
6
+ from pathlib import Path
7
+ from typing import Iterator, Sequence
8
+
9
+ from .models import Alternative, ImageAsset, Question
10
+
11
+
12
+ class PlutoLibrary:
13
+ """Biblioteca local de questões do ENEM para pesquisa e montagem de provas."""
14
+
15
+ def __init__(self, database_path: str | Path = "Pluto/database/pluto.db") -> None:
16
+ self.database_path = Path(database_path).expanduser().resolve()
17
+ if not self.database_path.exists():
18
+ raise FileNotFoundError(
19
+ f"Banco não encontrado: {self.database_path}. Execute python3 pluto_import.py primeiro."
20
+ )
21
+ self.library_dir = self.database_path.parent.parent
22
+ self.project_dir = self.library_dir.parent
23
+
24
+ def _connect(self) -> sqlite3.Connection:
25
+ connection = sqlite3.connect(self.database_path)
26
+ connection.row_factory = sqlite3.Row
27
+ connection.execute("PRAGMA foreign_keys = ON")
28
+ return connection
29
+
30
+ def _absolute_path(self, value: str | None) -> Path | None:
31
+ if not value:
32
+ return None
33
+ path = Path(value)
34
+ if path.is_absolute():
35
+ return path
36
+ return self.project_dir / path if path.parts[:1] != (self.project_dir.name,) else self.project_dir.parent / path
37
+
38
+ def years(self) -> list[int]:
39
+ with self._connect() as connection:
40
+ rows = connection.execute("SELECT DISTINCT ano FROM questoes ORDER BY ano").fetchall()
41
+ return [int(row["ano"]) for row in rows]
42
+
43
+ def disciplines(self, year: int | None = None) -> list[str]:
44
+ query = "SELECT DISTINCT disciplina FROM questoes WHERE disciplina IS NOT NULL"
45
+ parameters: list[object] = []
46
+ if year is not None:
47
+ query += " AND ano = ?"
48
+ parameters.append(year)
49
+ query += " ORDER BY disciplina"
50
+ with self._connect() as connection:
51
+ rows = connection.execute(query, parameters).fetchall()
52
+ return [str(row["disciplina"]) for row in rows]
53
+
54
+ def languages(self, year: int | None = None) -> list[str]:
55
+ query = "SELECT DISTINCT idioma FROM questoes WHERE idioma IS NOT NULL"
56
+ parameters: list[object] = []
57
+ if year is not None:
58
+ query += " AND ano = ?"
59
+ parameters.append(year)
60
+ query += " ORDER BY idioma"
61
+ with self._connect() as connection:
62
+ return [str(row[0]) for row in connection.execute(query, parameters)]
63
+
64
+ def count(
65
+ self,
66
+ year: int | None = None,
67
+ discipline: str | None = None,
68
+ language: str | None = None,
69
+ with_images: bool | None = None,
70
+ ) -> int:
71
+ query = "SELECT COUNT(DISTINCT q.id) FROM questoes q"
72
+ parameters: list[object] = []
73
+ if with_images is True:
74
+ query += " JOIN questao_imagens qi ON qi.questao_id = q.id"
75
+ query += " WHERE 1=1"
76
+ if year is not None:
77
+ query += " AND q.ano = ?"
78
+ parameters.append(year)
79
+ if discipline is not None:
80
+ query += " AND q.disciplina = ?"
81
+ parameters.append(discipline)
82
+ if language is not None:
83
+ query += " AND q.idioma = ?"
84
+ parameters.append(language)
85
+ with self._connect() as connection:
86
+ return int(connection.execute(query, parameters).fetchone()[0])
87
+
88
+ def questions(
89
+ self,
90
+ year: int | None = None,
91
+ discipline: str | None = None,
92
+ language: str | None = None,
93
+ with_images: bool | None = None,
94
+ exclude_years: Sequence[int] | None = None,
95
+ limit: int | None = None,
96
+ random_order: bool = False,
97
+ ) -> Iterator[Question]:
98
+ query = "SELECT DISTINCT q.* FROM questoes q"
99
+ parameters: list[object] = []
100
+ if with_images is True:
101
+ query += " JOIN questao_imagens qi ON qi.questao_id = q.id"
102
+ query += " WHERE 1=1"
103
+ if year is not None:
104
+ query += " AND q.ano = ?"
105
+ parameters.append(year)
106
+ if discipline is not None:
107
+ query += " AND q.disciplina = ?"
108
+ parameters.append(discipline)
109
+ if language is not None:
110
+ query += " AND q.idioma = ?"
111
+ parameters.append(language)
112
+ if exclude_years:
113
+ placeholders = ",".join("?" for _ in exclude_years)
114
+ query += f" AND q.ano NOT IN ({placeholders})"
115
+ parameters.extend(exclude_years)
116
+ query += " ORDER BY RANDOM()" if random_order else " ORDER BY q.ano, q.numero, q.disciplina, q.idioma"
117
+ if limit is not None:
118
+ if limit < 1:
119
+ return
120
+ query += " LIMIT ?"
121
+ parameters.append(limit)
122
+ with self._connect() as connection:
123
+ for row in connection.execute(query, parameters):
124
+ yield self._question(connection, row)
125
+
126
+ def sample(
127
+ self,
128
+ amount: int,
129
+ discipline: str | None = None,
130
+ year: int | None = None,
131
+ exclude_years: Sequence[int] | None = None,
132
+ seed: int | None = None,
133
+ ) -> list[Question]:
134
+ """Retorna uma amostra reprodutível, útil para montar um simulado."""
135
+ if amount < 1:
136
+ return []
137
+ candidates = list(self.questions(
138
+ year=year, discipline=discipline, exclude_years=exclude_years
139
+ ))
140
+ generator = random.Random(seed)
141
+ generator.shuffle(candidates)
142
+ return candidates[:amount]
143
+
144
+ def get_question(
145
+ self,
146
+ year: int,
147
+ number: int,
148
+ discipline: str | None = None,
149
+ language: str | None = None,
150
+ ) -> Question | None:
151
+ query = "SELECT * FROM questoes WHERE ano = ? AND numero = ?"
152
+ parameters: list[object] = [year, number]
153
+ if discipline is not None:
154
+ query += " AND disciplina = ?"
155
+ parameters.append(discipline)
156
+ if language is not None:
157
+ query += " AND idioma = ?"
158
+ parameters.append(language)
159
+ query += " ORDER BY idioma LIMIT 1"
160
+ with self._connect() as connection:
161
+ row = connection.execute(query, parameters).fetchone()
162
+ return self._question(connection, row) if row else None
163
+
164
+ def search(self, text: str, limit: int | None = None) -> Iterator[Question]:
165
+ pattern = f"%{text}%"
166
+ query = """
167
+ SELECT * FROM questoes
168
+ WHERE COALESCE(titulo, '') LIKE ?
169
+ OR COALESCE(contexto, '') LIKE ?
170
+ OR COALESCE(texto_completo, '') LIKE ?
171
+ ORDER BY ano, numero
172
+ """
173
+ parameters: list[object] = [pattern, pattern, pattern]
174
+ if limit is not None:
175
+ query += " LIMIT ?"
176
+ parameters.append(limit)
177
+ with self._connect() as connection:
178
+ for row in connection.execute(query, parameters):
179
+ yield self._question(connection, row)
180
+
181
+ def export_jsonl(self, output_path: str | Path, **filters: object) -> int:
182
+ """Exporta questões completas para treinamento ou processamento posterior."""
183
+ destination = Path(output_path)
184
+ destination.parent.mkdir(parents=True, exist_ok=True)
185
+ count = 0
186
+ with destination.open("w", encoding="utf-8") as output:
187
+ for question in self.questions(**filters):
188
+ record = {
189
+ "id": question.id, "ano": question.year, "numero": question.number,
190
+ "disciplina": question.discipline, "idioma": question.language,
191
+ "titulo": question.title, "contexto": question.context,
192
+ "texto_completo": question.full_text,
193
+ "alternativa_correta": question.correct_alternative,
194
+ "alternativas": [
195
+ {"letra": item.letter, "texto": item.text, "correta": item.correct}
196
+ for item in question.alternatives
197
+ ],
198
+ "imagens": [
199
+ {"arquivo": str(item.file), "thumbnail": str(item.thumbnail), "origem": item.origin}
200
+ for item in question.images
201
+ ],
202
+ "fonte": question.source,
203
+ }
204
+ output.write(json.dumps(record, ensure_ascii=False) + "\n")
205
+ count += 1
206
+ return count
207
+
208
+ def _question(self, connection: sqlite3.Connection, row: sqlite3.Row) -> Question:
209
+ alternatives = connection.execute(
210
+ "SELECT * FROM alternativas WHERE questao_id = ? ORDER BY letra", (row["id"],)
211
+ ).fetchall()
212
+ images = connection.execute(
213
+ """
214
+ SELECT imagens.*, questao_imagens.origem
215
+ FROM imagens JOIN questao_imagens ON questao_imagens.imagem_id = imagens.id
216
+ WHERE questao_imagens.questao_id = ? ORDER BY imagens.id
217
+ """,
218
+ (row["id"],),
219
+ ).fetchall()
220
+ keys = row.keys()
221
+ raw = row["dados_api"] if "dados_api" in keys else None
222
+ try:
223
+ api_data = json.loads(raw) if raw else None
224
+ except json.JSONDecodeError:
225
+ api_data = None
226
+ return Question(
227
+ id=int(row["id"]), year=int(row["ano"]), number=int(row["numero"]),
228
+ discipline=row["disciplina"], language=row["idioma"], title=row["titulo"],
229
+ context=row["contexto"], alternatives_intro=row["introducao_alternativas"],
230
+ correct_alternative=row["alternativa_correta"],
231
+ alternatives=tuple(
232
+ Alternative(
233
+ letter=str(item["letra"]), text=item["texto"],
234
+ file=self._absolute_path(item["arquivo"]), correct=bool(item["correta"])
235
+ ) for item in alternatives
236
+ ),
237
+ images=tuple(
238
+ ImageAsset(
239
+ id=str(item["id"]), year=int(item["ano"]), discipline=str(item["disciplina"]),
240
+ question_number=int(item["questao"]), file=self._absolute_path(item["arquivo"]),
241
+ thumbnail=self._absolute_path(item["thumbnail"]), sha256=str(item["sha256"]),
242
+ width=int(item["largura"]), height=int(item["altura"]), origin=item["origem"]
243
+ ) for item in images
244
+ ),
245
+ full_text=row["texto_completo"] if "texto_completo" in keys else row["contexto"],
246
+ source=(row["fonte"] if "fonte" in keys and row["fonte"] else "enem.dev"),
247
+ api_data=api_data,
248
+ )
249
+
250
+
251
+ __all__ = ["PlutoLibrary"]
@@ -0,0 +1,54 @@
1
+ from dataclasses import dataclass
2
+ from pathlib import Path
3
+ from typing import Any
4
+
5
+
6
+ @dataclass(frozen=True)
7
+ class Alternative:
8
+ letter: str
9
+ text: str | None
10
+ file: Path | None
11
+ correct: bool
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class ImageAsset:
16
+ id: str
17
+ year: int
18
+ discipline: str
19
+ question_number: int
20
+ file: Path
21
+ thumbnail: Path
22
+ sha256: str
23
+ width: int
24
+ height: int
25
+ origin: str | None = None
26
+
27
+
28
+ @dataclass(frozen=True)
29
+ class Question:
30
+ id: int
31
+ year: int
32
+ number: int
33
+ discipline: str | None
34
+ language: str | None
35
+ title: str | None
36
+ context: str | None
37
+ alternatives_intro: str | None
38
+ correct_alternative: str | None
39
+ alternatives: tuple[Alternative, ...]
40
+ images: tuple[ImageAsset, ...]
41
+ full_text: str | None = None
42
+ source: str = "enem.dev"
43
+ api_data: dict[str, Any] | None = None
44
+
45
+ @property
46
+ def correct_answer(self) -> Alternative | None:
47
+ return next((item for item in self.alternatives if item.correct), None)
48
+
49
+ @property
50
+ def has_images(self) -> bool:
51
+ return bool(self.images)
52
+
53
+
54
+ __all__ = ["Alternative", "ImageAsset", "Question"]
@@ -0,0 +1,5 @@
1
+ Metadata-Version: 2.4
2
+ Name: pluto-enem
3
+ Version: 0.1.0
4
+ Summary: Biblioteca Python para questões, alternativas e imagens do ENEM
5
+ Requires-Python: >=3.10
@@ -0,0 +1,8 @@
1
+ pyproject.toml
2
+ pluto/__init__.py
3
+ pluto/library.py
4
+ pluto/models.py
5
+ pluto_enem.egg-info/PKG-INFO
6
+ pluto_enem.egg-info/SOURCES.txt
7
+ pluto_enem.egg-info/dependency_links.txt
8
+ pluto_enem.egg-info/top_level.txt
@@ -0,0 +1 @@
1
+ pluto
@@ -0,0 +1,12 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pluto-enem"
7
+ version = "0.1.0"
8
+ description = "Biblioteca Python para questões, alternativas e imagens do ENEM"
9
+ requires-python = ">=3.10"
10
+
11
+ [tool.setuptools.packages.find]
12
+ include = ["pluto*"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+