pluto-enem 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pluto_enem-0.1.0/PKG-INFO +5 -0
- pluto_enem-0.1.0/pluto/__init__.py +5 -0
- pluto_enem-0.1.0/pluto/library.py +251 -0
- pluto_enem-0.1.0/pluto/models.py +54 -0
- pluto_enem-0.1.0/pluto_enem.egg-info/PKG-INFO +5 -0
- pluto_enem-0.1.0/pluto_enem.egg-info/SOURCES.txt +8 -0
- pluto_enem-0.1.0/pluto_enem.egg-info/dependency_links.txt +1 -0
- pluto_enem-0.1.0/pluto_enem.egg-info/top_level.txt +1 -0
- pluto_enem-0.1.0/pyproject.toml +12 -0
- pluto_enem-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import random
|
|
5
|
+
import sqlite3
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Iterator, Sequence
|
|
8
|
+
|
|
9
|
+
from .models import Alternative, ImageAsset, Question
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class PlutoLibrary:
|
|
13
|
+
"""Biblioteca local de questões do ENEM para pesquisa e montagem de provas."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, database_path: str | Path = "Pluto/database/pluto.db") -> None:
|
|
16
|
+
self.database_path = Path(database_path).expanduser().resolve()
|
|
17
|
+
if not self.database_path.exists():
|
|
18
|
+
raise FileNotFoundError(
|
|
19
|
+
f"Banco não encontrado: {self.database_path}. Execute python3 pluto_import.py primeiro."
|
|
20
|
+
)
|
|
21
|
+
self.library_dir = self.database_path.parent.parent
|
|
22
|
+
self.project_dir = self.library_dir.parent
|
|
23
|
+
|
|
24
|
+
def _connect(self) -> sqlite3.Connection:
|
|
25
|
+
connection = sqlite3.connect(self.database_path)
|
|
26
|
+
connection.row_factory = sqlite3.Row
|
|
27
|
+
connection.execute("PRAGMA foreign_keys = ON")
|
|
28
|
+
return connection
|
|
29
|
+
|
|
30
|
+
def _absolute_path(self, value: str | None) -> Path | None:
|
|
31
|
+
if not value:
|
|
32
|
+
return None
|
|
33
|
+
path = Path(value)
|
|
34
|
+
if path.is_absolute():
|
|
35
|
+
return path
|
|
36
|
+
return self.project_dir / path if path.parts[:1] != (self.project_dir.name,) else self.project_dir.parent / path
|
|
37
|
+
|
|
38
|
+
def years(self) -> list[int]:
|
|
39
|
+
with self._connect() as connection:
|
|
40
|
+
rows = connection.execute("SELECT DISTINCT ano FROM questoes ORDER BY ano").fetchall()
|
|
41
|
+
return [int(row["ano"]) for row in rows]
|
|
42
|
+
|
|
43
|
+
def disciplines(self, year: int | None = None) -> list[str]:
|
|
44
|
+
query = "SELECT DISTINCT disciplina FROM questoes WHERE disciplina IS NOT NULL"
|
|
45
|
+
parameters: list[object] = []
|
|
46
|
+
if year is not None:
|
|
47
|
+
query += " AND ano = ?"
|
|
48
|
+
parameters.append(year)
|
|
49
|
+
query += " ORDER BY disciplina"
|
|
50
|
+
with self._connect() as connection:
|
|
51
|
+
rows = connection.execute(query, parameters).fetchall()
|
|
52
|
+
return [str(row["disciplina"]) for row in rows]
|
|
53
|
+
|
|
54
|
+
def languages(self, year: int | None = None) -> list[str]:
|
|
55
|
+
query = "SELECT DISTINCT idioma FROM questoes WHERE idioma IS NOT NULL"
|
|
56
|
+
parameters: list[object] = []
|
|
57
|
+
if year is not None:
|
|
58
|
+
query += " AND ano = ?"
|
|
59
|
+
parameters.append(year)
|
|
60
|
+
query += " ORDER BY idioma"
|
|
61
|
+
with self._connect() as connection:
|
|
62
|
+
return [str(row[0]) for row in connection.execute(query, parameters)]
|
|
63
|
+
|
|
64
|
+
def count(
|
|
65
|
+
self,
|
|
66
|
+
year: int | None = None,
|
|
67
|
+
discipline: str | None = None,
|
|
68
|
+
language: str | None = None,
|
|
69
|
+
with_images: bool | None = None,
|
|
70
|
+
) -> int:
|
|
71
|
+
query = "SELECT COUNT(DISTINCT q.id) FROM questoes q"
|
|
72
|
+
parameters: list[object] = []
|
|
73
|
+
if with_images is True:
|
|
74
|
+
query += " JOIN questao_imagens qi ON qi.questao_id = q.id"
|
|
75
|
+
query += " WHERE 1=1"
|
|
76
|
+
if year is not None:
|
|
77
|
+
query += " AND q.ano = ?"
|
|
78
|
+
parameters.append(year)
|
|
79
|
+
if discipline is not None:
|
|
80
|
+
query += " AND q.disciplina = ?"
|
|
81
|
+
parameters.append(discipline)
|
|
82
|
+
if language is not None:
|
|
83
|
+
query += " AND q.idioma = ?"
|
|
84
|
+
parameters.append(language)
|
|
85
|
+
with self._connect() as connection:
|
|
86
|
+
return int(connection.execute(query, parameters).fetchone()[0])
|
|
87
|
+
|
|
88
|
+
def questions(
|
|
89
|
+
self,
|
|
90
|
+
year: int | None = None,
|
|
91
|
+
discipline: str | None = None,
|
|
92
|
+
language: str | None = None,
|
|
93
|
+
with_images: bool | None = None,
|
|
94
|
+
exclude_years: Sequence[int] | None = None,
|
|
95
|
+
limit: int | None = None,
|
|
96
|
+
random_order: bool = False,
|
|
97
|
+
) -> Iterator[Question]:
|
|
98
|
+
query = "SELECT DISTINCT q.* FROM questoes q"
|
|
99
|
+
parameters: list[object] = []
|
|
100
|
+
if with_images is True:
|
|
101
|
+
query += " JOIN questao_imagens qi ON qi.questao_id = q.id"
|
|
102
|
+
query += " WHERE 1=1"
|
|
103
|
+
if year is not None:
|
|
104
|
+
query += " AND q.ano = ?"
|
|
105
|
+
parameters.append(year)
|
|
106
|
+
if discipline is not None:
|
|
107
|
+
query += " AND q.disciplina = ?"
|
|
108
|
+
parameters.append(discipline)
|
|
109
|
+
if language is not None:
|
|
110
|
+
query += " AND q.idioma = ?"
|
|
111
|
+
parameters.append(language)
|
|
112
|
+
if exclude_years:
|
|
113
|
+
placeholders = ",".join("?" for _ in exclude_years)
|
|
114
|
+
query += f" AND q.ano NOT IN ({placeholders})"
|
|
115
|
+
parameters.extend(exclude_years)
|
|
116
|
+
query += " ORDER BY RANDOM()" if random_order else " ORDER BY q.ano, q.numero, q.disciplina, q.idioma"
|
|
117
|
+
if limit is not None:
|
|
118
|
+
if limit < 1:
|
|
119
|
+
return
|
|
120
|
+
query += " LIMIT ?"
|
|
121
|
+
parameters.append(limit)
|
|
122
|
+
with self._connect() as connection:
|
|
123
|
+
for row in connection.execute(query, parameters):
|
|
124
|
+
yield self._question(connection, row)
|
|
125
|
+
|
|
126
|
+
def sample(
|
|
127
|
+
self,
|
|
128
|
+
amount: int,
|
|
129
|
+
discipline: str | None = None,
|
|
130
|
+
year: int | None = None,
|
|
131
|
+
exclude_years: Sequence[int] | None = None,
|
|
132
|
+
seed: int | None = None,
|
|
133
|
+
) -> list[Question]:
|
|
134
|
+
"""Retorna uma amostra reprodutível, útil para montar um simulado."""
|
|
135
|
+
if amount < 1:
|
|
136
|
+
return []
|
|
137
|
+
candidates = list(self.questions(
|
|
138
|
+
year=year, discipline=discipline, exclude_years=exclude_years
|
|
139
|
+
))
|
|
140
|
+
generator = random.Random(seed)
|
|
141
|
+
generator.shuffle(candidates)
|
|
142
|
+
return candidates[:amount]
|
|
143
|
+
|
|
144
|
+
def get_question(
|
|
145
|
+
self,
|
|
146
|
+
year: int,
|
|
147
|
+
number: int,
|
|
148
|
+
discipline: str | None = None,
|
|
149
|
+
language: str | None = None,
|
|
150
|
+
) -> Question | None:
|
|
151
|
+
query = "SELECT * FROM questoes WHERE ano = ? AND numero = ?"
|
|
152
|
+
parameters: list[object] = [year, number]
|
|
153
|
+
if discipline is not None:
|
|
154
|
+
query += " AND disciplina = ?"
|
|
155
|
+
parameters.append(discipline)
|
|
156
|
+
if language is not None:
|
|
157
|
+
query += " AND idioma = ?"
|
|
158
|
+
parameters.append(language)
|
|
159
|
+
query += " ORDER BY idioma LIMIT 1"
|
|
160
|
+
with self._connect() as connection:
|
|
161
|
+
row = connection.execute(query, parameters).fetchone()
|
|
162
|
+
return self._question(connection, row) if row else None
|
|
163
|
+
|
|
164
|
+
def search(self, text: str, limit: int | None = None) -> Iterator[Question]:
|
|
165
|
+
pattern = f"%{text}%"
|
|
166
|
+
query = """
|
|
167
|
+
SELECT * FROM questoes
|
|
168
|
+
WHERE COALESCE(titulo, '') LIKE ?
|
|
169
|
+
OR COALESCE(contexto, '') LIKE ?
|
|
170
|
+
OR COALESCE(texto_completo, '') LIKE ?
|
|
171
|
+
ORDER BY ano, numero
|
|
172
|
+
"""
|
|
173
|
+
parameters: list[object] = [pattern, pattern, pattern]
|
|
174
|
+
if limit is not None:
|
|
175
|
+
query += " LIMIT ?"
|
|
176
|
+
parameters.append(limit)
|
|
177
|
+
with self._connect() as connection:
|
|
178
|
+
for row in connection.execute(query, parameters):
|
|
179
|
+
yield self._question(connection, row)
|
|
180
|
+
|
|
181
|
+
def export_jsonl(self, output_path: str | Path, **filters: object) -> int:
|
|
182
|
+
"""Exporta questões completas para treinamento ou processamento posterior."""
|
|
183
|
+
destination = Path(output_path)
|
|
184
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
185
|
+
count = 0
|
|
186
|
+
with destination.open("w", encoding="utf-8") as output:
|
|
187
|
+
for question in self.questions(**filters):
|
|
188
|
+
record = {
|
|
189
|
+
"id": question.id, "ano": question.year, "numero": question.number,
|
|
190
|
+
"disciplina": question.discipline, "idioma": question.language,
|
|
191
|
+
"titulo": question.title, "contexto": question.context,
|
|
192
|
+
"texto_completo": question.full_text,
|
|
193
|
+
"alternativa_correta": question.correct_alternative,
|
|
194
|
+
"alternativas": [
|
|
195
|
+
{"letra": item.letter, "texto": item.text, "correta": item.correct}
|
|
196
|
+
for item in question.alternatives
|
|
197
|
+
],
|
|
198
|
+
"imagens": [
|
|
199
|
+
{"arquivo": str(item.file), "thumbnail": str(item.thumbnail), "origem": item.origin}
|
|
200
|
+
for item in question.images
|
|
201
|
+
],
|
|
202
|
+
"fonte": question.source,
|
|
203
|
+
}
|
|
204
|
+
output.write(json.dumps(record, ensure_ascii=False) + "\n")
|
|
205
|
+
count += 1
|
|
206
|
+
return count
|
|
207
|
+
|
|
208
|
+
def _question(self, connection: sqlite3.Connection, row: sqlite3.Row) -> Question:
|
|
209
|
+
alternatives = connection.execute(
|
|
210
|
+
"SELECT * FROM alternativas WHERE questao_id = ? ORDER BY letra", (row["id"],)
|
|
211
|
+
).fetchall()
|
|
212
|
+
images = connection.execute(
|
|
213
|
+
"""
|
|
214
|
+
SELECT imagens.*, questao_imagens.origem
|
|
215
|
+
FROM imagens JOIN questao_imagens ON questao_imagens.imagem_id = imagens.id
|
|
216
|
+
WHERE questao_imagens.questao_id = ? ORDER BY imagens.id
|
|
217
|
+
""",
|
|
218
|
+
(row["id"],),
|
|
219
|
+
).fetchall()
|
|
220
|
+
keys = row.keys()
|
|
221
|
+
raw = row["dados_api"] if "dados_api" in keys else None
|
|
222
|
+
try:
|
|
223
|
+
api_data = json.loads(raw) if raw else None
|
|
224
|
+
except json.JSONDecodeError:
|
|
225
|
+
api_data = None
|
|
226
|
+
return Question(
|
|
227
|
+
id=int(row["id"]), year=int(row["ano"]), number=int(row["numero"]),
|
|
228
|
+
discipline=row["disciplina"], language=row["idioma"], title=row["titulo"],
|
|
229
|
+
context=row["contexto"], alternatives_intro=row["introducao_alternativas"],
|
|
230
|
+
correct_alternative=row["alternativa_correta"],
|
|
231
|
+
alternatives=tuple(
|
|
232
|
+
Alternative(
|
|
233
|
+
letter=str(item["letra"]), text=item["texto"],
|
|
234
|
+
file=self._absolute_path(item["arquivo"]), correct=bool(item["correta"])
|
|
235
|
+
) for item in alternatives
|
|
236
|
+
),
|
|
237
|
+
images=tuple(
|
|
238
|
+
ImageAsset(
|
|
239
|
+
id=str(item["id"]), year=int(item["ano"]), discipline=str(item["disciplina"]),
|
|
240
|
+
question_number=int(item["questao"]), file=self._absolute_path(item["arquivo"]),
|
|
241
|
+
thumbnail=self._absolute_path(item["thumbnail"]), sha256=str(item["sha256"]),
|
|
242
|
+
width=int(item["largura"]), height=int(item["altura"]), origin=item["origem"]
|
|
243
|
+
) for item in images
|
|
244
|
+
),
|
|
245
|
+
full_text=row["texto_completo"] if "texto_completo" in keys else row["contexto"],
|
|
246
|
+
source=(row["fonte"] if "fonte" in keys and row["fonte"] else "enem.dev"),
|
|
247
|
+
api_data=api_data,
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
__all__ = ["PlutoLibrary"]
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass(frozen=True)
|
|
7
|
+
class Alternative:
|
|
8
|
+
letter: str
|
|
9
|
+
text: str | None
|
|
10
|
+
file: Path | None
|
|
11
|
+
correct: bool
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class ImageAsset:
|
|
16
|
+
id: str
|
|
17
|
+
year: int
|
|
18
|
+
discipline: str
|
|
19
|
+
question_number: int
|
|
20
|
+
file: Path
|
|
21
|
+
thumbnail: Path
|
|
22
|
+
sha256: str
|
|
23
|
+
width: int
|
|
24
|
+
height: int
|
|
25
|
+
origin: str | None = None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class Question:
|
|
30
|
+
id: int
|
|
31
|
+
year: int
|
|
32
|
+
number: int
|
|
33
|
+
discipline: str | None
|
|
34
|
+
language: str | None
|
|
35
|
+
title: str | None
|
|
36
|
+
context: str | None
|
|
37
|
+
alternatives_intro: str | None
|
|
38
|
+
correct_alternative: str | None
|
|
39
|
+
alternatives: tuple[Alternative, ...]
|
|
40
|
+
images: tuple[ImageAsset, ...]
|
|
41
|
+
full_text: str | None = None
|
|
42
|
+
source: str = "enem.dev"
|
|
43
|
+
api_data: dict[str, Any] | None = None
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def correct_answer(self) -> Alternative | None:
|
|
47
|
+
return next((item for item in self.alternatives if item.correct), None)
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def has_images(self) -> bool:
|
|
51
|
+
return bool(self.images)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
__all__ = ["Alternative", "ImageAsset", "Question"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pluto
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pluto-enem"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Biblioteca Python para questões, alternativas e imagens do ENEM"
|
|
9
|
+
requires-python = ">=3.10"
|
|
10
|
+
|
|
11
|
+
[tool.setuptools.packages.find]
|
|
12
|
+
include = ["pluto*"]
|