examforge 0.2.0.post9__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
app/__init__.py ADDED
File without changes
app/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from app.cli import main
2
+
3
+ raise SystemExit(main())
File without changes
@@ -0,0 +1,14 @@
1
+ from datetime import datetime, timedelta
2
+ from app.core.models import MasteryState
3
+
4
+ def schedule_review(state: MasteryState, recalled: bool) -> MasteryState:
5
+ now=datetime.utcnow(); state.last_review=now
6
+ if recalled:
7
+ state.stability=max(1.0,state.stability*1.35+1.0)
8
+ state.retrievability=min(1.0,state.retrievability+0.2)
9
+ else:
10
+ state.stability=max(0.2,state.stability*0.55)
11
+ state.retrievability=max(0.0,state.retrievability-0.3)
12
+ state.next_review=now+timedelta(days=max(1,round(state.stability)))
13
+ state.review_history.append({"at":now.isoformat(),"recalled":recalled})
14
+ return state
app/assessment/quiz.py ADDED
@@ -0,0 +1,13 @@
1
+ from app.core.ids import stable_id
2
+ from app.core.models import Concept, Question, QuestionType
3
+
4
+ def generate_questions(concept: Concept, count: int=4) -> list[Question]:
5
+ templates=[
6
+ (QuestionType.recall,f"What is {concept.name}?",concept.description or "State the definition precisely."),
7
+ (QuestionType.conceptual,f"Explain the central idea behind {concept.name}.",concept.description or "Explain it in your own words."),
8
+ (QuestionType.application,f"Give one concrete application of {concept.name}.",concept.examples[0] if concept.examples else "Provide a valid application."),
9
+ (QuestionType.error_detection,f"What is a common mistake when using {concept.name}?","Identify the incorrect assumption and explain why it fails."),
10
+ ]
11
+ return [Question(question_id=stable_id("question",concept.concept_id,str(i)),
12
+ concept_id=concept.concept_id,type=k,prompt=p,answer=a)
13
+ for i,(k,p,a) in enumerate(templates[:count])]
app/audio/__init__.py ADDED
File without changes
app/audio/sync.py ADDED
@@ -0,0 +1,5 @@
1
+ from app.core.models import SceneAction
2
+
3
+ def build_timeline(actions: list[SceneAction], word_timestamps: list[dict[str,float]]) -> list[dict[str,object]]:
4
+ return [{"start":a.start,"end":a.end,"action":a.action,
5
+ "words":[w for w in word_timestamps if a.start<=w["start"]<=a.end]} for a in actions]
File without changes
@@ -0,0 +1,15 @@
1
+ from app.core.ids import stable_id
2
+ from app.core.models import Concept, SceneAction, VisualBlueprint
3
+
4
+ def build_blueprint(concept: Concept) -> VisualBlueprint:
5
+ scene_type="graph_plot" if concept.visuals else "comparison"
6
+ return VisualBlueprint(
7
+ blueprint_id=stable_id("blueprint", concept.concept_id),
8
+ concept_id=concept.concept_id,
9
+ scene_type=scene_type,
10
+ learning_objective=f"Understand {concept.name}",
11
+ objects=[{"concept_id":concept.concept_id,"name":concept.name}],
12
+ animations=[SceneAction(start=0,end=2,action="introduce_concept")],
13
+ narration_cues=[{"at":0,"text":concept.description[:500]}],
14
+ checkpoint={"type":"active_recall","concept_id":concept.concept_id},
15
+ )
app/cli.py ADDED
@@ -0,0 +1,186 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from app.assessment.mastery import schedule_review
10
+ from app.assessment.quiz import generate_questions
11
+ from app.blueprints.engine import build_blueprint
12
+ from app.core.models import Concept, MasteryState
13
+ from app.ingestion.text import ingest_text
14
+ from app.llm.client import LLMClient, LLMConfig, config_path, load_config, save_config
15
+ from app.video.pipeline import generate_video
16
+
17
+
18
+ def _emit(value: Any) -> None:
19
+ print(json.dumps(value, indent=2, ensure_ascii=False, default=str))
20
+
21
+
22
+ def _read_text(path: str | None, inline: str | None) -> str:
23
+ if path:
24
+ return Path(path).read_text(encoding="utf-8")
25
+ if inline is not None:
26
+ return inline
27
+ if not sys.stdin.isatty():
28
+ return sys.stdin.read()
29
+ raise SystemExit("Provide --text, --file, or pipe text on stdin.")
30
+
31
+
32
+ def _concept(args: argparse.Namespace) -> Concept:
33
+ document = ingest_text(args.document_id, _read_text(args.file, args.text))
34
+ if not document.concepts:
35
+ raise SystemExit("No concepts were produced.")
36
+ return document.concepts[0]
37
+
38
+
39
+ def _review(args: argparse.Namespace) -> None:
40
+ if not args.state:
41
+ state = MasteryState(concept_id="concept")
42
+ else:
43
+ state = MasteryState.model_validate_json(Path(args.state).read_text(encoding="utf-8"))
44
+ _emit(schedule_review(state, args.recalled).model_dump(mode="json"))
45
+
46
+
47
+ def _video(args: argparse.Namespace) -> None:
48
+ result = generate_video(args.file, args.output)
49
+ _emit(
50
+ {
51
+ "source": str(result.source_path),
52
+ "output": str(result.output_path),
53
+ "scene": str(result.scene_path),
54
+ "document_id": result.document_id,
55
+ "provider": result.provider,
56
+ "model": result.model,
57
+ }
58
+ )
59
+
60
+
61
+ def _llm_connect_api(args: argparse.Namespace) -> None:
62
+ config = LLMConfig(provider="api", endpoint=args.endpoint, model=args.model, api_key_env=args.api_key_env)
63
+ path = save_config(config)
64
+ _emit({"provider": config.provider, "endpoint": config.endpoint, "model": config.model, "api_key_env": config.api_key_env, "config_path": str(path)})
65
+
66
+
67
+ def _llm_connect_ollama(args: argparse.Namespace) -> None:
68
+ config = LLMConfig(provider="ollama", endpoint=args.endpoint, model=args.model)
69
+ path = save_config(config)
70
+ _emit({"provider": config.provider, "endpoint": config.endpoint, "model": config.model, "config_path": str(path)})
71
+
72
+
73
+ def _llm_status(_: argparse.Namespace) -> None:
74
+ config = load_config()
75
+ if config is None:
76
+ _emit({"connected": False, "config_path": str(config_path())})
77
+ return
78
+ _emit({
79
+ "connected": True,
80
+ "provider": config.provider,
81
+ "endpoint": config.endpoint,
82
+ "model": config.model,
83
+ "api_key_env": config.api_key_env,
84
+ "api_key_configured": bool(config.api_key),
85
+ "config_path": str(config_path()),
86
+ })
87
+
88
+
89
+ def _llm_chat(args: argparse.Namespace) -> None:
90
+ config = load_config()
91
+ if config is None:
92
+ raise SystemExit("No LLM linked. Run examforge llm connect-api or examforge llm connect-ollama.")
93
+ response = LLMClient(config).chat(args.prompt, args.system)
94
+ _emit({"provider": config.provider, "model": config.model, "response": response})
95
+
96
+
97
+ def _llm_parser(sub: argparse._SubParsersAction) -> None:
98
+ llm = sub.add_parser("llm", help="Link ExamForge to an LLM provider.")
99
+ llm_sub = llm.add_subparsers(dest="llm_command", required=True)
100
+
101
+ api = llm_sub.add_parser("connect-api", help="Link an OpenAI-compatible LLM API.")
102
+ api.add_argument("--endpoint", default="https://api.openai.com/v1")
103
+ api.add_argument("--model", required=True)
104
+ api.add_argument("--api-key-env", default="OPENAI_API_KEY", help="Environment variable containing the API key.")
105
+ api.set_defaults(handler=_llm_connect_api)
106
+
107
+ ollama = llm_sub.add_parser("connect-ollama", help="Link a local Ollama server.")
108
+ ollama.add_argument("--endpoint", default="http://localhost:11434")
109
+ ollama.add_argument("--model", required=True, help="Installed Ollama model, e.g. llama3.2:3b.")
110
+ ollama.set_defaults(handler=_llm_connect_ollama)
111
+
112
+ status = llm_sub.add_parser("status", help="Show the active LLM connection.")
113
+ status.set_defaults(handler=_llm_status)
114
+
115
+ chat = llm_sub.add_parser("chat", help="Send one prompt through the active LLM.")
116
+ chat.add_argument("prompt")
117
+ chat.add_argument("--system")
118
+ chat.set_defaults(handler=_llm_chat)
119
+
120
+
121
+ def build_parser() -> argparse.ArgumentParser:
122
+ parser = argparse.ArgumentParser(
123
+ prog="examforge",
124
+ description="ExamForge CLI: multimodal exam-preparation engine over canonical Learning IR.",
125
+ )
126
+ sub = parser.add_subparsers(dest="command", required=True)
127
+
128
+ ingest = sub.add_parser("ingest-text", help="Convert text into canonical Learning IR.")
129
+ ingest.add_argument("--document-id", default="document")
130
+ ingest.add_argument("--text")
131
+ ingest.add_argument("--file", help="UTF-8 text file; omit to read stdin.")
132
+ ingest.set_defaults(
133
+ handler=lambda a: _emit(
134
+ ingest_text(a.document_id, _read_text(a.file, a.text)).model_dump(mode="json")
135
+ )
136
+ )
137
+
138
+ blueprint = sub.add_parser("blueprint", help="Build a visual teaching blueprint from text.")
139
+ blueprint.add_argument("--document-id", default="document")
140
+ blueprint.add_argument("--text")
141
+ blueprint.add_argument("--file")
142
+ blueprint.set_defaults(handler=lambda a: _emit(build_blueprint(_concept(a)).model_dump(mode="json")))
143
+
144
+ questions = sub.add_parser("questions", help="Generate deterministic active-recall questions.")
145
+ questions.add_argument("--document-id", default="document")
146
+ questions.add_argument("--text")
147
+ questions.add_argument("--file")
148
+ questions.add_argument("--count", type=int, default=4)
149
+ questions.set_defaults(
150
+ handler=lambda a: _emit(
151
+ [q.model_dump(mode="json") for q in generate_questions(_concept(a), a.count)]
152
+ )
153
+ )
154
+
155
+ video = sub.add_parser("video", help="Convert a PDF or DOCX document into a rendered Manim video.")
156
+ video.add_argument("--file", required=True, help="Path to a .pdf or .docx document.")
157
+ video.add_argument("--output", help="Output .mp4 path; defaults to the input filename with .mp4.")
158
+ video.set_defaults(handler=_video)
159
+
160
+ review = sub.add_parser("review", help="Schedule a mastery review from a JSON state.")
161
+ review.add_argument("--state", help="Path to a MasteryState JSON file.")
162
+ review.add_argument("--recalled", action="store_true", help="Mark the review as successfully recalled.")
163
+ review.set_defaults(handler=_review)
164
+
165
+ _llm_parser(sub)
166
+
167
+ architecture = sub.add_parser("architecture", help="Print the ExamForge pipeline.")
168
+ architecture.set_defaults(
169
+ handler=lambda _: _emit(
170
+ {
171
+ "source_of_truth": "canonical_learning_ir",
172
+ "stages": ["multimodal_extraction", "canonical_learning_ir", "knowledge_graph", "vector_retrieval", "visual_blueprint", "manim_validation", "narration_sync", "active_recall", "mastery", "exam_intelligence", "llm_integration"],
173
+ }
174
+ )
175
+ )
176
+ return parser
177
+
178
+
179
+ def main(argv: list[str] | None = None) -> int:
180
+ args = build_parser().parse_args(argv)
181
+ args.handler(args)
182
+ return 0
183
+
184
+
185
+ if __name__ == "__main__":
186
+ raise SystemExit(main())
app/core/__init__.py ADDED
File without changes
app/core/ids.py ADDED
@@ -0,0 +1,6 @@
1
+ from hashlib import sha256
2
+
3
+
4
+ def stable_id(kind: str, *parts: str) -> str:
5
+ raw = "::".join((kind, *parts))
6
+ return f"{kind}_{sha256(raw.encode('utf-8')).hexdigest()[:16]}"
app/core/models.py ADDED
@@ -0,0 +1,85 @@
1
+ from __future__ import annotations
2
+ from datetime import datetime
3
+ from enum import StrEnum
4
+ from pydantic import BaseModel, Field
5
+
6
+ class VisualType(StrEnum):
7
+ text="text"; equation="equation"; graph="graph"; chart="chart"; table="table"
8
+ diagram="diagram"; flowchart="flowchart"; geometric_figure="geometric_figure"
9
+ photograph="photograph"; illustration="illustration"; unknown="unknown"
10
+
11
+ class Reconstructability(StrEnum):
12
+ reconstructable="RECONSTRUCTABLE"; partial="PARTIALLY_RECONSTRUCTABLE"; none="NON_RECONSTRUCTABLE"
13
+
14
+ class BoundingBox(BaseModel):
15
+ x: float; y: float; width: float; height: float
16
+
17
+ class SourceRegion(BaseModel):
18
+ page_id: str; bbox: BoundingBox | None=None
19
+ extracted_text: str | None=None; artifact_path: str | None=None
20
+
21
+ class VisualObject(BaseModel):
22
+ id: str; type: str; label: str | None=None; direction: str | None=None
23
+ source_region: SourceRegion | None=None
24
+
25
+ class GraphIR(BaseModel):
26
+ visual_id: str; type: str="graph"; axes: dict[str, object]=Field(default_factory=dict)
27
+ objects: list[VisualObject]=Field(default_factory=list)
28
+ intersections: list[dict[str, object]]=Field(default_factory=list)
29
+ annotations: list[dict[str, object]]=Field(default_factory=list)
30
+ semantic_interpretation: dict[str, str]=Field(default_factory=dict)
31
+ reconstructability: Reconstructability=Reconstructability.partial
32
+ confidence: float=0.0
33
+
34
+ class EquationIR(BaseModel):
35
+ latex: str; type: str; variables: list[str]=Field(default_factory=list); operation: str | None=None
36
+
37
+ class TableIR(BaseModel):
38
+ columns: list[str]; rows: list[list[object]]
39
+
40
+ class VisualArtifact(BaseModel):
41
+ visual_id: str; visual_type: VisualType; source: SourceRegion
42
+ graph: GraphIR | None=None; equation: EquationIR | None=None; table: TableIR | None=None
43
+ confidence: float=0.0
44
+
45
+ class PageIR(BaseModel):
46
+ page_id: str; page_number: int; text: str=""
47
+ blocks: list[dict[str, object]]=Field(default_factory=list)
48
+ figures: list[str]=Field(default_factory=list); tables: list[str]=Field(default_factory=list)
49
+ equations: list[str]=Field(default_factory=list); visual_regions: list[VisualArtifact]=Field(default_factory=list)
50
+
51
+ class Concept(BaseModel):
52
+ concept_id: str; name: str; description: str=""
53
+ prerequisites: list[str]=Field(default_factory=list); definitions: list[str]=Field(default_factory=list)
54
+ formulas: list[str]=Field(default_factory=list); examples: list[str]=Field(default_factory=list)
55
+ visuals: list[str]=Field(default_factory=list); evidence: list[SourceRegion]=Field(default_factory=list)
56
+
57
+ class LearningDocument(BaseModel):
58
+ document_id: str; title: str=""; pages: list[PageIR]=Field(default_factory=list)
59
+ concepts: list[Concept]=Field(default_factory=list); created_at: datetime=Field(default_factory=datetime.utcnow)
60
+
61
+ class SceneAction(BaseModel):
62
+ start: float=0.0; end: float=0.0; action: str; payload: dict[str, object]=Field(default_factory=dict)
63
+
64
+ class VisualBlueprint(BaseModel):
65
+ blueprint_id: str; concept_id: str; scene_type: str; learning_objective: str
66
+ objects: list[dict[str, object]]=Field(default_factory=list)
67
+ animations: list[SceneAction]=Field(default_factory=list)
68
+ narration_cues: list[dict[str, object]]=Field(default_factory=list)
69
+ checkpoint: dict[str, object]=Field(default_factory=dict)
70
+
71
+ class QuestionType(StrEnum):
72
+ recall="RECALL"; conceptual="CONCEPTUAL"; calculation="CALCULATION"; application="APPLICATION"
73
+ transfer="TRANSFER"; error_detection="ERROR_DETECTION"; graph_interpretation="GRAPH_INTERPRETATION"
74
+ diagram_interpretation="DIAGRAM_INTERPRETATION"
75
+
76
+ class Question(BaseModel):
77
+ question_id: str; concept_id: str; type: QuestionType; prompt: str; answer: str; explanation: str=""
78
+
79
+ class QuestionRequest(BaseModel):
80
+ concept: Concept; count: int=Field(default=4, ge=1, le=20)
81
+
82
+ class MasteryState(BaseModel):
83
+ concept_id: str; stability: float=0.0; difficulty: float=0.0; retrievability: float=0.0
84
+ last_review: datetime | None=None; next_review: datetime | None=None
85
+ review_history: list[dict[str, object]]=Field(default_factory=list)
File without changes
@@ -0,0 +1,92 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ from app.core.ids import stable_id
6
+ from app.core.models import Concept, LearningDocument, PageIR, SourceRegion
7
+
8
+
9
+ def ingest_document(path: Path) -> LearningDocument:
10
+ """Extract text from a PDF or DOCX while preserving page boundaries where possible."""
11
+ path = path.expanduser().resolve()
12
+ suffix = path.suffix.lower()
13
+ document_id = stable_id("document", str(path))
14
+ title = path.stem
15
+
16
+ if suffix == ".pdf":
17
+ return _ingest_pdf(path, document_id, title)
18
+ if suffix == ".docx":
19
+ return _ingest_docx(path, document_id, title)
20
+ raise ValueError("Unsupported document type. Use .pdf or .docx.")
21
+
22
+
23
+ def _ingest_pdf(path: Path, document_id: str, title: str) -> LearningDocument:
24
+ try:
25
+ import fitz
26
+ except ImportError as exc:
27
+ raise RuntimeError(
28
+ "PDF support is not installed. Install the document extra: "
29
+ "python -m pip install -e '.[document]'"
30
+ ) from exc
31
+
32
+ pages: list[PageIR] = []
33
+ with fitz.open(path) as pdf:
34
+ for number, page in enumerate(pdf, start=1):
35
+ text = page.get_text("text").strip()
36
+ page_id = stable_id("page", document_id, str(number))
37
+ pages.append(
38
+ PageIR(
39
+ page_id=page_id,
40
+ page_number=number,
41
+ text=text,
42
+ blocks=[],
43
+ )
44
+ )
45
+
46
+ return _document(document_id, title, pages)
47
+
48
+
49
+ def _ingest_docx(path: Path, document_id: str, title: str) -> LearningDocument:
50
+ try:
51
+ from docx import Document
52
+ except ImportError as exc:
53
+ raise RuntimeError(
54
+ "DOCX support is not installed. Install the document extra: "
55
+ "python -m pip install -e '.[document]'"
56
+ ) from exc
57
+
58
+ document = Document(path)
59
+ text = "\n".join(
60
+ paragraph.text.strip()
61
+ for paragraph in document.paragraphs
62
+ if paragraph.text.strip()
63
+ )
64
+ page_id = stable_id("page", document_id, "1")
65
+ pages = [PageIR(page_id=page_id, page_number=1, text=text)]
66
+ return _document(document_id, title, pages)
67
+
68
+
69
+ def _document(document_id: str, title: str, pages: list[PageIR]) -> LearningDocument:
70
+ text = "\n\n".join(page.text for page in pages if page.text.strip())
71
+ page_id = pages[0].page_id if pages else stable_id("page", document_id, "1")
72
+ concept = Concept(
73
+ concept_id=stable_id("concept", document_id, text[:160]),
74
+ name=title,
75
+ description=text[:1000],
76
+ evidence=[
77
+ SourceRegion(page_id=page_id, extracted_text=text[:4000])
78
+ ],
79
+ )
80
+ return LearningDocument(
81
+ document_id=document_id,
82
+ title=title,
83
+ pages=pages,
84
+ concepts=[concept] if text else [],
85
+ )
86
+
87
+
88
+ class DocumentIngestor:
89
+ """Compatibility boundary for document adapters."""
90
+
91
+ def ingest(self, path: Path) -> LearningDocument:
92
+ return ingest_document(path)
app/ingestion/text.py ADDED
@@ -0,0 +1,11 @@
1
+ from app.core.ids import stable_id
2
+ from app.core.models import Concept, LearningDocument, PageIR, SourceRegion
3
+
4
+ def ingest_text(document_id: str, text: str) -> LearningDocument:
5
+ page_id = stable_id("page", document_id, "1")
6
+ concept_id = stable_id("concept", document_id, text[:160])
7
+ concept = Concept(concept_id=concept_id, name="Imported concept", description=text[:1000],
8
+ evidence=[SourceRegion(page_id=page_id, extracted_text=text[:4000])])
9
+ return LearningDocument(document_id=document_id, title=document_id,
10
+ pages=[PageIR(page_id=page_id, page_number=1, text=text)],
11
+ concepts=[concept])
File without changes
@@ -0,0 +1,15 @@
1
+ from app.core.models import Concept
2
+
3
+ class InMemoryKnowledgeGraph:
4
+ def __init__(self): self.nodes={}; self.edges=[]
5
+ def upsert_concept(self, concept: Concept) -> None: self.nodes[concept.concept_id]=concept
6
+ def add_edge(self, source: str, relation: str, target: str) -> None: self.edges.append((source,relation,target))
7
+
8
+ class InMemoryVectorStore:
9
+ def __init__(self): self.items=[]
10
+ def upsert(self, document_id: str, text: str, metadata: dict[str,str]) -> None:
11
+ self.items.append({"id":document_id,"text":text,"metadata":metadata})
12
+ def search(self, query: str, limit: int=5) -> list[dict[str,object]]:
13
+ terms=set(query.lower().split())
14
+ ranked=sorted(self.items,key=lambda x:len(terms & set(str(x["text"]).lower().split())),reverse=True)
15
+ return ranked[:limit]
app/knowledge/ports.py ADDED
@@ -0,0 +1,10 @@
1
+ from typing import Protocol
2
+ from app.core.models import Concept
3
+
4
+ class KnowledgeGraph(Protocol):
5
+ def upsert_concept(self, concept: Concept) -> None: ...
6
+ def add_edge(self, source: str, relation: str, target: str) -> None: ...
7
+
8
+ class VectorStore(Protocol):
9
+ def upsert(self, document_id: str, text: str, metadata: dict[str,str]) -> None: ...
10
+ def search(self, query: str, limit: int=5) -> list[dict[str,object]]: ...
app/llm/client.py ADDED
@@ -0,0 +1,92 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+ from urllib.error import HTTPError, URLError
8
+ from urllib.request import Request, urlopen
9
+
10
+ _CONFIG_DIR = Path.home() / ".config" / "examforge"
11
+ _CONFIG_FILE = _CONFIG_DIR / "llm.json"
12
+
13
+ @dataclass(frozen=True)
14
+ class LLMConfig:
15
+ provider: str
16
+ endpoint: str
17
+ model: str
18
+ api_key_env: str | None = None
19
+
20
+ @property
21
+ def api_key(self) -> str | None:
22
+ return os.getenv(self.api_key_env) if self.api_key_env else None
23
+
24
+ def save_config(config: LLMConfig) -> Path:
25
+ _CONFIG_DIR.mkdir(parents=True, exist_ok=True)
26
+ _CONFIG_FILE.write_text(
27
+ json.dumps(
28
+ {
29
+ "provider": config.provider,
30
+ "endpoint": config.endpoint,
31
+ "model": config.model,
32
+ "api_key_env": config.api_key_env,
33
+ },
34
+ indent=2,
35
+ )
36
+ + "\n",
37
+ encoding="utf-8",
38
+ )
39
+ return _CONFIG_FILE
40
+
41
+ def load_config() -> LLMConfig | None:
42
+ if not _CONFIG_FILE.exists():
43
+ return None
44
+ data = json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
45
+ return LLMConfig(
46
+ provider=data["provider"],
47
+ endpoint=data["endpoint"],
48
+ model=data["model"],
49
+ api_key_env=data.get("api_key_env"),
50
+ )
51
+
52
+ class LLMClient:
53
+ """Small dependency-free client for Ollama and OpenAI-compatible APIs."""
54
+
55
+ def __init__(self, config: LLMConfig):
56
+ self.config = config
57
+
58
+ def _post(self, url: str, payload: dict[str, object], headers: dict[str, str] | None = None) -> dict[str, object]:
59
+ body = json.dumps(payload).encode("utf-8")
60
+ request = Request(url, data=body, method="POST")
61
+ request.add_header("Content-Type", "application/json")
62
+ for key, value in (headers or {}).items():
63
+ request.add_header(key, value)
64
+ try:
65
+ with urlopen(request, timeout=120) as response:
66
+ return json.loads(response.read().decode("utf-8"))
67
+ except HTTPError as exc:
68
+ detail = exc.read().decode("utf-8", errors="replace")
69
+ raise RuntimeError(f"LLM request failed ({exc.code}): {detail}") from exc
70
+ except URLError as exc:
71
+ raise RuntimeError(f"Could not reach LLM at {url}: {exc.reason}") from exc
72
+
73
+ def chat(self, prompt: str, system: str | None = None) -> str:
74
+ messages: list[dict[str, str]] = []
75
+ if system:
76
+ messages.append({"role": "system", "content": system})
77
+ messages.append({"role": "user", "content": prompt})
78
+
79
+ if self.config.provider == "ollama":
80
+ payload = {"model": self.config.model, "messages": messages, "stream": False}
81
+ result = self._post(self.config.endpoint.rstrip("/") + "/api/chat", payload)
82
+ return str(result["message"]["content"])
83
+
84
+ headers = {}
85
+ if self.config.api_key:
86
+ headers["Authorization"] = f"Bearer {self.config.api_key}"
87
+ payload = {"model": self.config.model, "messages": messages, "stream": False}
88
+ result = self._post(self.config.endpoint.rstrip("/") + "/chat/completions", payload, headers)
89
+ return str(result["choices"][0]["message"]["content"])
90
+
91
+ def config_path() -> Path:
92
+ return _CONFIG_FILE
app/manim/__init__.py ADDED
File without changes
app/manim/compiler.py ADDED
@@ -0,0 +1,17 @@
1
+ from dataclasses import dataclass
2
+ from app.core.models import VisualBlueprint
3
+
4
+ @dataclass
5
+ class CompilationResult:
6
+ valid: bool
7
+ source: str
8
+ errors: list[str]
9
+
10
+ class ManimCompiler:
11
+ """Blueprint -> template -> Manim source. The LLM never owns this interface."""
12
+ def compile(self, blueprint: VisualBlueprint) -> CompilationResult:
13
+ source=("from manim import *\n\nclass ExamForgeScene(Scene):\n"
14
+ " def construct(self):\n"
15
+ f" title = Text({blueprint.learning_objective!r})\n"
16
+ " self.play(Write(title))\n self.wait(1)\n")
17
+ return CompilationResult(True, source, [])
@@ -0,0 +1,15 @@
1
+ from dataclasses import dataclass
2
+ from app.core.models import VisualBlueprint
3
+
4
+ @dataclass
5
+ class ValidationReport:
6
+ code_valid: bool
7
+ render_valid: bool
8
+ pedagogical_valid: bool
9
+ errors: list[str]
10
+
11
+ def validate_blueprint(blueprint: VisualBlueprint) -> ValidationReport:
12
+ errors=[]
13
+ if not blueprint.learning_objective.strip(): errors.append("Missing learning objective.")
14
+ if not blueprint.animations: errors.append("Blueprint contains no animation actions.")
15
+ return ValidationReport(not errors, False, not errors, errors)
app/video/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """Document-to-Manim video generation pipeline."""
app/video/pipeline.py ADDED
@@ -0,0 +1,154 @@
1
+ from __future__ import annotations
2
+
3
+ import ast
4
+ import re
5
+ import subprocess
6
+ import sys
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+
10
+ from app.ingestion.document import ingest_document
11
+ from app.llm.client import LLMClient, load_config
12
+
13
+ _ALLOWED_CLASS_PATTERN = re.compile(r"class\s+ExamForgeScene\s*\(\s*Scene\s*\)")
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class VideoResult:
18
+ source_path: Path
19
+ output_path: Path
20
+ scene_path: Path
21
+ document_id: str
22
+ provider: str
23
+ model: str
24
+
25
+
26
+ def _extract_code(response: str) -> str:
27
+ match = re.search(r"```(?:python)?\s*(.*?)```", response, re.DOTALL | re.IGNORECASE)
28
+ return match.group(1).strip() if match else response.strip()
29
+
30
+
31
+ def _validate_source(source: str) -> None:
32
+ try:
33
+ tree = ast.parse(source)
34
+ except SyntaxError as exc:
35
+ raise RuntimeError(f"LLM returned invalid Python: {exc}") from exc
36
+
37
+ for node in ast.walk(tree):
38
+ if isinstance(node, ast.Import):
39
+ raise RuntimeError("Generated source may only import from manim.")
40
+ if isinstance(node, ast.ImportFrom) and node.module != "manim":
41
+ raise RuntimeError("Generated source may only import from manim.")
42
+ if isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and node.func.id in {
43
+ "eval", "exec", "open", "compile", "__import__", "input"
44
+ }:
45
+ raise RuntimeError(f"Generated source uses forbidden function: {node.func.id}")
46
+ if isinstance(node, ast.Name) and node.id in {
47
+ "os", "sys", "subprocess", "pathlib", "socket", "requests"
48
+ }:
49
+ raise RuntimeError(f"Generated source uses forbidden module/name: {node.id}")
50
+
51
+ if not _ALLOWED_CLASS_PATTERN.search(source):
52
+ raise RuntimeError("Generated Manim source must define an ExamForgeScene(Scene) class.")
53
+
54
+ classes = [node for node in tree.body if isinstance(node, ast.ClassDef)]
55
+ if not any(node.name == "ExamForgeScene" for node in classes):
56
+ raise RuntimeError("Generated source does not contain ExamForgeScene.")
57
+
58
+
59
+ def _prompt(title: str, text: str) -> str:
60
+ return f"""Create a concise educational Manim Community Edition video from the document below.
61
+
62
+ Document title: {title}
63
+
64
+ Requirements:
65
+ - Return ONLY complete Python source code.
66
+ - Import from manim with: from manim import *
67
+ - Define exactly one main scene named ExamForgeScene(Scene).
68
+ - Teach the most important concepts in a clear sequence using Text, MathTex, axes, shapes, arrows, and simple animations where useful.
69
+ - Prefer deterministic, readable Manim primitives over external assets.
70
+ - Keep rendering practical: target roughly 30-90 seconds and avoid expensive simulations.
71
+ - Do not use network access, file I/O, subprocesses, eval, exec, or arbitrary imports.
72
+ - Escape text safely and use MathTex only for valid LaTeX.
73
+ - The scene must run with standard Manim Community Edition.
74
+
75
+ DOCUMENT:
76
+ {text[:30000]}
77
+ """
78
+
79
+
80
+ def generate_video(
81
+ source_path: str | Path,
82
+ output_path: str | Path | None = None,
83
+ ) -> VideoResult:
84
+ source = Path(source_path).expanduser().resolve()
85
+ if not source.exists() or not source.is_file():
86
+ raise FileNotFoundError(f"Input file not found: {source}")
87
+ if source.suffix.lower() not in {".pdf", ".docx"}:
88
+ raise ValueError("Input must be a .pdf or .docx file.")
89
+
90
+ config = load_config()
91
+ if config is None:
92
+ raise RuntimeError(
93
+ "No LLM linked. Run examforge llm connect-api or "
94
+ "examforge llm connect-ollama first."
95
+ )
96
+
97
+ document = ingest_document(source)
98
+ if not document.pages:
99
+ raise RuntimeError("No readable content was extracted from the document.")
100
+
101
+ text = "\n\n".join(page.text for page in document.pages if page.text.strip())
102
+ if not text.strip():
103
+ raise RuntimeError("The document contains no extractable text.")
104
+
105
+ response = LLMClient(config).chat(_prompt(document.title, text))
106
+ manim_source = _extract_code(response)
107
+ _validate_source(manim_source)
108
+
109
+ target = Path(output_path).expanduser() if output_path else source.with_suffix(".mp4")
110
+ target = target.resolve()
111
+ target.parent.mkdir(parents=True, exist_ok=True)
112
+
113
+ scene_dir = target.parent / ".examforge" / source.stem
114
+ scene_dir.mkdir(parents=True, exist_ok=True)
115
+ scene_path = scene_dir / "scene.py"
116
+ scene_path.write_text(manim_source + "\n", encoding="utf-8")
117
+
118
+ command = [
119
+ sys.executable, "-m", "manim", "-q", "m", str(scene_path), "ExamForgeScene",
120
+ "--media_dir", str(scene_dir / "media"), "-o", target.name,
121
+ ]
122
+ try:
123
+ completed = subprocess.run(command, check=True, capture_output=True, text=True)
124
+ except FileNotFoundError as exc:
125
+ raise RuntimeError(
126
+ "Manim is not installed. Install the generation extra: "
127
+ "python -m pip install -e '.[generation]'"
128
+ ) from exc
129
+ except subprocess.CalledProcessError as exc:
130
+ detail = (exc.stderr or exc.stdout or "unknown Manim error").strip()
131
+ raise RuntimeError(f"Manim render failed: {detail}") from exc
132
+
133
+ rendered = scene_dir / "media" / "videos" / "scene" / "720p30" / target.name
134
+ if rendered.exists() and rendered != target:
135
+ target.write_bytes(rendered.read_bytes())
136
+ elif not target.exists():
137
+ candidates = list((scene_dir / "media").rglob(target.name))
138
+ if candidates:
139
+ target.write_bytes(candidates[0].read_bytes())
140
+
141
+ if not target.exists():
142
+ raise RuntimeError(
143
+ "Manim completed without producing the expected video output. "
144
+ f"stdout: {completed.stdout[-1000:]}"
145
+ )
146
+
147
+ return VideoResult(
148
+ source_path=source,
149
+ output_path=target,
150
+ scene_path=scene_path,
151
+ document_id=document.document_id,
152
+ provider=config.provider,
153
+ model=config.model,
154
+ )
app/visual/__init__.py ADDED
File without changes
app/visual/equation.py ADDED
@@ -0,0 +1,6 @@
1
+ import re
2
+ from app.core.models import EquationIR
3
+
4
+ def parse_equation(latex: str, equation_type: str="unknown", operation: str|None=None) -> EquationIR:
5
+ variables=sorted(set(re.findall(r"(?<![A-Za-z])[A-Za-z](?![A-Za-z])", latex)))
6
+ return EquationIR(latex=latex, type=equation_type, variables=variables, operation=operation)
app/visual/graph.py ADDED
@@ -0,0 +1,10 @@
1
+ from app.core.models import GraphIR, Reconstructability, VisualObject
2
+
3
+ def interpret_graph(visual_id: str, x_label: str, y_label: str,
4
+ objects: list[tuple[str, str, str]]) -> GraphIR:
5
+ parsed=[VisualObject(id=f"{visual_id}_{label.lower()}", type=kind, label=label, direction=direction)
6
+ for label, kind, direction in objects]
7
+ return GraphIR(visual_id=visual_id,
8
+ axes={"x":{"label":x_label,"scale":"unknown"},"y":{"label":y_label,"scale":"unknown"}},
9
+ objects=parsed, reconstructability=Reconstructability.reconstructable,
10
+ confidence=0.9 if parsed else 0.5)
app/visual/table.py ADDED
@@ -0,0 +1,6 @@
1
+ from app.core.models import TableIR
2
+
3
+ def parse_table(columns: list[str], rows: list[list[object]]) -> TableIR:
4
+ if any(len(row)!=len(columns) for row in rows):
5
+ raise ValueError("Every table row must match the column count.")
6
+ return TableIR(columns=columns, rows=rows)
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: examforge
3
+ Version: 0.2.0.post9
4
+ Summary: CLI-first multimodal exam-preparation engine with a canonical learning IR
5
+ Requires-Python: >=3.12
6
+ Requires-Dist: pydantic<3,>=2.8
7
+ Provides-Extra: document
8
+ Requires-Dist: pymupdf>=1.24; extra == "document"
9
+ Requires-Dist: pymupdf4llm>=0.0.17; extra == "document"
10
+ Requires-Dist: python-docx>=1.1; extra == "document"
11
+ Requires-Dist: pytesseract>=0.3.13; extra == "document"
12
+ Requires-Dist: opencv-python>=4.10; extra == "document"
13
+ Provides-Extra: knowledge
14
+ Requires-Dist: neo4j>=5.25; extra == "knowledge"
15
+ Requires-Dist: qdrant-client>=1.12; extra == "knowledge"
16
+ Provides-Extra: generation
17
+ Requires-Dist: manim>=0.18; extra == "generation"
18
+ Provides-Extra: audio
19
+ Requires-Dist: openai-whisper>=20240930; extra == "audio"
20
+ Provides-Extra: dev
21
+ Requires-Dist: pytest>=8.3; extra == "dev"
22
+ Requires-Dist: httpx>=0.27; extra == "dev"
23
+ Requires-Dist: ruff>=0.6; extra == "dev"
24
+ Requires-Dist: mypy>=1.13; extra == "dev"
@@ -0,0 +1,34 @@
1
+ app/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ app/__main__.py,sha256=BKbRG3z9Q7X0xHXRBEbzvbOr9oLFI7cqsAnCVzwLGk4,51
3
+ app/cli.py,sha256=GhVqU4L1B1RTC576OiZrXB22igtlGIbors4W2FuIqdk,7463
4
+ app/assessment/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ app/assessment/mastery.py,sha256=xO6NcFd_aqyT7pE2mWy4eF1DqMVELRARcvoXiwXWNU4,636
6
+ app/assessment/quiz.py,sha256=5cPYHTmrowtKVh2CRmRN2SHN_590dUlUfUHAki7Huc4,979
7
+ app/audio/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
8
+ app/audio/sync.py,sha256=zAEXlKC0UGvyOCiCsIxM9sylz6R-MnvhncP7MCQLihA,314
9
+ app/blueprints/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
+ app/blueprints/engine.py,sha256=qo6K3Fsd28WE5usj3A4WjzPNWyjx0ZxXD5iZxgkTcUw,745
11
+ app/core/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
12
+ app/core/ids.py,sha256=EEPUnJln8HKl3Dtb-pbaFqtJc7qB5YEJ9eaLOgta3zs,179
13
+ app/core/models.py,sha256=Zzkr0hFqPFCFENP6bPOVQVKcPKeTihsMYWF6VI2J8cc,4123
14
+ app/ingestion/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
15
+ app/ingestion/document.py,sha256=-Z02sB6W7K_khJ8FC-TxYuC3qjFySUa2eYFIgtmkMxc,2987
16
+ app/ingestion/text.py,sha256=EX-c5mvX6jOegE1-5m2yW4-vwbaqvNp_0WbVuYGln74,685
17
+ app/knowledge/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
18
+ app/knowledge/in_memory.py,sha256=mmFJJELGSNEphZeEJBxcPiIT0_brEn7sQt4lv2bOF4g,814
19
+ app/knowledge/ports.py,sha256=-fK-569QUs-KeQaoDVEFjZyoHkCdXik-X5H69vRuTkM,430
20
+ app/llm/client.py,sha256=ndi-jpb690b52Kcs0wDvG25pWxSfWJPAzJZUoEjlR_k,3303
21
+ app/manim/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
22
+ app/manim/compiler.py,sha256=Glfjh07zq1Rhrm8ShVG7_v1y5YHPanwBPRoCSIfbchI,666
23
+ app/manim/validation.py,sha256=VOUCo7THJUrAFpkJiFJ4hJ3pBkfd3gr6e4NKK8SZPo4,547
24
+ app/video/__init__.py,sha256=_foq7s1pIK0ufl3XZ7uuRU2aZQrp6ETJXgzs5vGHS-w,51
25
+ app/video/pipeline.py,sha256=umszNB1fQKdSTgYgCk-PteVYLSd9qJ9DfmXoRM2CTpI,5849
26
+ app/visual/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
27
+ app/visual/equation.py,sha256=fPG8Vr5SZOM8pcv81hRPE9TLWWlRJ3sgZh31YfKZkb4,332
28
+ app/visual/graph.py,sha256=pLg6rVfDeiE7tBN9lZe2_6bjycRRFpcZGn4m2s1DwHQ,651
29
+ app/visual/table.py,sha256=Z1y3XFNVKaskEPkveH0HK_Q8wnHhul2gUufSb-5Skh8,283
30
+ examforge-0.2.0.post9.dist-info/METADATA,sha256=8jkzKMrC0nzTFQqQsKPnbcjWmEXatzQfBss6xUdpzCM,972
31
+ examforge-0.2.0.post9.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
32
+ examforge-0.2.0.post9.dist-info/entry_points.txt,sha256=tXRQjTcdrnQ6RlC6H_q-Td1hZQ0R1Qt6neXWT8Jpamk,43
33
+ examforge-0.2.0.post9.dist-info/top_level.txt,sha256=io9g7LCbfmTG1SFKgEOGXmCFB9uMP2H5lerm0HiHWQE,4
34
+ examforge-0.2.0.post9.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ examforge = app.cli:main
@@ -0,0 +1 @@
1
+ app