examforge 0.2.0.post9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. examforge-0.2.0.post9/PKG-INFO +24 -0
  2. examforge-0.2.0.post9/README.md +79 -0
  3. examforge-0.2.0.post9/app/__init__.py +0 -0
  4. examforge-0.2.0.post9/app/__main__.py +3 -0
  5. examforge-0.2.0.post9/app/assessment/__init__.py +0 -0
  6. examforge-0.2.0.post9/app/assessment/mastery.py +14 -0
  7. examforge-0.2.0.post9/app/assessment/quiz.py +13 -0
  8. examforge-0.2.0.post9/app/audio/__init__.py +0 -0
  9. examforge-0.2.0.post9/app/audio/sync.py +5 -0
  10. examforge-0.2.0.post9/app/blueprints/__init__.py +0 -0
  11. examforge-0.2.0.post9/app/blueprints/engine.py +15 -0
  12. examforge-0.2.0.post9/app/cli.py +186 -0
  13. examforge-0.2.0.post9/app/core/__init__.py +0 -0
  14. examforge-0.2.0.post9/app/core/ids.py +6 -0
  15. examforge-0.2.0.post9/app/core/models.py +85 -0
  16. examforge-0.2.0.post9/app/ingestion/__init__.py +0 -0
  17. examforge-0.2.0.post9/app/ingestion/document.py +92 -0
  18. examforge-0.2.0.post9/app/ingestion/text.py +11 -0
  19. examforge-0.2.0.post9/app/knowledge/__init__.py +0 -0
  20. examforge-0.2.0.post9/app/knowledge/in_memory.py +15 -0
  21. examforge-0.2.0.post9/app/knowledge/ports.py +10 -0
  22. examforge-0.2.0.post9/app/llm/client.py +92 -0
  23. examforge-0.2.0.post9/app/manim/__init__.py +0 -0
  24. examforge-0.2.0.post9/app/manim/compiler.py +17 -0
  25. examforge-0.2.0.post9/app/manim/validation.py +15 -0
  26. examforge-0.2.0.post9/app/video/__init__.py +1 -0
  27. examforge-0.2.0.post9/app/video/pipeline.py +154 -0
  28. examforge-0.2.0.post9/app/visual/__init__.py +0 -0
  29. examforge-0.2.0.post9/app/visual/equation.py +6 -0
  30. examforge-0.2.0.post9/app/visual/graph.py +10 -0
  31. examforge-0.2.0.post9/app/visual/table.py +6 -0
  32. examforge-0.2.0.post9/examforge.egg-info/PKG-INFO +24 -0
  33. examforge-0.2.0.post9/examforge.egg-info/SOURCES.txt +41 -0
  34. examforge-0.2.0.post9/examforge.egg-info/dependency_links.txt +1 -0
  35. examforge-0.2.0.post9/examforge.egg-info/entry_points.txt +2 -0
  36. examforge-0.2.0.post9/examforge.egg-info/requires.txt +24 -0
  37. examforge-0.2.0.post9/examforge.egg-info/top_level.txt +1 -0
  38. examforge-0.2.0.post9/pyproject.toml +34 -0
  39. examforge-0.2.0.post9/setup.cfg +4 -0
  40. examforge-0.2.0.post9/tests/test_cli.py +25 -0
  41. examforge-0.2.0.post9/tests/test_document_video.py +32 -0
  42. examforge-0.2.0.post9/tests/test_ids.py +5 -0
  43. examforge-0.2.0.post9/tests/test_visual.py +14 -0
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: examforge
3
+ Version: 0.2.0.post9
4
+ Summary: CLI-first multimodal exam-preparation engine with a canonical learning IR
5
+ Requires-Python: >=3.12
6
+ Requires-Dist: pydantic<3,>=2.8
7
+ Provides-Extra: document
8
+ Requires-Dist: pymupdf>=1.24; extra == "document"
9
+ Requires-Dist: pymupdf4llm>=0.0.17; extra == "document"
10
+ Requires-Dist: python-docx>=1.1; extra == "document"
11
+ Requires-Dist: pytesseract>=0.3.13; extra == "document"
12
+ Requires-Dist: opencv-python>=4.10; extra == "document"
13
+ Provides-Extra: knowledge
14
+ Requires-Dist: neo4j>=5.25; extra == "knowledge"
15
+ Requires-Dist: qdrant-client>=1.12; extra == "knowledge"
16
+ Provides-Extra: generation
17
+ Requires-Dist: manim>=0.18; extra == "generation"
18
+ Provides-Extra: audio
19
+ Requires-Dist: openai-whisper>=20240930; extra == "audio"
20
+ Provides-Extra: dev
21
+ Requires-Dist: pytest>=8.3; extra == "dev"
22
+ Requires-Dist: httpx>=0.27; extra == "dev"
23
+ Requires-Dist: ruff>=0.6; extra == "dev"
24
+ Requires-Dist: mypy>=1.13; extra == "dev"
@@ -0,0 +1,79 @@
1
+ # ExamForge
2
+
3
+ Vekkam is now a **CLI-first** multimodal exam-preparation engine built around a canonical Learning IR.
4
+
5
+ There is no frontend or web server in the runtime path. The CLI is the user-facing product surface; the domain engine remains importable as a Python package.
6
+
7
+ ## Pipeline
8
+
9
+ ```
10
+ SOURCE MATERIAL -> MULTIMODAL EXTRACTION -> CANONICAL LEARNING IR
11
+ -> KNOWLEDGE GRAPH + VECTOR RETRIEVAL -> VISUAL BLUEPRINT
12
+ -> MANIM COMPILER + VALIDATION -> NARRATION + SYNC -> VIDEO
13
+ -> ACTIVE RECALL + FSRS -> EXAM INTELLIGENCE
14
+ ```
15
+
16
+ ## CLI
17
+
18
+ Install locally:
19
+
20
+ ```bash
21
+ python -m pip install -e ".[dev]"
22
+ ```
23
+
24
+ For PDF/DOCX conversion and Manim rendering:
25
+
26
+ ```bash
27
+ python -m pip install -e ".[document,generation]"
28
+ ```
29
+
30
+ Then:
31
+
32
+ ```bash
33
+ examforge architecture
34
+ examforge llm connect-ollama --model llama3.2:3b
35
+ examforge llm chat "Explain opportunity cost"
36
+ examforge llm connect-api --model gpt-4.1
37
+ examforge ingest-text --file notes.txt
38
+ examforge blueprint --file notes.txt
39
+ examforge questions --file notes.txt --count 4
40
+ examforge video --file ./notes.pdf
41
+ examforge video --file ./chapter.docx --output ./chapter.mp4
42
+ type notes.txt | examforge ingest-text --document-id economics
43
+ ```
44
+
45
+ `video` extracts the PDF/DOCX, sends the extracted learning material through the linked LLM to generate a deterministic Manim scene, validates the returned Python, and renders `ExamForgeScene` to MP4. The generated scene is retained under `.examforge/` beside the requested output.
46
+
47
+ Every command emits JSON so the CLI can be composed with shell scripts, CI, other tools, or future TUI clients.
48
+
49
+ ### Commands
50
+
51
+ - `ingest-text`: text -> canonical Learning IR
52
+ - `blueprint`: text/IR-derived concept -> visual teaching blueprint
53
+ - `questions`: concept -> active-recall question set
54
+ - `video`: PDF/DOCX -> LLM-generated Manim scene -> MP4
55
+ - `review`: mastery state -> next review state
56
+ - `architecture`: inspect the engine pipeline
57
+ - `llm connect-api`: link an OpenAI-compatible API using an environment variable for the key
58
+ - `llm connect-ollama`: link a local Ollama server and installed model
59
+ - `llm status`: inspect the active provider without exposing secrets
60
+ - `llm chat`: send a prompt through the active provider
61
+
62
+ ## Repository layout
63
+
64
+ - `app/core/`: canonical domain models and stable IDs
65
+ - `app/ingestion/`: document/OCR/image extraction
66
+ - `app/visual/`: graph, table and equation understanding
67
+ - `app/knowledge/`: graph and vector-store ports
68
+ - `app/blueprints/`: declarative teaching blueprints
69
+ - `app/manim/`: scene compilation and validation
70
+ - `app/video/`: document-to-Manim video pipeline
71
+ - `app/audio/`: narration and synchronization
72
+ - `app/assessment/`: recall, rescue and mastery
73
+ - `app/cli.py`: product-facing CLI
74
+ - `tests/`: contracts
75
+ - `docs/`: architecture and roadmap
76
+
77
+ LLM integrations remain isolated behind `app/llm/`; provider configuration is stored locally without storing API keys. API keys are read from environment variables at request time.
78
+
79
+ Optional integrations remain isolated behind adapters: PyMuPDF/OCR/OpenCV, Neo4j, Qdrant, Manim, Kokoro/Piper and WhisperX.
File without changes
@@ -0,0 +1,3 @@
1
+ from app.cli import main
2
+
3
+ raise SystemExit(main())
File without changes
@@ -0,0 +1,14 @@
1
+ from datetime import datetime, timedelta
2
+ from app.core.models import MasteryState
3
+
4
+ def schedule_review(state: MasteryState, recalled: bool) -> MasteryState:
5
+ now=datetime.utcnow(); state.last_review=now
6
+ if recalled:
7
+ state.stability=max(1.0,state.stability*1.35+1.0)
8
+ state.retrievability=min(1.0,state.retrievability+0.2)
9
+ else:
10
+ state.stability=max(0.2,state.stability*0.55)
11
+ state.retrievability=max(0.0,state.retrievability-0.3)
12
+ state.next_review=now+timedelta(days=max(1,round(state.stability)))
13
+ state.review_history.append({"at":now.isoformat(),"recalled":recalled})
14
+ return state
@@ -0,0 +1,13 @@
1
+ from app.core.ids import stable_id
2
+ from app.core.models import Concept, Question, QuestionType
3
+
4
+ def generate_questions(concept: Concept, count: int=4) -> list[Question]:
5
+ templates=[
6
+ (QuestionType.recall,f"What is {concept.name}?",concept.description or "State the definition precisely."),
7
+ (QuestionType.conceptual,f"Explain the central idea behind {concept.name}.",concept.description or "Explain it in your own words."),
8
+ (QuestionType.application,f"Give one concrete application of {concept.name}.",concept.examples[0] if concept.examples else "Provide a valid application."),
9
+ (QuestionType.error_detection,f"What is a common mistake when using {concept.name}?","Identify the incorrect assumption and explain why it fails."),
10
+ ]
11
+ return [Question(question_id=stable_id("question",concept.concept_id,str(i)),
12
+ concept_id=concept.concept_id,type=k,prompt=p,answer=a)
13
+ for i,(k,p,a) in enumerate(templates[:count])]
File without changes
@@ -0,0 +1,5 @@
1
+ from app.core.models import SceneAction
2
+
3
+ def build_timeline(actions: list[SceneAction], word_timestamps: list[dict[str,float]]) -> list[dict[str,object]]:
4
+ return [{"start":a.start,"end":a.end,"action":a.action,
5
+ "words":[w for w in word_timestamps if a.start<=w["start"]<=a.end]} for a in actions]
File without changes
@@ -0,0 +1,15 @@
1
+ from app.core.ids import stable_id
2
+ from app.core.models import Concept, SceneAction, VisualBlueprint
3
+
4
+ def build_blueprint(concept: Concept) -> VisualBlueprint:
5
+ scene_type="graph_plot" if concept.visuals else "comparison"
6
+ return VisualBlueprint(
7
+ blueprint_id=stable_id("blueprint", concept.concept_id),
8
+ concept_id=concept.concept_id,
9
+ scene_type=scene_type,
10
+ learning_objective=f"Understand {concept.name}",
11
+ objects=[{"concept_id":concept.concept_id,"name":concept.name}],
12
+ animations=[SceneAction(start=0,end=2,action="introduce_concept")],
13
+ narration_cues=[{"at":0,"text":concept.description[:500]}],
14
+ checkpoint={"type":"active_recall","concept_id":concept.concept_id},
15
+ )
@@ -0,0 +1,186 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from app.assessment.mastery import schedule_review
10
+ from app.assessment.quiz import generate_questions
11
+ from app.blueprints.engine import build_blueprint
12
+ from app.core.models import Concept, MasteryState
13
+ from app.ingestion.text import ingest_text
14
+ from app.llm.client import LLMClient, LLMConfig, config_path, load_config, save_config
15
+ from app.video.pipeline import generate_video
16
+
17
+
18
+ def _emit(value: Any) -> None:
19
+ print(json.dumps(value, indent=2, ensure_ascii=False, default=str))
20
+
21
+
22
+ def _read_text(path: str | None, inline: str | None) -> str:
23
+ if path:
24
+ return Path(path).read_text(encoding="utf-8")
25
+ if inline is not None:
26
+ return inline
27
+ if not sys.stdin.isatty():
28
+ return sys.stdin.read()
29
+ raise SystemExit("Provide --text, --file, or pipe text on stdin.")
30
+
31
+
32
+ def _concept(args: argparse.Namespace) -> Concept:
33
+ document = ingest_text(args.document_id, _read_text(args.file, args.text))
34
+ if not document.concepts:
35
+ raise SystemExit("No concepts were produced.")
36
+ return document.concepts[0]
37
+
38
+
39
+ def _review(args: argparse.Namespace) -> None:
40
+ if not args.state:
41
+ state = MasteryState(concept_id="concept")
42
+ else:
43
+ state = MasteryState.model_validate_json(Path(args.state).read_text(encoding="utf-8"))
44
+ _emit(schedule_review(state, args.recalled).model_dump(mode="json"))
45
+
46
+
47
+ def _video(args: argparse.Namespace) -> None:
48
+ result = generate_video(args.file, args.output)
49
+ _emit(
50
+ {
51
+ "source": str(result.source_path),
52
+ "output": str(result.output_path),
53
+ "scene": str(result.scene_path),
54
+ "document_id": result.document_id,
55
+ "provider": result.provider,
56
+ "model": result.model,
57
+ }
58
+ )
59
+
60
+
61
+ def _llm_connect_api(args: argparse.Namespace) -> None:
62
+ config = LLMConfig(provider="api", endpoint=args.endpoint, model=args.model, api_key_env=args.api_key_env)
63
+ path = save_config(config)
64
+ _emit({"provider": config.provider, "endpoint": config.endpoint, "model": config.model, "api_key_env": config.api_key_env, "config_path": str(path)})
65
+
66
+
67
+ def _llm_connect_ollama(args: argparse.Namespace) -> None:
68
+ config = LLMConfig(provider="ollama", endpoint=args.endpoint, model=args.model)
69
+ path = save_config(config)
70
+ _emit({"provider": config.provider, "endpoint": config.endpoint, "model": config.model, "config_path": str(path)})
71
+
72
+
73
+ def _llm_status(_: argparse.Namespace) -> None:
74
+ config = load_config()
75
+ if config is None:
76
+ _emit({"connected": False, "config_path": str(config_path())})
77
+ return
78
+ _emit({
79
+ "connected": True,
80
+ "provider": config.provider,
81
+ "endpoint": config.endpoint,
82
+ "model": config.model,
83
+ "api_key_env": config.api_key_env,
84
+ "api_key_configured": bool(config.api_key),
85
+ "config_path": str(config_path()),
86
+ })
87
+
88
+
89
+ def _llm_chat(args: argparse.Namespace) -> None:
90
+ config = load_config()
91
+ if config is None:
92
+ raise SystemExit("No LLM linked. Run examforge llm connect-api or examforge llm connect-ollama.")
93
+ response = LLMClient(config).chat(args.prompt, args.system)
94
+ _emit({"provider": config.provider, "model": config.model, "response": response})
95
+
96
+
97
+ def _llm_parser(sub: argparse._SubParsersAction) -> None:
98
+ llm = sub.add_parser("llm", help="Link ExamForge to an LLM provider.")
99
+ llm_sub = llm.add_subparsers(dest="llm_command", required=True)
100
+
101
+ api = llm_sub.add_parser("connect-api", help="Link an OpenAI-compatible LLM API.")
102
+ api.add_argument("--endpoint", default="https://api.openai.com/v1")
103
+ api.add_argument("--model", required=True)
104
+ api.add_argument("--api-key-env", default="OPENAI_API_KEY", help="Environment variable containing the API key.")
105
+ api.set_defaults(handler=_llm_connect_api)
106
+
107
+ ollama = llm_sub.add_parser("connect-ollama", help="Link a local Ollama server.")
108
+ ollama.add_argument("--endpoint", default="http://localhost:11434")
109
+ ollama.add_argument("--model", required=True, help="Installed Ollama model, e.g. llama3.2:3b.")
110
+ ollama.set_defaults(handler=_llm_connect_ollama)
111
+
112
+ status = llm_sub.add_parser("status", help="Show the active LLM connection.")
113
+ status.set_defaults(handler=_llm_status)
114
+
115
+ chat = llm_sub.add_parser("chat", help="Send one prompt through the active LLM.")
116
+ chat.add_argument("prompt")
117
+ chat.add_argument("--system")
118
+ chat.set_defaults(handler=_llm_chat)
119
+
120
+
121
+ def build_parser() -> argparse.ArgumentParser:
122
+ parser = argparse.ArgumentParser(
123
+ prog="examforge",
124
+ description="ExamForge CLI: multimodal exam-preparation engine over canonical Learning IR.",
125
+ )
126
+ sub = parser.add_subparsers(dest="command", required=True)
127
+
128
+ ingest = sub.add_parser("ingest-text", help="Convert text into canonical Learning IR.")
129
+ ingest.add_argument("--document-id", default="document")
130
+ ingest.add_argument("--text")
131
+ ingest.add_argument("--file", help="UTF-8 text file; omit to read stdin.")
132
+ ingest.set_defaults(
133
+ handler=lambda a: _emit(
134
+ ingest_text(a.document_id, _read_text(a.file, a.text)).model_dump(mode="json")
135
+ )
136
+ )
137
+
138
+ blueprint = sub.add_parser("blueprint", help="Build a visual teaching blueprint from text.")
139
+ blueprint.add_argument("--document-id", default="document")
140
+ blueprint.add_argument("--text")
141
+ blueprint.add_argument("--file")
142
+ blueprint.set_defaults(handler=lambda a: _emit(build_blueprint(_concept(a)).model_dump(mode="json")))
143
+
144
+ questions = sub.add_parser("questions", help="Generate deterministic active-recall questions.")
145
+ questions.add_argument("--document-id", default="document")
146
+ questions.add_argument("--text")
147
+ questions.add_argument("--file")
148
+ questions.add_argument("--count", type=int, default=4)
149
+ questions.set_defaults(
150
+ handler=lambda a: _emit(
151
+ [q.model_dump(mode="json") for q in generate_questions(_concept(a), a.count)]
152
+ )
153
+ )
154
+
155
+ video = sub.add_parser("video", help="Convert a PDF or DOCX document into a rendered Manim video.")
156
+ video.add_argument("--file", required=True, help="Path to a .pdf or .docx document.")
157
+ video.add_argument("--output", help="Output .mp4 path; defaults to the input filename with .mp4.")
158
+ video.set_defaults(handler=_video)
159
+
160
+ review = sub.add_parser("review", help="Schedule a mastery review from a JSON state.")
161
+ review.add_argument("--state", help="Path to a MasteryState JSON file.")
162
+ review.add_argument("--recalled", action="store_true", help="Mark the review as successfully recalled.")
163
+ review.set_defaults(handler=_review)
164
+
165
+ _llm_parser(sub)
166
+
167
+ architecture = sub.add_parser("architecture", help="Print the ExamForge pipeline.")
168
+ architecture.set_defaults(
169
+ handler=lambda _: _emit(
170
+ {
171
+ "source_of_truth": "canonical_learning_ir",
172
+ "stages": ["multimodal_extraction", "canonical_learning_ir", "knowledge_graph", "vector_retrieval", "visual_blueprint", "manim_validation", "narration_sync", "active_recall", "mastery", "exam_intelligence", "llm_integration"],
173
+ }
174
+ )
175
+ )
176
+ return parser
177
+
178
+
179
+ def main(argv: list[str] | None = None) -> int:
180
+ args = build_parser().parse_args(argv)
181
+ args.handler(args)
182
+ return 0
183
+
184
+
185
+ if __name__ == "__main__":
186
+ raise SystemExit(main())
File without changes
@@ -0,0 +1,6 @@
1
+ from hashlib import sha256
2
+
3
+
4
+ def stable_id(kind: str, *parts: str) -> str:
5
+ raw = "::".join((kind, *parts))
6
+ return f"{kind}_{sha256(raw.encode('utf-8')).hexdigest()[:16]}"
@@ -0,0 +1,85 @@
1
+ from __future__ import annotations
2
+ from datetime import datetime
3
+ from enum import StrEnum
4
+ from pydantic import BaseModel, Field
5
+
6
+ class VisualType(StrEnum):
7
+ text="text"; equation="equation"; graph="graph"; chart="chart"; table="table"
8
+ diagram="diagram"; flowchart="flowchart"; geometric_figure="geometric_figure"
9
+ photograph="photograph"; illustration="illustration"; unknown="unknown"
10
+
11
+ class Reconstructability(StrEnum):
12
+ reconstructable="RECONSTRUCTABLE"; partial="PARTIALLY_RECONSTRUCTABLE"; none="NON_RECONSTRUCTABLE"
13
+
14
+ class BoundingBox(BaseModel):
15
+ x: float; y: float; width: float; height: float
16
+
17
+ class SourceRegion(BaseModel):
18
+ page_id: str; bbox: BoundingBox | None=None
19
+ extracted_text: str | None=None; artifact_path: str | None=None
20
+
21
+ class VisualObject(BaseModel):
22
+ id: str; type: str; label: str | None=None; direction: str | None=None
23
+ source_region: SourceRegion | None=None
24
+
25
+ class GraphIR(BaseModel):
26
+ visual_id: str; type: str="graph"; axes: dict[str, object]=Field(default_factory=dict)
27
+ objects: list[VisualObject]=Field(default_factory=list)
28
+ intersections: list[dict[str, object]]=Field(default_factory=list)
29
+ annotations: list[dict[str, object]]=Field(default_factory=list)
30
+ semantic_interpretation: dict[str, str]=Field(default_factory=dict)
31
+ reconstructability: Reconstructability=Reconstructability.partial
32
+ confidence: float=0.0
33
+
34
+ class EquationIR(BaseModel):
35
+ latex: str; type: str; variables: list[str]=Field(default_factory=list); operation: str | None=None
36
+
37
+ class TableIR(BaseModel):
38
+ columns: list[str]; rows: list[list[object]]
39
+
40
+ class VisualArtifact(BaseModel):
41
+ visual_id: str; visual_type: VisualType; source: SourceRegion
42
+ graph: GraphIR | None=None; equation: EquationIR | None=None; table: TableIR | None=None
43
+ confidence: float=0.0
44
+
45
+ class PageIR(BaseModel):
46
+ page_id: str; page_number: int; text: str=""
47
+ blocks: list[dict[str, object]]=Field(default_factory=list)
48
+ figures: list[str]=Field(default_factory=list); tables: list[str]=Field(default_factory=list)
49
+ equations: list[str]=Field(default_factory=list); visual_regions: list[VisualArtifact]=Field(default_factory=list)
50
+
51
+ class Concept(BaseModel):
52
+ concept_id: str; name: str; description: str=""
53
+ prerequisites: list[str]=Field(default_factory=list); definitions: list[str]=Field(default_factory=list)
54
+ formulas: list[str]=Field(default_factory=list); examples: list[str]=Field(default_factory=list)
55
+ visuals: list[str]=Field(default_factory=list); evidence: list[SourceRegion]=Field(default_factory=list)
56
+
57
+ class LearningDocument(BaseModel):
58
+ document_id: str; title: str=""; pages: list[PageIR]=Field(default_factory=list)
59
+ concepts: list[Concept]=Field(default_factory=list); created_at: datetime=Field(default_factory=datetime.utcnow)
60
+
61
+ class SceneAction(BaseModel):
62
+ start: float=0.0; end: float=0.0; action: str; payload: dict[str, object]=Field(default_factory=dict)
63
+
64
+ class VisualBlueprint(BaseModel):
65
+ blueprint_id: str; concept_id: str; scene_type: str; learning_objective: str
66
+ objects: list[dict[str, object]]=Field(default_factory=list)
67
+ animations: list[SceneAction]=Field(default_factory=list)
68
+ narration_cues: list[dict[str, object]]=Field(default_factory=list)
69
+ checkpoint: dict[str, object]=Field(default_factory=dict)
70
+
71
+ class QuestionType(StrEnum):
72
+ recall="RECALL"; conceptual="CONCEPTUAL"; calculation="CALCULATION"; application="APPLICATION"
73
+ transfer="TRANSFER"; error_detection="ERROR_DETECTION"; graph_interpretation="GRAPH_INTERPRETATION"
74
+ diagram_interpretation="DIAGRAM_INTERPRETATION"
75
+
76
+ class Question(BaseModel):
77
+ question_id: str; concept_id: str; type: QuestionType; prompt: str; answer: str; explanation: str=""
78
+
79
+ class QuestionRequest(BaseModel):
80
+ concept: Concept; count: int=Field(default=4, ge=1, le=20)
81
+
82
+ class MasteryState(BaseModel):
83
+ concept_id: str; stability: float=0.0; difficulty: float=0.0; retrievability: float=0.0
84
+ last_review: datetime | None=None; next_review: datetime | None=None
85
+ review_history: list[dict[str, object]]=Field(default_factory=list)
File without changes
@@ -0,0 +1,92 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ from app.core.ids import stable_id
6
+ from app.core.models import Concept, LearningDocument, PageIR, SourceRegion
7
+
8
+
9
+ def ingest_document(path: Path) -> LearningDocument:
10
+ """Extract text from a PDF or DOCX while preserving page boundaries where possible."""
11
+ path = path.expanduser().resolve()
12
+ suffix = path.suffix.lower()
13
+ document_id = stable_id("document", str(path))
14
+ title = path.stem
15
+
16
+ if suffix == ".pdf":
17
+ return _ingest_pdf(path, document_id, title)
18
+ if suffix == ".docx":
19
+ return _ingest_docx(path, document_id, title)
20
+ raise ValueError("Unsupported document type. Use .pdf or .docx.")
21
+
22
+
23
+ def _ingest_pdf(path: Path, document_id: str, title: str) -> LearningDocument:
24
+ try:
25
+ import fitz
26
+ except ImportError as exc:
27
+ raise RuntimeError(
28
+ "PDF support is not installed. Install the document extra: "
29
+ "python -m pip install -e '.[document]'"
30
+ ) from exc
31
+
32
+ pages: list[PageIR] = []
33
+ with fitz.open(path) as pdf:
34
+ for number, page in enumerate(pdf, start=1):
35
+ text = page.get_text("text").strip()
36
+ page_id = stable_id("page", document_id, str(number))
37
+ pages.append(
38
+ PageIR(
39
+ page_id=page_id,
40
+ page_number=number,
41
+ text=text,
42
+ blocks=[],
43
+ )
44
+ )
45
+
46
+ return _document(document_id, title, pages)
47
+
48
+
49
+ def _ingest_docx(path: Path, document_id: str, title: str) -> LearningDocument:
50
+ try:
51
+ from docx import Document
52
+ except ImportError as exc:
53
+ raise RuntimeError(
54
+ "DOCX support is not installed. Install the document extra: "
55
+ "python -m pip install -e '.[document]'"
56
+ ) from exc
57
+
58
+ document = Document(path)
59
+ text = "\n".join(
60
+ paragraph.text.strip()
61
+ for paragraph in document.paragraphs
62
+ if paragraph.text.strip()
63
+ )
64
+ page_id = stable_id("page", document_id, "1")
65
+ pages = [PageIR(page_id=page_id, page_number=1, text=text)]
66
+ return _document(document_id, title, pages)
67
+
68
+
69
+ def _document(document_id: str, title: str, pages: list[PageIR]) -> LearningDocument:
70
+ text = "\n\n".join(page.text for page in pages if page.text.strip())
71
+ page_id = pages[0].page_id if pages else stable_id("page", document_id, "1")
72
+ concept = Concept(
73
+ concept_id=stable_id("concept", document_id, text[:160]),
74
+ name=title,
75
+ description=text[:1000],
76
+ evidence=[
77
+ SourceRegion(page_id=page_id, extracted_text=text[:4000])
78
+ ],
79
+ )
80
+ return LearningDocument(
81
+ document_id=document_id,
82
+ title=title,
83
+ pages=pages,
84
+ concepts=[concept] if text else [],
85
+ )
86
+
87
+
88
+ class DocumentIngestor:
89
+ """Compatibility boundary for document adapters."""
90
+
91
+ def ingest(self, path: Path) -> LearningDocument:
92
+ return ingest_document(path)
@@ -0,0 +1,11 @@
1
+ from app.core.ids import stable_id
2
+ from app.core.models import Concept, LearningDocument, PageIR, SourceRegion
3
+
4
+ def ingest_text(document_id: str, text: str) -> LearningDocument:
5
+ page_id = stable_id("page", document_id, "1")
6
+ concept_id = stable_id("concept", document_id, text[:160])
7
+ concept = Concept(concept_id=concept_id, name="Imported concept", description=text[:1000],
8
+ evidence=[SourceRegion(page_id=page_id, extracted_text=text[:4000])])
9
+ return LearningDocument(document_id=document_id, title=document_id,
10
+ pages=[PageIR(page_id=page_id, page_number=1, text=text)],
11
+ concepts=[concept])
File without changes
@@ -0,0 +1,15 @@
1
+ from app.core.models import Concept
2
+
3
+ class InMemoryKnowledgeGraph:
4
+ def __init__(self): self.nodes={}; self.edges=[]
5
+ def upsert_concept(self, concept: Concept) -> None: self.nodes[concept.concept_id]=concept
6
+ def add_edge(self, source: str, relation: str, target: str) -> None: self.edges.append((source,relation,target))
7
+
8
+ class InMemoryVectorStore:
9
+ def __init__(self): self.items=[]
10
+ def upsert(self, document_id: str, text: str, metadata: dict[str,str]) -> None:
11
+ self.items.append({"id":document_id,"text":text,"metadata":metadata})
12
+ def search(self, query: str, limit: int=5) -> list[dict[str,object]]:
13
+ terms=set(query.lower().split())
14
+ ranked=sorted(self.items,key=lambda x:len(terms & set(str(x["text"]).lower().split())),reverse=True)
15
+ return ranked[:limit]
@@ -0,0 +1,10 @@
1
+ from typing import Protocol
2
+ from app.core.models import Concept
3
+
4
+ class KnowledgeGraph(Protocol):
5
+ def upsert_concept(self, concept: Concept) -> None: ...
6
+ def add_edge(self, source: str, relation: str, target: str) -> None: ...
7
+
8
+ class VectorStore(Protocol):
9
+ def upsert(self, document_id: str, text: str, metadata: dict[str,str]) -> None: ...
10
+ def search(self, query: str, limit: int=5) -> list[dict[str,object]]: ...
@@ -0,0 +1,92 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+ from urllib.error import HTTPError, URLError
8
+ from urllib.request import Request, urlopen
9
+
10
+ _CONFIG_DIR = Path.home() / ".config" / "examforge"
11
+ _CONFIG_FILE = _CONFIG_DIR / "llm.json"
12
+
13
+ @dataclass(frozen=True)
14
+ class LLMConfig:
15
+ provider: str
16
+ endpoint: str
17
+ model: str
18
+ api_key_env: str | None = None
19
+
20
+ @property
21
+ def api_key(self) -> str | None:
22
+ return os.getenv(self.api_key_env) if self.api_key_env else None
23
+
24
+ def save_config(config: LLMConfig) -> Path:
25
+ _CONFIG_DIR.mkdir(parents=True, exist_ok=True)
26
+ _CONFIG_FILE.write_text(
27
+ json.dumps(
28
+ {
29
+ "provider": config.provider,
30
+ "endpoint": config.endpoint,
31
+ "model": config.model,
32
+ "api_key_env": config.api_key_env,
33
+ },
34
+ indent=2,
35
+ )
36
+ + "\n",
37
+ encoding="utf-8",
38
+ )
39
+ return _CONFIG_FILE
40
+
41
+ def load_config() -> LLMConfig | None:
42
+ if not _CONFIG_FILE.exists():
43
+ return None
44
+ data = json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
45
+ return LLMConfig(
46
+ provider=data["provider"],
47
+ endpoint=data["endpoint"],
48
+ model=data["model"],
49
+ api_key_env=data.get("api_key_env"),
50
+ )
51
+
52
+ class LLMClient:
53
+ """Small dependency-free client for Ollama and OpenAI-compatible APIs."""
54
+
55
+ def __init__(self, config: LLMConfig):
56
+ self.config = config
57
+
58
+ def _post(self, url: str, payload: dict[str, object], headers: dict[str, str] | None = None) -> dict[str, object]:
59
+ body = json.dumps(payload).encode("utf-8")
60
+ request = Request(url, data=body, method="POST")
61
+ request.add_header("Content-Type", "application/json")
62
+ for key, value in (headers or {}).items():
63
+ request.add_header(key, value)
64
+ try:
65
+ with urlopen(request, timeout=120) as response:
66
+ return json.loads(response.read().decode("utf-8"))
67
+ except HTTPError as exc:
68
+ detail = exc.read().decode("utf-8", errors="replace")
69
+ raise RuntimeError(f"LLM request failed ({exc.code}): {detail}") from exc
70
+ except URLError as exc:
71
+ raise RuntimeError(f"Could not reach LLM at {url}: {exc.reason}") from exc
72
+
73
+ def chat(self, prompt: str, system: str | None = None) -> str:
74
+ messages: list[dict[str, str]] = []
75
+ if system:
76
+ messages.append({"role": "system", "content": system})
77
+ messages.append({"role": "user", "content": prompt})
78
+
79
+ if self.config.provider == "ollama":
80
+ payload = {"model": self.config.model, "messages": messages, "stream": False}
81
+ result = self._post(self.config.endpoint.rstrip("/") + "/api/chat", payload)
82
+ return str(result["message"]["content"])
83
+
84
+ headers = {}
85
+ if self.config.api_key:
86
+ headers["Authorization"] = f"Bearer {self.config.api_key}"
87
+ payload = {"model": self.config.model, "messages": messages, "stream": False}
88
+ result = self._post(self.config.endpoint.rstrip("/") + "/chat/completions", payload, headers)
89
+ return str(result["choices"][0]["message"]["content"])
90
+
91
+ def config_path() -> Path:
92
+ return _CONFIG_FILE
File without changes
@@ -0,0 +1,17 @@
1
+ from dataclasses import dataclass
2
+ from app.core.models import VisualBlueprint
3
+
4
+ @dataclass
5
+ class CompilationResult:
6
+ valid: bool
7
+ source: str
8
+ errors: list[str]
9
+
10
+ class ManimCompiler:
11
+ """Blueprint -> template -> Manim source. The LLM never owns this interface."""
12
+ def compile(self, blueprint: VisualBlueprint) -> CompilationResult:
13
+ source=("from manim import *\n\nclass ExamForgeScene(Scene):\n"
14
+ " def construct(self):\n"
15
+ f" title = Text({blueprint.learning_objective!r})\n"
16
+ " self.play(Write(title))\n self.wait(1)\n")
17
+ return CompilationResult(True, source, [])
@@ -0,0 +1,15 @@
1
+ from dataclasses import dataclass
2
+ from app.core.models import VisualBlueprint
3
+
4
+ @dataclass
5
+ class ValidationReport:
6
+ code_valid: bool
7
+ render_valid: bool
8
+ pedagogical_valid: bool
9
+ errors: list[str]
10
+
11
+ def validate_blueprint(blueprint: VisualBlueprint) -> ValidationReport:
12
+ errors=[]
13
+ if not blueprint.learning_objective.strip(): errors.append("Missing learning objective.")
14
+ if not blueprint.animations: errors.append("Blueprint contains no animation actions.")
15
+ return ValidationReport(not errors, False, not errors, errors)
@@ -0,0 +1 @@
1
+ """Document-to-Manim video generation pipeline."""
@@ -0,0 +1,154 @@
1
+ from __future__ import annotations
2
+
3
+ import ast
4
+ import re
5
+ import subprocess
6
+ import sys
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+
10
+ from app.ingestion.document import ingest_document
11
+ from app.llm.client import LLMClient, load_config
12
+
13
+ _ALLOWED_CLASS_PATTERN = re.compile(r"class\s+ExamForgeScene\s*\(\s*Scene\s*\)")
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class VideoResult:
18
+ source_path: Path
19
+ output_path: Path
20
+ scene_path: Path
21
+ document_id: str
22
+ provider: str
23
+ model: str
24
+
25
+
26
+ def _extract_code(response: str) -> str:
27
+ match = re.search(r"```(?:python)?\s*(.*?)```", response, re.DOTALL | re.IGNORECASE)
28
+ return match.group(1).strip() if match else response.strip()
29
+
30
+
31
+ def _validate_source(source: str) -> None:
32
+ try:
33
+ tree = ast.parse(source)
34
+ except SyntaxError as exc:
35
+ raise RuntimeError(f"LLM returned invalid Python: {exc}") from exc
36
+
37
+ for node in ast.walk(tree):
38
+ if isinstance(node, ast.Import):
39
+ raise RuntimeError("Generated source may only import from manim.")
40
+ if isinstance(node, ast.ImportFrom) and node.module != "manim":
41
+ raise RuntimeError("Generated source may only import from manim.")
42
+ if isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and node.func.id in {
43
+ "eval", "exec", "open", "compile", "__import__", "input"
44
+ }:
45
+ raise RuntimeError(f"Generated source uses forbidden function: {node.func.id}")
46
+ if isinstance(node, ast.Name) and node.id in {
47
+ "os", "sys", "subprocess", "pathlib", "socket", "requests"
48
+ }:
49
+ raise RuntimeError(f"Generated source uses forbidden module/name: {node.id}")
50
+
51
+ if not _ALLOWED_CLASS_PATTERN.search(source):
52
+ raise RuntimeError("Generated Manim source must define an ExamForgeScene(Scene) class.")
53
+
54
+ classes = [node for node in tree.body if isinstance(node, ast.ClassDef)]
55
+ if not any(node.name == "ExamForgeScene" for node in classes):
56
+ raise RuntimeError("Generated source does not contain ExamForgeScene.")
57
+
58
+
59
+ def _prompt(title: str, text: str) -> str:
60
+ return f"""Create a concise educational Manim Community Edition video from the document below.
61
+
62
+ Document title: {title}
63
+
64
+ Requirements:
65
+ - Return ONLY complete Python source code.
66
+ - Import from manim with: from manim import *
67
+ - Define exactly one main scene named ExamForgeScene(Scene).
68
+ - Teach the most important concepts in a clear sequence using Text, MathTex, axes, shapes, arrows, and simple animations where useful.
69
+ - Prefer deterministic, readable Manim primitives over external assets.
70
+ - Keep rendering practical: target roughly 30-90 seconds and avoid expensive simulations.
71
+ - Do not use network access, file I/O, subprocesses, eval, exec, or arbitrary imports.
72
+ - Escape text safely and use MathTex only for valid LaTeX.
73
+ - The scene must run with standard Manim Community Edition.
74
+
75
+ DOCUMENT:
76
+ {text[:30000]}
77
+ """
78
+
79
+
80
+ def generate_video(
81
+ source_path: str | Path,
82
+ output_path: str | Path | None = None,
83
+ ) -> VideoResult:
84
+ source = Path(source_path).expanduser().resolve()
85
+ if not source.exists() or not source.is_file():
86
+ raise FileNotFoundError(f"Input file not found: {source}")
87
+ if source.suffix.lower() not in {".pdf", ".docx"}:
88
+ raise ValueError("Input must be a .pdf or .docx file.")
89
+
90
+ config = load_config()
91
+ if config is None:
92
+ raise RuntimeError(
93
+ "No LLM linked. Run examforge llm connect-api or "
94
+ "examforge llm connect-ollama first."
95
+ )
96
+
97
+ document = ingest_document(source)
98
+ if not document.pages:
99
+ raise RuntimeError("No readable content was extracted from the document.")
100
+
101
+ text = "\n\n".join(page.text for page in document.pages if page.text.strip())
102
+ if not text.strip():
103
+ raise RuntimeError("The document contains no extractable text.")
104
+
105
+ response = LLMClient(config).chat(_prompt(document.title, text))
106
+ manim_source = _extract_code(response)
107
+ _validate_source(manim_source)
108
+
109
+ target = Path(output_path).expanduser() if output_path else source.with_suffix(".mp4")
110
+ target = target.resolve()
111
+ target.parent.mkdir(parents=True, exist_ok=True)
112
+
113
+ scene_dir = target.parent / ".examforge" / source.stem
114
+ scene_dir.mkdir(parents=True, exist_ok=True)
115
+ scene_path = scene_dir / "scene.py"
116
+ scene_path.write_text(manim_source + "\n", encoding="utf-8")
117
+
118
+ command = [
119
+ sys.executable, "-m", "manim", "-q", "m", str(scene_path), "ExamForgeScene",
120
+ "--media_dir", str(scene_dir / "media"), "-o", target.name,
121
+ ]
122
+ try:
123
+ completed = subprocess.run(command, check=True, capture_output=True, text=True)
124
+ except FileNotFoundError as exc:
125
+ raise RuntimeError(
126
+ "Manim is not installed. Install the generation extra: "
127
+ "python -m pip install -e '.[generation]'"
128
+ ) from exc
129
+ except subprocess.CalledProcessError as exc:
130
+ detail = (exc.stderr or exc.stdout or "unknown Manim error").strip()
131
+ raise RuntimeError(f"Manim render failed: {detail}") from exc
132
+
133
+ rendered = scene_dir / "media" / "videos" / "scene" / "720p30" / target.name
134
+ if rendered.exists() and rendered != target:
135
+ target.write_bytes(rendered.read_bytes())
136
+ elif not target.exists():
137
+ candidates = list((scene_dir / "media").rglob(target.name))
138
+ if candidates:
139
+ target.write_bytes(candidates[0].read_bytes())
140
+
141
+ if not target.exists():
142
+ raise RuntimeError(
143
+ "Manim completed without producing the expected video output. "
144
+ f"stdout: {completed.stdout[-1000:]}"
145
+ )
146
+
147
+ return VideoResult(
148
+ source_path=source,
149
+ output_path=target,
150
+ scene_path=scene_path,
151
+ document_id=document.document_id,
152
+ provider=config.provider,
153
+ model=config.model,
154
+ )
File without changes
@@ -0,0 +1,6 @@
1
+ import re
2
+ from app.core.models import EquationIR
3
+
4
+ def parse_equation(latex: str, equation_type: str="unknown", operation: str|None=None) -> EquationIR:
5
+ variables=sorted(set(re.findall(r"(?<![A-Za-z])[A-Za-z](?![A-Za-z])", latex)))
6
+ return EquationIR(latex=latex, type=equation_type, variables=variables, operation=operation)
@@ -0,0 +1,10 @@
1
+ from app.core.models import GraphIR, Reconstructability, VisualObject
2
+
3
+ def interpret_graph(visual_id: str, x_label: str, y_label: str,
4
+ objects: list[tuple[str, str, str]]) -> GraphIR:
5
+ parsed=[VisualObject(id=f"{visual_id}_{label.lower()}", type=kind, label=label, direction=direction)
6
+ for label, kind, direction in objects]
7
+ return GraphIR(visual_id=visual_id,
8
+ axes={"x":{"label":x_label,"scale":"unknown"},"y":{"label":y_label,"scale":"unknown"}},
9
+ objects=parsed, reconstructability=Reconstructability.reconstructable,
10
+ confidence=0.9 if parsed else 0.5)
@@ -0,0 +1,6 @@
1
+ from app.core.models import TableIR
2
+
3
+ def parse_table(columns: list[str], rows: list[list[object]]) -> TableIR:
4
+ if any(len(row)!=len(columns) for row in rows):
5
+ raise ValueError("Every table row must match the column count.")
6
+ return TableIR(columns=columns, rows=rows)
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: examforge
3
+ Version: 0.2.0.post9
4
+ Summary: CLI-first multimodal exam-preparation engine with a canonical learning IR
5
+ Requires-Python: >=3.12
6
+ Requires-Dist: pydantic<3,>=2.8
7
+ Provides-Extra: document
8
+ Requires-Dist: pymupdf>=1.24; extra == "document"
9
+ Requires-Dist: pymupdf4llm>=0.0.17; extra == "document"
10
+ Requires-Dist: python-docx>=1.1; extra == "document"
11
+ Requires-Dist: pytesseract>=0.3.13; extra == "document"
12
+ Requires-Dist: opencv-python>=4.10; extra == "document"
13
+ Provides-Extra: knowledge
14
+ Requires-Dist: neo4j>=5.25; extra == "knowledge"
15
+ Requires-Dist: qdrant-client>=1.12; extra == "knowledge"
16
+ Provides-Extra: generation
17
+ Requires-Dist: manim>=0.18; extra == "generation"
18
+ Provides-Extra: audio
19
+ Requires-Dist: openai-whisper>=20240930; extra == "audio"
20
+ Provides-Extra: dev
21
+ Requires-Dist: pytest>=8.3; extra == "dev"
22
+ Requires-Dist: httpx>=0.27; extra == "dev"
23
+ Requires-Dist: ruff>=0.6; extra == "dev"
24
+ Requires-Dist: mypy>=1.13; extra == "dev"
@@ -0,0 +1,41 @@
1
+ README.md
2
+ pyproject.toml
3
+ app/__init__.py
4
+ app/__main__.py
5
+ app/cli.py
6
+ app/assessment/__init__.py
7
+ app/assessment/mastery.py
8
+ app/assessment/quiz.py
9
+ app/audio/__init__.py
10
+ app/audio/sync.py
11
+ app/blueprints/__init__.py
12
+ app/blueprints/engine.py
13
+ app/core/__init__.py
14
+ app/core/ids.py
15
+ app/core/models.py
16
+ app/ingestion/__init__.py
17
+ app/ingestion/document.py
18
+ app/ingestion/text.py
19
+ app/knowledge/__init__.py
20
+ app/knowledge/in_memory.py
21
+ app/knowledge/ports.py
22
+ app/llm/client.py
23
+ app/manim/__init__.py
24
+ app/manim/compiler.py
25
+ app/manim/validation.py
26
+ app/video/__init__.py
27
+ app/video/pipeline.py
28
+ app/visual/__init__.py
29
+ app/visual/equation.py
30
+ app/visual/graph.py
31
+ app/visual/table.py
32
+ examforge.egg-info/PKG-INFO
33
+ examforge.egg-info/SOURCES.txt
34
+ examforge.egg-info/dependency_links.txt
35
+ examforge.egg-info/entry_points.txt
36
+ examforge.egg-info/requires.txt
37
+ examforge.egg-info/top_level.txt
38
+ tests/test_cli.py
39
+ tests/test_document_video.py
40
+ tests/test_ids.py
41
+ tests/test_visual.py
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ examforge = app.cli:main
@@ -0,0 +1,24 @@
1
+ pydantic<3,>=2.8
2
+
3
+ [audio]
4
+ openai-whisper>=20240930
5
+
6
+ [dev]
7
+ pytest>=8.3
8
+ httpx>=0.27
9
+ ruff>=0.6
10
+ mypy>=1.13
11
+
12
+ [document]
13
+ pymupdf>=1.24
14
+ pymupdf4llm>=0.0.17
15
+ python-docx>=1.1
16
+ pytesseract>=0.3.13
17
+ opencv-python>=4.10
18
+
19
+ [generation]
20
+ manim>=0.18
21
+
22
+ [knowledge]
23
+ neo4j>=5.25
24
+ qdrant-client>=1.12
@@ -0,0 +1,34 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "examforge"
7
+ version = "0.2.0.post9"
8
+ description = "CLI-first multimodal exam-preparation engine with a canonical learning IR"
9
+ requires-python = ">=3.12"
10
+ dependencies = ["pydantic>=2.8,<3"]
11
+
12
+ [project.optional-dependencies]
13
+ document = ["pymupdf>=1.24", "pymupdf4llm>=0.0.17", "python-docx>=1.1", "pytesseract>=0.3.13", "opencv-python>=4.10"]
14
+ knowledge = ["neo4j>=5.25", "qdrant-client>=1.12"]
15
+ generation = ["manim>=0.18"]
16
+ audio = ["openai-whisper>=20240930"]
17
+ dev = ["pytest>=8.3", "httpx>=0.27", "ruff>=0.6", "mypy>=1.13"]
18
+
19
+ [project.scripts]
20
+ examforge = "app.cli:main"
21
+
22
+ [tool.setuptools.packages.find]
23
+ include = ["app*"]
24
+
25
+ [tool.pytest.ini_options]
26
+ testpaths = ["tests"]
27
+
28
+ [tool.ruff]
29
+ line-length = 100
30
+ target-version = "py312"
31
+
32
+ [tool.mypy]
33
+ python_version = "3.12"
34
+ strict = true
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,25 @@
1
+ from app.cli import main
2
+
3
+
4
+ def test_architecture_command(capsys):
5
+ assert main(["architecture"]) == 0
6
+ assert "canonical_learning_ir" in capsys.readouterr().out
7
+
8
+
9
+ def test_ingest_text_command(capsys):
10
+ assert main(["ingest-text", "--document-id", "demo", "--text", "Supply and demand"]) == 0
11
+ output = capsys.readouterr().out
12
+ assert '"document_id": "demo"' in output
13
+
14
+
15
+ def test_llm_connect_ollama(capsys, monkeypatch, tmp_path):
16
+ import app.llm.client as client
17
+
18
+ monkeypatch.setattr(client, "_CONFIG_DIR", tmp_path)
19
+ monkeypatch.setattr(client, "_CONFIG_FILE", tmp_path / "llm.json")
20
+ assert main(["llm", "connect-ollama", "--model", "llama3.2:3b"]) == 0
21
+ output = capsys.readouterr().out
22
+ assert '"provider": "ollama"' in output
23
+ assert '"model": "llama3.2:3b"' in output
24
+ assert main(["llm", "status"]) == 0
25
+ assert '"connected": true' in capsys.readouterr().out
@@ -0,0 +1,32 @@
1
+ from pathlib import Path
2
+
3
+ import pytest
4
+
5
+ from app.ingestion.document import ingest_document
6
+ from app.video.pipeline import _extract_code, _validate_source
7
+
8
+
9
+ def test_extract_code_block():
10
+ source = "```python\nfrom manim import *\nclass ExamForgeScene(Scene):\n pass\n```"
11
+ assert _extract_code(source).startswith("from manim import *")
12
+
13
+
14
+ def test_validate_source_accepts_scene():
15
+ _validate_source("from manim import *\nclass ExamForgeScene(Scene):\n def construct(self):\n pass\n")
16
+
17
+
18
+ def test_validate_source_rejects_invalid_python():
19
+ with pytest.raises(RuntimeError, match="invalid Python"):
20
+ _validate_source("class ExamForgeScene(Scene):\n def construct(")
21
+
22
+
23
+ def test_validate_source_rejects_wrong_scene_name():
24
+ with pytest.raises(RuntimeError, match="ExamForgeScene"):
25
+ _validate_source("from manim import *\nclass OtherScene(Scene):\n pass\n")
26
+
27
+
28
+ def test_ingest_document_rejects_unknown_suffix(tmp_path: Path):
29
+ source = tmp_path / "notes.txt"
30
+ source.write_text("hello", encoding="utf-8")
31
+ with pytest.raises(ValueError, match="\\.pdf or \\.docx"):
32
+ ingest_document(source)
@@ -0,0 +1,5 @@
1
+ from app.core.ids import stable_id
2
+
3
+ def test_stable_ids_are_deterministic():
4
+ assert stable_id("concept","doc","equilibrium")==stable_id("concept","doc","equilibrium")
5
+ assert stable_id("concept","doc","equilibrium")!=stable_id("concept","doc","elasticity")
@@ -0,0 +1,14 @@
1
+ from app.visual.equation import parse_equation
2
+ from app.visual.graph import interpret_graph
3
+ from app.visual.table import parse_table
4
+
5
+ def test_equation_ir():
6
+ eq=parse_equation(r"\frac{d}{dx}x^2 = 2x","derivative","differentiation")
7
+ assert eq.operation=="differentiation" and "x" in eq.variables
8
+
9
+ def test_graph_ir():
10
+ graph=interpret_graph("fig_1","Quantity","Price",[("D","curve","negative"),("S","curve","positive")])
11
+ assert graph.reconstructability.value=="RECONSTRUCTABLE" and len(graph.objects)==2
12
+
13
+ def test_table_ir():
14
+ assert parse_table(["Year","GDP"],[[2024,100]]).rows[0][1]==100