examforge 0.2.0.post9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- examforge-0.2.0.post9/PKG-INFO +24 -0
- examforge-0.2.0.post9/README.md +79 -0
- examforge-0.2.0.post9/app/__init__.py +0 -0
- examforge-0.2.0.post9/app/__main__.py +3 -0
- examforge-0.2.0.post9/app/assessment/__init__.py +0 -0
- examforge-0.2.0.post9/app/assessment/mastery.py +14 -0
- examforge-0.2.0.post9/app/assessment/quiz.py +13 -0
- examforge-0.2.0.post9/app/audio/__init__.py +0 -0
- examforge-0.2.0.post9/app/audio/sync.py +5 -0
- examforge-0.2.0.post9/app/blueprints/__init__.py +0 -0
- examforge-0.2.0.post9/app/blueprints/engine.py +15 -0
- examforge-0.2.0.post9/app/cli.py +186 -0
- examforge-0.2.0.post9/app/core/__init__.py +0 -0
- examforge-0.2.0.post9/app/core/ids.py +6 -0
- examforge-0.2.0.post9/app/core/models.py +85 -0
- examforge-0.2.0.post9/app/ingestion/__init__.py +0 -0
- examforge-0.2.0.post9/app/ingestion/document.py +92 -0
- examforge-0.2.0.post9/app/ingestion/text.py +11 -0
- examforge-0.2.0.post9/app/knowledge/__init__.py +0 -0
- examforge-0.2.0.post9/app/knowledge/in_memory.py +15 -0
- examforge-0.2.0.post9/app/knowledge/ports.py +10 -0
- examforge-0.2.0.post9/app/llm/client.py +92 -0
- examforge-0.2.0.post9/app/manim/__init__.py +0 -0
- examforge-0.2.0.post9/app/manim/compiler.py +17 -0
- examforge-0.2.0.post9/app/manim/validation.py +15 -0
- examforge-0.2.0.post9/app/video/__init__.py +1 -0
- examforge-0.2.0.post9/app/video/pipeline.py +154 -0
- examforge-0.2.0.post9/app/visual/__init__.py +0 -0
- examforge-0.2.0.post9/app/visual/equation.py +6 -0
- examforge-0.2.0.post9/app/visual/graph.py +10 -0
- examforge-0.2.0.post9/app/visual/table.py +6 -0
- examforge-0.2.0.post9/examforge.egg-info/PKG-INFO +24 -0
- examforge-0.2.0.post9/examforge.egg-info/SOURCES.txt +41 -0
- examforge-0.2.0.post9/examforge.egg-info/dependency_links.txt +1 -0
- examforge-0.2.0.post9/examforge.egg-info/entry_points.txt +2 -0
- examforge-0.2.0.post9/examforge.egg-info/requires.txt +24 -0
- examforge-0.2.0.post9/examforge.egg-info/top_level.txt +1 -0
- examforge-0.2.0.post9/pyproject.toml +34 -0
- examforge-0.2.0.post9/setup.cfg +4 -0
- examforge-0.2.0.post9/tests/test_cli.py +25 -0
- examforge-0.2.0.post9/tests/test_document_video.py +32 -0
- examforge-0.2.0.post9/tests/test_ids.py +5 -0
- examforge-0.2.0.post9/tests/test_visual.py +14 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: examforge
|
|
3
|
+
Version: 0.2.0.post9
|
|
4
|
+
Summary: CLI-first multimodal exam-preparation engine with a canonical learning IR
|
|
5
|
+
Requires-Python: >=3.12
|
|
6
|
+
Requires-Dist: pydantic<3,>=2.8
|
|
7
|
+
Provides-Extra: document
|
|
8
|
+
Requires-Dist: pymupdf>=1.24; extra == "document"
|
|
9
|
+
Requires-Dist: pymupdf4llm>=0.0.17; extra == "document"
|
|
10
|
+
Requires-Dist: python-docx>=1.1; extra == "document"
|
|
11
|
+
Requires-Dist: pytesseract>=0.3.13; extra == "document"
|
|
12
|
+
Requires-Dist: opencv-python>=4.10; extra == "document"
|
|
13
|
+
Provides-Extra: knowledge
|
|
14
|
+
Requires-Dist: neo4j>=5.25; extra == "knowledge"
|
|
15
|
+
Requires-Dist: qdrant-client>=1.12; extra == "knowledge"
|
|
16
|
+
Provides-Extra: generation
|
|
17
|
+
Requires-Dist: manim>=0.18; extra == "generation"
|
|
18
|
+
Provides-Extra: audio
|
|
19
|
+
Requires-Dist: openai-whisper>=20240930; extra == "audio"
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: pytest>=8.3; extra == "dev"
|
|
22
|
+
Requires-Dist: httpx>=0.27; extra == "dev"
|
|
23
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
24
|
+
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# ExamForge
|
|
2
|
+
|
|
3
|
+
Vekkam is now a **CLI-first** multimodal exam-preparation engine built around a canonical Learning IR.
|
|
4
|
+
|
|
5
|
+
There is no frontend or web server in the runtime path. The CLI is the user-facing product surface; the domain engine remains importable as a Python package.
|
|
6
|
+
|
|
7
|
+
## Pipeline
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
SOURCE MATERIAL -> MULTIMODAL EXTRACTION -> CANONICAL LEARNING IR
|
|
11
|
+
-> KNOWLEDGE GRAPH + VECTOR RETRIEVAL -> VISUAL BLUEPRINT
|
|
12
|
+
-> MANIM COMPILER + VALIDATION -> NARRATION + SYNC -> VIDEO
|
|
13
|
+
-> ACTIVE RECALL + FSRS -> EXAM INTELLIGENCE
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
## CLI
|
|
17
|
+
|
|
18
|
+
Install locally:
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
python -m pip install -e ".[dev]"
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
For PDF/DOCX conversion and Manim rendering:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
python -m pip install -e ".[document,generation]"
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Then:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
examforge architecture
|
|
34
|
+
examforge llm connect-ollama --model llama3.2:3b
|
|
35
|
+
examforge llm chat "Explain opportunity cost"
|
|
36
|
+
examforge llm connect-api --model gpt-4.1
|
|
37
|
+
examforge ingest-text --file notes.txt
|
|
38
|
+
examforge blueprint --file notes.txt
|
|
39
|
+
examforge questions --file notes.txt --count 4
|
|
40
|
+
examforge video --file ./notes.pdf
|
|
41
|
+
examforge video --file ./chapter.docx --output ./chapter.mp4
|
|
42
|
+
type notes.txt | examforge ingest-text --document-id economics
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`video` extracts the PDF/DOCX, sends the extracted learning material through the linked LLM to generate a deterministic Manim scene, validates the returned Python, and renders `ExamForgeScene` to MP4. The generated scene is retained under `.examforge/` beside the requested output.
|
|
46
|
+
|
|
47
|
+
Every command emits JSON so the CLI can be composed with shell scripts, CI, other tools, or future TUI clients.
|
|
48
|
+
|
|
49
|
+
### Commands
|
|
50
|
+
|
|
51
|
+
- `ingest-text`: text -> canonical Learning IR
|
|
52
|
+
- `blueprint`: text/IR-derived concept -> visual teaching blueprint
|
|
53
|
+
- `questions`: concept -> active-recall question set
|
|
54
|
+
- `video`: PDF/DOCX -> LLM-generated Manim scene -> MP4
|
|
55
|
+
- `review`: mastery state -> next review state
|
|
56
|
+
- `architecture`: inspect the engine pipeline
|
|
57
|
+
- `llm connect-api`: link an OpenAI-compatible API using an environment variable for the key
|
|
58
|
+
- `llm connect-ollama`: link a local Ollama server and installed model
|
|
59
|
+
- `llm status`: inspect the active provider without exposing secrets
|
|
60
|
+
- `llm chat`: send a prompt through the active provider
|
|
61
|
+
|
|
62
|
+
## Repository layout
|
|
63
|
+
|
|
64
|
+
- `app/core/`: canonical domain models and stable IDs
|
|
65
|
+
- `app/ingestion/`: document/OCR/image extraction
|
|
66
|
+
- `app/visual/`: graph, table and equation understanding
|
|
67
|
+
- `app/knowledge/`: graph and vector-store ports
|
|
68
|
+
- `app/blueprints/`: declarative teaching blueprints
|
|
69
|
+
- `app/manim/`: scene compilation and validation
|
|
70
|
+
- `app/video/`: document-to-Manim video pipeline
|
|
71
|
+
- `app/audio/`: narration and synchronization
|
|
72
|
+
- `app/assessment/`: recall, rescue and mastery
|
|
73
|
+
- `app/cli.py`: product-facing CLI
|
|
74
|
+
- `tests/`: contracts
|
|
75
|
+
- `docs/`: architecture and roadmap
|
|
76
|
+
|
|
77
|
+
LLM integrations remain isolated behind `app/llm/`; provider configuration is stored locally without storing API keys. API keys are read from environment variables at request time.
|
|
78
|
+
|
|
79
|
+
Optional integrations remain isolated behind adapters: PyMuPDF/OCR/OpenCV, Neo4j, Qdrant, Manim, Kokoro/Piper and WhisperX.
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from datetime import datetime, timedelta
|
|
2
|
+
from app.core.models import MasteryState
|
|
3
|
+
|
|
4
|
+
def schedule_review(state: MasteryState, recalled: bool) -> MasteryState:
|
|
5
|
+
now=datetime.utcnow(); state.last_review=now
|
|
6
|
+
if recalled:
|
|
7
|
+
state.stability=max(1.0,state.stability*1.35+1.0)
|
|
8
|
+
state.retrievability=min(1.0,state.retrievability+0.2)
|
|
9
|
+
else:
|
|
10
|
+
state.stability=max(0.2,state.stability*0.55)
|
|
11
|
+
state.retrievability=max(0.0,state.retrievability-0.3)
|
|
12
|
+
state.next_review=now+timedelta(days=max(1,round(state.stability)))
|
|
13
|
+
state.review_history.append({"at":now.isoformat(),"recalled":recalled})
|
|
14
|
+
return state
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
from app.core.ids import stable_id
|
|
2
|
+
from app.core.models import Concept, Question, QuestionType
|
|
3
|
+
|
|
4
|
+
def generate_questions(concept: Concept, count: int=4) -> list[Question]:
|
|
5
|
+
templates=[
|
|
6
|
+
(QuestionType.recall,f"What is {concept.name}?",concept.description or "State the definition precisely."),
|
|
7
|
+
(QuestionType.conceptual,f"Explain the central idea behind {concept.name}.",concept.description or "Explain it in your own words."),
|
|
8
|
+
(QuestionType.application,f"Give one concrete application of {concept.name}.",concept.examples[0] if concept.examples else "Provide a valid application."),
|
|
9
|
+
(QuestionType.error_detection,f"What is a common mistake when using {concept.name}?","Identify the incorrect assumption and explain why it fails."),
|
|
10
|
+
]
|
|
11
|
+
return [Question(question_id=stable_id("question",concept.concept_id,str(i)),
|
|
12
|
+
concept_id=concept.concept_id,type=k,prompt=p,answer=a)
|
|
13
|
+
for i,(k,p,a) in enumerate(templates[:count])]
|
|
File without changes
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
from app.core.models import SceneAction
|
|
2
|
+
|
|
3
|
+
def build_timeline(actions: list[SceneAction], word_timestamps: list[dict[str,float]]) -> list[dict[str,object]]:
|
|
4
|
+
return [{"start":a.start,"end":a.end,"action":a.action,
|
|
5
|
+
"words":[w for w in word_timestamps if a.start<=w["start"]<=a.end]} for a in actions]
|
|
File without changes
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from app.core.ids import stable_id
|
|
2
|
+
from app.core.models import Concept, SceneAction, VisualBlueprint
|
|
3
|
+
|
|
4
|
+
def build_blueprint(concept: Concept) -> VisualBlueprint:
|
|
5
|
+
scene_type="graph_plot" if concept.visuals else "comparison"
|
|
6
|
+
return VisualBlueprint(
|
|
7
|
+
blueprint_id=stable_id("blueprint", concept.concept_id),
|
|
8
|
+
concept_id=concept.concept_id,
|
|
9
|
+
scene_type=scene_type,
|
|
10
|
+
learning_objective=f"Understand {concept.name}",
|
|
11
|
+
objects=[{"concept_id":concept.concept_id,"name":concept.name}],
|
|
12
|
+
animations=[SceneAction(start=0,end=2,action="introduce_concept")],
|
|
13
|
+
narration_cues=[{"at":0,"text":concept.description[:500]}],
|
|
14
|
+
checkpoint={"type":"active_recall","concept_id":concept.concept_id},
|
|
15
|
+
)
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from app.assessment.mastery import schedule_review
|
|
10
|
+
from app.assessment.quiz import generate_questions
|
|
11
|
+
from app.blueprints.engine import build_blueprint
|
|
12
|
+
from app.core.models import Concept, MasteryState
|
|
13
|
+
from app.ingestion.text import ingest_text
|
|
14
|
+
from app.llm.client import LLMClient, LLMConfig, config_path, load_config, save_config
|
|
15
|
+
from app.video.pipeline import generate_video
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _emit(value: Any) -> None:
|
|
19
|
+
print(json.dumps(value, indent=2, ensure_ascii=False, default=str))
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _read_text(path: str | None, inline: str | None) -> str:
|
|
23
|
+
if path:
|
|
24
|
+
return Path(path).read_text(encoding="utf-8")
|
|
25
|
+
if inline is not None:
|
|
26
|
+
return inline
|
|
27
|
+
if not sys.stdin.isatty():
|
|
28
|
+
return sys.stdin.read()
|
|
29
|
+
raise SystemExit("Provide --text, --file, or pipe text on stdin.")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _concept(args: argparse.Namespace) -> Concept:
|
|
33
|
+
document = ingest_text(args.document_id, _read_text(args.file, args.text))
|
|
34
|
+
if not document.concepts:
|
|
35
|
+
raise SystemExit("No concepts were produced.")
|
|
36
|
+
return document.concepts[0]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _review(args: argparse.Namespace) -> None:
|
|
40
|
+
if not args.state:
|
|
41
|
+
state = MasteryState(concept_id="concept")
|
|
42
|
+
else:
|
|
43
|
+
state = MasteryState.model_validate_json(Path(args.state).read_text(encoding="utf-8"))
|
|
44
|
+
_emit(schedule_review(state, args.recalled).model_dump(mode="json"))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _video(args: argparse.Namespace) -> None:
|
|
48
|
+
result = generate_video(args.file, args.output)
|
|
49
|
+
_emit(
|
|
50
|
+
{
|
|
51
|
+
"source": str(result.source_path),
|
|
52
|
+
"output": str(result.output_path),
|
|
53
|
+
"scene": str(result.scene_path),
|
|
54
|
+
"document_id": result.document_id,
|
|
55
|
+
"provider": result.provider,
|
|
56
|
+
"model": result.model,
|
|
57
|
+
}
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _llm_connect_api(args: argparse.Namespace) -> None:
|
|
62
|
+
config = LLMConfig(provider="api", endpoint=args.endpoint, model=args.model, api_key_env=args.api_key_env)
|
|
63
|
+
path = save_config(config)
|
|
64
|
+
_emit({"provider": config.provider, "endpoint": config.endpoint, "model": config.model, "api_key_env": config.api_key_env, "config_path": str(path)})
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _llm_connect_ollama(args: argparse.Namespace) -> None:
|
|
68
|
+
config = LLMConfig(provider="ollama", endpoint=args.endpoint, model=args.model)
|
|
69
|
+
path = save_config(config)
|
|
70
|
+
_emit({"provider": config.provider, "endpoint": config.endpoint, "model": config.model, "config_path": str(path)})
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _llm_status(_: argparse.Namespace) -> None:
|
|
74
|
+
config = load_config()
|
|
75
|
+
if config is None:
|
|
76
|
+
_emit({"connected": False, "config_path": str(config_path())})
|
|
77
|
+
return
|
|
78
|
+
_emit({
|
|
79
|
+
"connected": True,
|
|
80
|
+
"provider": config.provider,
|
|
81
|
+
"endpoint": config.endpoint,
|
|
82
|
+
"model": config.model,
|
|
83
|
+
"api_key_env": config.api_key_env,
|
|
84
|
+
"api_key_configured": bool(config.api_key),
|
|
85
|
+
"config_path": str(config_path()),
|
|
86
|
+
})
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _llm_chat(args: argparse.Namespace) -> None:
|
|
90
|
+
config = load_config()
|
|
91
|
+
if config is None:
|
|
92
|
+
raise SystemExit("No LLM linked. Run examforge llm connect-api or examforge llm connect-ollama.")
|
|
93
|
+
response = LLMClient(config).chat(args.prompt, args.system)
|
|
94
|
+
_emit({"provider": config.provider, "model": config.model, "response": response})
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _llm_parser(sub: argparse._SubParsersAction) -> None:
|
|
98
|
+
llm = sub.add_parser("llm", help="Link ExamForge to an LLM provider.")
|
|
99
|
+
llm_sub = llm.add_subparsers(dest="llm_command", required=True)
|
|
100
|
+
|
|
101
|
+
api = llm_sub.add_parser("connect-api", help="Link an OpenAI-compatible LLM API.")
|
|
102
|
+
api.add_argument("--endpoint", default="https://api.openai.com/v1")
|
|
103
|
+
api.add_argument("--model", required=True)
|
|
104
|
+
api.add_argument("--api-key-env", default="OPENAI_API_KEY", help="Environment variable containing the API key.")
|
|
105
|
+
api.set_defaults(handler=_llm_connect_api)
|
|
106
|
+
|
|
107
|
+
ollama = llm_sub.add_parser("connect-ollama", help="Link a local Ollama server.")
|
|
108
|
+
ollama.add_argument("--endpoint", default="http://localhost:11434")
|
|
109
|
+
ollama.add_argument("--model", required=True, help="Installed Ollama model, e.g. llama3.2:3b.")
|
|
110
|
+
ollama.set_defaults(handler=_llm_connect_ollama)
|
|
111
|
+
|
|
112
|
+
status = llm_sub.add_parser("status", help="Show the active LLM connection.")
|
|
113
|
+
status.set_defaults(handler=_llm_status)
|
|
114
|
+
|
|
115
|
+
chat = llm_sub.add_parser("chat", help="Send one prompt through the active LLM.")
|
|
116
|
+
chat.add_argument("prompt")
|
|
117
|
+
chat.add_argument("--system")
|
|
118
|
+
chat.set_defaults(handler=_llm_chat)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
122
|
+
parser = argparse.ArgumentParser(
|
|
123
|
+
prog="examforge",
|
|
124
|
+
description="ExamForge CLI: multimodal exam-preparation engine over canonical Learning IR.",
|
|
125
|
+
)
|
|
126
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
127
|
+
|
|
128
|
+
ingest = sub.add_parser("ingest-text", help="Convert text into canonical Learning IR.")
|
|
129
|
+
ingest.add_argument("--document-id", default="document")
|
|
130
|
+
ingest.add_argument("--text")
|
|
131
|
+
ingest.add_argument("--file", help="UTF-8 text file; omit to read stdin.")
|
|
132
|
+
ingest.set_defaults(
|
|
133
|
+
handler=lambda a: _emit(
|
|
134
|
+
ingest_text(a.document_id, _read_text(a.file, a.text)).model_dump(mode="json")
|
|
135
|
+
)
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
blueprint = sub.add_parser("blueprint", help="Build a visual teaching blueprint from text.")
|
|
139
|
+
blueprint.add_argument("--document-id", default="document")
|
|
140
|
+
blueprint.add_argument("--text")
|
|
141
|
+
blueprint.add_argument("--file")
|
|
142
|
+
blueprint.set_defaults(handler=lambda a: _emit(build_blueprint(_concept(a)).model_dump(mode="json")))
|
|
143
|
+
|
|
144
|
+
questions = sub.add_parser("questions", help="Generate deterministic active-recall questions.")
|
|
145
|
+
questions.add_argument("--document-id", default="document")
|
|
146
|
+
questions.add_argument("--text")
|
|
147
|
+
questions.add_argument("--file")
|
|
148
|
+
questions.add_argument("--count", type=int, default=4)
|
|
149
|
+
questions.set_defaults(
|
|
150
|
+
handler=lambda a: _emit(
|
|
151
|
+
[q.model_dump(mode="json") for q in generate_questions(_concept(a), a.count)]
|
|
152
|
+
)
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
video = sub.add_parser("video", help="Convert a PDF or DOCX document into a rendered Manim video.")
|
|
156
|
+
video.add_argument("--file", required=True, help="Path to a .pdf or .docx document.")
|
|
157
|
+
video.add_argument("--output", help="Output .mp4 path; defaults to the input filename with .mp4.")
|
|
158
|
+
video.set_defaults(handler=_video)
|
|
159
|
+
|
|
160
|
+
review = sub.add_parser("review", help="Schedule a mastery review from a JSON state.")
|
|
161
|
+
review.add_argument("--state", help="Path to a MasteryState JSON file.")
|
|
162
|
+
review.add_argument("--recalled", action="store_true", help="Mark the review as successfully recalled.")
|
|
163
|
+
review.set_defaults(handler=_review)
|
|
164
|
+
|
|
165
|
+
_llm_parser(sub)
|
|
166
|
+
|
|
167
|
+
architecture = sub.add_parser("architecture", help="Print the ExamForge pipeline.")
|
|
168
|
+
architecture.set_defaults(
|
|
169
|
+
handler=lambda _: _emit(
|
|
170
|
+
{
|
|
171
|
+
"source_of_truth": "canonical_learning_ir",
|
|
172
|
+
"stages": ["multimodal_extraction", "canonical_learning_ir", "knowledge_graph", "vector_retrieval", "visual_blueprint", "manim_validation", "narration_sync", "active_recall", "mastery", "exam_intelligence", "llm_integration"],
|
|
173
|
+
}
|
|
174
|
+
)
|
|
175
|
+
)
|
|
176
|
+
return parser
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def main(argv: list[str] | None = None) -> int:
|
|
180
|
+
args = build_parser().parse_args(argv)
|
|
181
|
+
args.handler(args)
|
|
182
|
+
return 0
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
if __name__ == "__main__":
|
|
186
|
+
raise SystemExit(main())
|
|
File without changes
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from enum import StrEnum
|
|
4
|
+
from pydantic import BaseModel, Field
|
|
5
|
+
|
|
6
|
+
class VisualType(StrEnum):
|
|
7
|
+
text="text"; equation="equation"; graph="graph"; chart="chart"; table="table"
|
|
8
|
+
diagram="diagram"; flowchart="flowchart"; geometric_figure="geometric_figure"
|
|
9
|
+
photograph="photograph"; illustration="illustration"; unknown="unknown"
|
|
10
|
+
|
|
11
|
+
class Reconstructability(StrEnum):
|
|
12
|
+
reconstructable="RECONSTRUCTABLE"; partial="PARTIALLY_RECONSTRUCTABLE"; none="NON_RECONSTRUCTABLE"
|
|
13
|
+
|
|
14
|
+
class BoundingBox(BaseModel):
|
|
15
|
+
x: float; y: float; width: float; height: float
|
|
16
|
+
|
|
17
|
+
class SourceRegion(BaseModel):
|
|
18
|
+
page_id: str; bbox: BoundingBox | None=None
|
|
19
|
+
extracted_text: str | None=None; artifact_path: str | None=None
|
|
20
|
+
|
|
21
|
+
class VisualObject(BaseModel):
|
|
22
|
+
id: str; type: str; label: str | None=None; direction: str | None=None
|
|
23
|
+
source_region: SourceRegion | None=None
|
|
24
|
+
|
|
25
|
+
class GraphIR(BaseModel):
|
|
26
|
+
visual_id: str; type: str="graph"; axes: dict[str, object]=Field(default_factory=dict)
|
|
27
|
+
objects: list[VisualObject]=Field(default_factory=list)
|
|
28
|
+
intersections: list[dict[str, object]]=Field(default_factory=list)
|
|
29
|
+
annotations: list[dict[str, object]]=Field(default_factory=list)
|
|
30
|
+
semantic_interpretation: dict[str, str]=Field(default_factory=dict)
|
|
31
|
+
reconstructability: Reconstructability=Reconstructability.partial
|
|
32
|
+
confidence: float=0.0
|
|
33
|
+
|
|
34
|
+
class EquationIR(BaseModel):
|
|
35
|
+
latex: str; type: str; variables: list[str]=Field(default_factory=list); operation: str | None=None
|
|
36
|
+
|
|
37
|
+
class TableIR(BaseModel):
|
|
38
|
+
columns: list[str]; rows: list[list[object]]
|
|
39
|
+
|
|
40
|
+
class VisualArtifact(BaseModel):
|
|
41
|
+
visual_id: str; visual_type: VisualType; source: SourceRegion
|
|
42
|
+
graph: GraphIR | None=None; equation: EquationIR | None=None; table: TableIR | None=None
|
|
43
|
+
confidence: float=0.0
|
|
44
|
+
|
|
45
|
+
class PageIR(BaseModel):
|
|
46
|
+
page_id: str; page_number: int; text: str=""
|
|
47
|
+
blocks: list[dict[str, object]]=Field(default_factory=list)
|
|
48
|
+
figures: list[str]=Field(default_factory=list); tables: list[str]=Field(default_factory=list)
|
|
49
|
+
equations: list[str]=Field(default_factory=list); visual_regions: list[VisualArtifact]=Field(default_factory=list)
|
|
50
|
+
|
|
51
|
+
class Concept(BaseModel):
|
|
52
|
+
concept_id: str; name: str; description: str=""
|
|
53
|
+
prerequisites: list[str]=Field(default_factory=list); definitions: list[str]=Field(default_factory=list)
|
|
54
|
+
formulas: list[str]=Field(default_factory=list); examples: list[str]=Field(default_factory=list)
|
|
55
|
+
visuals: list[str]=Field(default_factory=list); evidence: list[SourceRegion]=Field(default_factory=list)
|
|
56
|
+
|
|
57
|
+
class LearningDocument(BaseModel):
|
|
58
|
+
document_id: str; title: str=""; pages: list[PageIR]=Field(default_factory=list)
|
|
59
|
+
concepts: list[Concept]=Field(default_factory=list); created_at: datetime=Field(default_factory=datetime.utcnow)
|
|
60
|
+
|
|
61
|
+
class SceneAction(BaseModel):
|
|
62
|
+
start: float=0.0; end: float=0.0; action: str; payload: dict[str, object]=Field(default_factory=dict)
|
|
63
|
+
|
|
64
|
+
class VisualBlueprint(BaseModel):
|
|
65
|
+
blueprint_id: str; concept_id: str; scene_type: str; learning_objective: str
|
|
66
|
+
objects: list[dict[str, object]]=Field(default_factory=list)
|
|
67
|
+
animations: list[SceneAction]=Field(default_factory=list)
|
|
68
|
+
narration_cues: list[dict[str, object]]=Field(default_factory=list)
|
|
69
|
+
checkpoint: dict[str, object]=Field(default_factory=dict)
|
|
70
|
+
|
|
71
|
+
class QuestionType(StrEnum):
|
|
72
|
+
recall="RECALL"; conceptual="CONCEPTUAL"; calculation="CALCULATION"; application="APPLICATION"
|
|
73
|
+
transfer="TRANSFER"; error_detection="ERROR_DETECTION"; graph_interpretation="GRAPH_INTERPRETATION"
|
|
74
|
+
diagram_interpretation="DIAGRAM_INTERPRETATION"
|
|
75
|
+
|
|
76
|
+
class Question(BaseModel):
|
|
77
|
+
question_id: str; concept_id: str; type: QuestionType; prompt: str; answer: str; explanation: str=""
|
|
78
|
+
|
|
79
|
+
class QuestionRequest(BaseModel):
|
|
80
|
+
concept: Concept; count: int=Field(default=4, ge=1, le=20)
|
|
81
|
+
|
|
82
|
+
class MasteryState(BaseModel):
|
|
83
|
+
concept_id: str; stability: float=0.0; difficulty: float=0.0; retrievability: float=0.0
|
|
84
|
+
last_review: datetime | None=None; next_review: datetime | None=None
|
|
85
|
+
review_history: list[dict[str, object]]=Field(default_factory=list)
|
|
File without changes
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from app.core.ids import stable_id
|
|
6
|
+
from app.core.models import Concept, LearningDocument, PageIR, SourceRegion
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def ingest_document(path: Path) -> LearningDocument:
|
|
10
|
+
"""Extract text from a PDF or DOCX while preserving page boundaries where possible."""
|
|
11
|
+
path = path.expanduser().resolve()
|
|
12
|
+
suffix = path.suffix.lower()
|
|
13
|
+
document_id = stable_id("document", str(path))
|
|
14
|
+
title = path.stem
|
|
15
|
+
|
|
16
|
+
if suffix == ".pdf":
|
|
17
|
+
return _ingest_pdf(path, document_id, title)
|
|
18
|
+
if suffix == ".docx":
|
|
19
|
+
return _ingest_docx(path, document_id, title)
|
|
20
|
+
raise ValueError("Unsupported document type. Use .pdf or .docx.")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _ingest_pdf(path: Path, document_id: str, title: str) -> LearningDocument:
|
|
24
|
+
try:
|
|
25
|
+
import fitz
|
|
26
|
+
except ImportError as exc:
|
|
27
|
+
raise RuntimeError(
|
|
28
|
+
"PDF support is not installed. Install the document extra: "
|
|
29
|
+
"python -m pip install -e '.[document]'"
|
|
30
|
+
) from exc
|
|
31
|
+
|
|
32
|
+
pages: list[PageIR] = []
|
|
33
|
+
with fitz.open(path) as pdf:
|
|
34
|
+
for number, page in enumerate(pdf, start=1):
|
|
35
|
+
text = page.get_text("text").strip()
|
|
36
|
+
page_id = stable_id("page", document_id, str(number))
|
|
37
|
+
pages.append(
|
|
38
|
+
PageIR(
|
|
39
|
+
page_id=page_id,
|
|
40
|
+
page_number=number,
|
|
41
|
+
text=text,
|
|
42
|
+
blocks=[],
|
|
43
|
+
)
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
return _document(document_id, title, pages)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _ingest_docx(path: Path, document_id: str, title: str) -> LearningDocument:
|
|
50
|
+
try:
|
|
51
|
+
from docx import Document
|
|
52
|
+
except ImportError as exc:
|
|
53
|
+
raise RuntimeError(
|
|
54
|
+
"DOCX support is not installed. Install the document extra: "
|
|
55
|
+
"python -m pip install -e '.[document]'"
|
|
56
|
+
) from exc
|
|
57
|
+
|
|
58
|
+
document = Document(path)
|
|
59
|
+
text = "\n".join(
|
|
60
|
+
paragraph.text.strip()
|
|
61
|
+
for paragraph in document.paragraphs
|
|
62
|
+
if paragraph.text.strip()
|
|
63
|
+
)
|
|
64
|
+
page_id = stable_id("page", document_id, "1")
|
|
65
|
+
pages = [PageIR(page_id=page_id, page_number=1, text=text)]
|
|
66
|
+
return _document(document_id, title, pages)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _document(document_id: str, title: str, pages: list[PageIR]) -> LearningDocument:
|
|
70
|
+
text = "\n\n".join(page.text for page in pages if page.text.strip())
|
|
71
|
+
page_id = pages[0].page_id if pages else stable_id("page", document_id, "1")
|
|
72
|
+
concept = Concept(
|
|
73
|
+
concept_id=stable_id("concept", document_id, text[:160]),
|
|
74
|
+
name=title,
|
|
75
|
+
description=text[:1000],
|
|
76
|
+
evidence=[
|
|
77
|
+
SourceRegion(page_id=page_id, extracted_text=text[:4000])
|
|
78
|
+
],
|
|
79
|
+
)
|
|
80
|
+
return LearningDocument(
|
|
81
|
+
document_id=document_id,
|
|
82
|
+
title=title,
|
|
83
|
+
pages=pages,
|
|
84
|
+
concepts=[concept] if text else [],
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class DocumentIngestor:
|
|
89
|
+
"""Compatibility boundary for document adapters."""
|
|
90
|
+
|
|
91
|
+
def ingest(self, path: Path) -> LearningDocument:
|
|
92
|
+
return ingest_document(path)
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
from app.core.ids import stable_id
|
|
2
|
+
from app.core.models import Concept, LearningDocument, PageIR, SourceRegion
|
|
3
|
+
|
|
4
|
+
def ingest_text(document_id: str, text: str) -> LearningDocument:
|
|
5
|
+
page_id = stable_id("page", document_id, "1")
|
|
6
|
+
concept_id = stable_id("concept", document_id, text[:160])
|
|
7
|
+
concept = Concept(concept_id=concept_id, name="Imported concept", description=text[:1000],
|
|
8
|
+
evidence=[SourceRegion(page_id=page_id, extracted_text=text[:4000])])
|
|
9
|
+
return LearningDocument(document_id=document_id, title=document_id,
|
|
10
|
+
pages=[PageIR(page_id=page_id, page_number=1, text=text)],
|
|
11
|
+
concepts=[concept])
|
|
File without changes
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from app.core.models import Concept
|
|
2
|
+
|
|
3
|
+
class InMemoryKnowledgeGraph:
|
|
4
|
+
def __init__(self): self.nodes={}; self.edges=[]
|
|
5
|
+
def upsert_concept(self, concept: Concept) -> None: self.nodes[concept.concept_id]=concept
|
|
6
|
+
def add_edge(self, source: str, relation: str, target: str) -> None: self.edges.append((source,relation,target))
|
|
7
|
+
|
|
8
|
+
class InMemoryVectorStore:
|
|
9
|
+
def __init__(self): self.items=[]
|
|
10
|
+
def upsert(self, document_id: str, text: str, metadata: dict[str,str]) -> None:
|
|
11
|
+
self.items.append({"id":document_id,"text":text,"metadata":metadata})
|
|
12
|
+
def search(self, query: str, limit: int=5) -> list[dict[str,object]]:
|
|
13
|
+
terms=set(query.lower().split())
|
|
14
|
+
ranked=sorted(self.items,key=lambda x:len(terms & set(str(x["text"]).lower().split())),reverse=True)
|
|
15
|
+
return ranked[:limit]
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
from typing import Protocol
|
|
2
|
+
from app.core.models import Concept
|
|
3
|
+
|
|
4
|
+
class KnowledgeGraph(Protocol):
|
|
5
|
+
def upsert_concept(self, concept: Concept) -> None: ...
|
|
6
|
+
def add_edge(self, source: str, relation: str, target: str) -> None: ...
|
|
7
|
+
|
|
8
|
+
class VectorStore(Protocol):
|
|
9
|
+
def upsert(self, document_id: str, text: str, metadata: dict[str,str]) -> None: ...
|
|
10
|
+
def search(self, query: str, limit: int=5) -> list[dict[str,object]]: ...
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from urllib.error import HTTPError, URLError
|
|
8
|
+
from urllib.request import Request, urlopen
|
|
9
|
+
|
|
10
|
+
_CONFIG_DIR = Path.home() / ".config" / "examforge"
|
|
11
|
+
_CONFIG_FILE = _CONFIG_DIR / "llm.json"
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class LLMConfig:
|
|
15
|
+
provider: str
|
|
16
|
+
endpoint: str
|
|
17
|
+
model: str
|
|
18
|
+
api_key_env: str | None = None
|
|
19
|
+
|
|
20
|
+
@property
|
|
21
|
+
def api_key(self) -> str | None:
|
|
22
|
+
return os.getenv(self.api_key_env) if self.api_key_env else None
|
|
23
|
+
|
|
24
|
+
def save_config(config: LLMConfig) -> Path:
|
|
25
|
+
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
|
|
26
|
+
_CONFIG_FILE.write_text(
|
|
27
|
+
json.dumps(
|
|
28
|
+
{
|
|
29
|
+
"provider": config.provider,
|
|
30
|
+
"endpoint": config.endpoint,
|
|
31
|
+
"model": config.model,
|
|
32
|
+
"api_key_env": config.api_key_env,
|
|
33
|
+
},
|
|
34
|
+
indent=2,
|
|
35
|
+
)
|
|
36
|
+
+ "\n",
|
|
37
|
+
encoding="utf-8",
|
|
38
|
+
)
|
|
39
|
+
return _CONFIG_FILE
|
|
40
|
+
|
|
41
|
+
def load_config() -> LLMConfig | None:
|
|
42
|
+
if not _CONFIG_FILE.exists():
|
|
43
|
+
return None
|
|
44
|
+
data = json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
|
|
45
|
+
return LLMConfig(
|
|
46
|
+
provider=data["provider"],
|
|
47
|
+
endpoint=data["endpoint"],
|
|
48
|
+
model=data["model"],
|
|
49
|
+
api_key_env=data.get("api_key_env"),
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
class LLMClient:
|
|
53
|
+
"""Small dependency-free client for Ollama and OpenAI-compatible APIs."""
|
|
54
|
+
|
|
55
|
+
def __init__(self, config: LLMConfig):
|
|
56
|
+
self.config = config
|
|
57
|
+
|
|
58
|
+
def _post(self, url: str, payload: dict[str, object], headers: dict[str, str] | None = None) -> dict[str, object]:
|
|
59
|
+
body = json.dumps(payload).encode("utf-8")
|
|
60
|
+
request = Request(url, data=body, method="POST")
|
|
61
|
+
request.add_header("Content-Type", "application/json")
|
|
62
|
+
for key, value in (headers or {}).items():
|
|
63
|
+
request.add_header(key, value)
|
|
64
|
+
try:
|
|
65
|
+
with urlopen(request, timeout=120) as response:
|
|
66
|
+
return json.loads(response.read().decode("utf-8"))
|
|
67
|
+
except HTTPError as exc:
|
|
68
|
+
detail = exc.read().decode("utf-8", errors="replace")
|
|
69
|
+
raise RuntimeError(f"LLM request failed ({exc.code}): {detail}") from exc
|
|
70
|
+
except URLError as exc:
|
|
71
|
+
raise RuntimeError(f"Could not reach LLM at {url}: {exc.reason}") from exc
|
|
72
|
+
|
|
73
|
+
def chat(self, prompt: str, system: str | None = None) -> str:
|
|
74
|
+
messages: list[dict[str, str]] = []
|
|
75
|
+
if system:
|
|
76
|
+
messages.append({"role": "system", "content": system})
|
|
77
|
+
messages.append({"role": "user", "content": prompt})
|
|
78
|
+
|
|
79
|
+
if self.config.provider == "ollama":
|
|
80
|
+
payload = {"model": self.config.model, "messages": messages, "stream": False}
|
|
81
|
+
result = self._post(self.config.endpoint.rstrip("/") + "/api/chat", payload)
|
|
82
|
+
return str(result["message"]["content"])
|
|
83
|
+
|
|
84
|
+
headers = {}
|
|
85
|
+
if self.config.api_key:
|
|
86
|
+
headers["Authorization"] = f"Bearer {self.config.api_key}"
|
|
87
|
+
payload = {"model": self.config.model, "messages": messages, "stream": False}
|
|
88
|
+
result = self._post(self.config.endpoint.rstrip("/") + "/chat/completions", payload, headers)
|
|
89
|
+
return str(result["choices"][0]["message"]["content"])
|
|
90
|
+
|
|
91
|
+
def config_path() -> Path:
|
|
92
|
+
return _CONFIG_FILE
|
|
File without changes
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from app.core.models import VisualBlueprint
|
|
3
|
+
|
|
4
|
+
@dataclass
|
|
5
|
+
class CompilationResult:
|
|
6
|
+
valid: bool
|
|
7
|
+
source: str
|
|
8
|
+
errors: list[str]
|
|
9
|
+
|
|
10
|
+
class ManimCompiler:
|
|
11
|
+
"""Blueprint -> template -> Manim source. The LLM never owns this interface."""
|
|
12
|
+
def compile(self, blueprint: VisualBlueprint) -> CompilationResult:
|
|
13
|
+
source=("from manim import *\n\nclass ExamForgeScene(Scene):\n"
|
|
14
|
+
" def construct(self):\n"
|
|
15
|
+
f" title = Text({blueprint.learning_objective!r})\n"
|
|
16
|
+
" self.play(Write(title))\n self.wait(1)\n")
|
|
17
|
+
return CompilationResult(True, source, [])
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from app.core.models import VisualBlueprint
|
|
3
|
+
|
|
4
|
+
@dataclass
|
|
5
|
+
class ValidationReport:
|
|
6
|
+
code_valid: bool
|
|
7
|
+
render_valid: bool
|
|
8
|
+
pedagogical_valid: bool
|
|
9
|
+
errors: list[str]
|
|
10
|
+
|
|
11
|
+
def validate_blueprint(blueprint: VisualBlueprint) -> ValidationReport:
|
|
12
|
+
errors=[]
|
|
13
|
+
if not blueprint.learning_objective.strip(): errors.append("Missing learning objective.")
|
|
14
|
+
if not blueprint.animations: errors.append("Blueprint contains no animation actions.")
|
|
15
|
+
return ValidationReport(not errors, False, not errors, errors)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Document-to-Manim video generation pipeline."""
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import ast
|
|
4
|
+
import re
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from app.ingestion.document import ingest_document
|
|
11
|
+
from app.llm.client import LLMClient, load_config
|
|
12
|
+
|
|
13
|
+
_ALLOWED_CLASS_PATTERN = re.compile(r"class\s+ExamForgeScene\s*\(\s*Scene\s*\)")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class VideoResult:
|
|
18
|
+
source_path: Path
|
|
19
|
+
output_path: Path
|
|
20
|
+
scene_path: Path
|
|
21
|
+
document_id: str
|
|
22
|
+
provider: str
|
|
23
|
+
model: str
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _extract_code(response: str) -> str:
|
|
27
|
+
match = re.search(r"```(?:python)?\s*(.*?)```", response, re.DOTALL | re.IGNORECASE)
|
|
28
|
+
return match.group(1).strip() if match else response.strip()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _validate_source(source: str) -> None:
|
|
32
|
+
try:
|
|
33
|
+
tree = ast.parse(source)
|
|
34
|
+
except SyntaxError as exc:
|
|
35
|
+
raise RuntimeError(f"LLM returned invalid Python: {exc}") from exc
|
|
36
|
+
|
|
37
|
+
for node in ast.walk(tree):
|
|
38
|
+
if isinstance(node, ast.Import):
|
|
39
|
+
raise RuntimeError("Generated source may only import from manim.")
|
|
40
|
+
if isinstance(node, ast.ImportFrom) and node.module != "manim":
|
|
41
|
+
raise RuntimeError("Generated source may only import from manim.")
|
|
42
|
+
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and node.func.id in {
|
|
43
|
+
"eval", "exec", "open", "compile", "__import__", "input"
|
|
44
|
+
}:
|
|
45
|
+
raise RuntimeError(f"Generated source uses forbidden function: {node.func.id}")
|
|
46
|
+
if isinstance(node, ast.Name) and node.id in {
|
|
47
|
+
"os", "sys", "subprocess", "pathlib", "socket", "requests"
|
|
48
|
+
}:
|
|
49
|
+
raise RuntimeError(f"Generated source uses forbidden module/name: {node.id}")
|
|
50
|
+
|
|
51
|
+
if not _ALLOWED_CLASS_PATTERN.search(source):
|
|
52
|
+
raise RuntimeError("Generated Manim source must define an ExamForgeScene(Scene) class.")
|
|
53
|
+
|
|
54
|
+
classes = [node for node in tree.body if isinstance(node, ast.ClassDef)]
|
|
55
|
+
if not any(node.name == "ExamForgeScene" for node in classes):
|
|
56
|
+
raise RuntimeError("Generated source does not contain ExamForgeScene.")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _prompt(title: str, text: str) -> str:
|
|
60
|
+
return f"""Create a concise educational Manim Community Edition video from the document below.
|
|
61
|
+
|
|
62
|
+
Document title: {title}
|
|
63
|
+
|
|
64
|
+
Requirements:
|
|
65
|
+
- Return ONLY complete Python source code.
|
|
66
|
+
- Import from manim with: from manim import *
|
|
67
|
+
- Define exactly one main scene named ExamForgeScene(Scene).
|
|
68
|
+
- Teach the most important concepts in a clear sequence using Text, MathTex, axes, shapes, arrows, and simple animations where useful.
|
|
69
|
+
- Prefer deterministic, readable Manim primitives over external assets.
|
|
70
|
+
- Keep rendering practical: target roughly 30-90 seconds and avoid expensive simulations.
|
|
71
|
+
- Do not use network access, file I/O, subprocesses, eval, exec, or arbitrary imports.
|
|
72
|
+
- Escape text safely and use MathTex only for valid LaTeX.
|
|
73
|
+
- The scene must run with standard Manim Community Edition.
|
|
74
|
+
|
|
75
|
+
DOCUMENT:
|
|
76
|
+
{text[:30000]}
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def generate_video(
|
|
81
|
+
source_path: str | Path,
|
|
82
|
+
output_path: str | Path | None = None,
|
|
83
|
+
) -> VideoResult:
|
|
84
|
+
source = Path(source_path).expanduser().resolve()
|
|
85
|
+
if not source.exists() or not source.is_file():
|
|
86
|
+
raise FileNotFoundError(f"Input file not found: {source}")
|
|
87
|
+
if source.suffix.lower() not in {".pdf", ".docx"}:
|
|
88
|
+
raise ValueError("Input must be a .pdf or .docx file.")
|
|
89
|
+
|
|
90
|
+
config = load_config()
|
|
91
|
+
if config is None:
|
|
92
|
+
raise RuntimeError(
|
|
93
|
+
"No LLM linked. Run examforge llm connect-api or "
|
|
94
|
+
"examforge llm connect-ollama first."
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
document = ingest_document(source)
|
|
98
|
+
if not document.pages:
|
|
99
|
+
raise RuntimeError("No readable content was extracted from the document.")
|
|
100
|
+
|
|
101
|
+
text = "\n\n".join(page.text for page in document.pages if page.text.strip())
|
|
102
|
+
if not text.strip():
|
|
103
|
+
raise RuntimeError("The document contains no extractable text.")
|
|
104
|
+
|
|
105
|
+
response = LLMClient(config).chat(_prompt(document.title, text))
|
|
106
|
+
manim_source = _extract_code(response)
|
|
107
|
+
_validate_source(manim_source)
|
|
108
|
+
|
|
109
|
+
target = Path(output_path).expanduser() if output_path else source.with_suffix(".mp4")
|
|
110
|
+
target = target.resolve()
|
|
111
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
112
|
+
|
|
113
|
+
scene_dir = target.parent / ".examforge" / source.stem
|
|
114
|
+
scene_dir.mkdir(parents=True, exist_ok=True)
|
|
115
|
+
scene_path = scene_dir / "scene.py"
|
|
116
|
+
scene_path.write_text(manim_source + "\n", encoding="utf-8")
|
|
117
|
+
|
|
118
|
+
command = [
|
|
119
|
+
sys.executable, "-m", "manim", "-q", "m", str(scene_path), "ExamForgeScene",
|
|
120
|
+
"--media_dir", str(scene_dir / "media"), "-o", target.name,
|
|
121
|
+
]
|
|
122
|
+
try:
|
|
123
|
+
completed = subprocess.run(command, check=True, capture_output=True, text=True)
|
|
124
|
+
except FileNotFoundError as exc:
|
|
125
|
+
raise RuntimeError(
|
|
126
|
+
"Manim is not installed. Install the generation extra: "
|
|
127
|
+
"python -m pip install -e '.[generation]'"
|
|
128
|
+
) from exc
|
|
129
|
+
except subprocess.CalledProcessError as exc:
|
|
130
|
+
detail = (exc.stderr or exc.stdout or "unknown Manim error").strip()
|
|
131
|
+
raise RuntimeError(f"Manim render failed: {detail}") from exc
|
|
132
|
+
|
|
133
|
+
rendered = scene_dir / "media" / "videos" / "scene" / "720p30" / target.name
|
|
134
|
+
if rendered.exists() and rendered != target:
|
|
135
|
+
target.write_bytes(rendered.read_bytes())
|
|
136
|
+
elif not target.exists():
|
|
137
|
+
candidates = list((scene_dir / "media").rglob(target.name))
|
|
138
|
+
if candidates:
|
|
139
|
+
target.write_bytes(candidates[0].read_bytes())
|
|
140
|
+
|
|
141
|
+
if not target.exists():
|
|
142
|
+
raise RuntimeError(
|
|
143
|
+
"Manim completed without producing the expected video output. "
|
|
144
|
+
f"stdout: {completed.stdout[-1000:]}"
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
return VideoResult(
|
|
148
|
+
source_path=source,
|
|
149
|
+
output_path=target,
|
|
150
|
+
scene_path=scene_path,
|
|
151
|
+
document_id=document.document_id,
|
|
152
|
+
provider=config.provider,
|
|
153
|
+
model=config.model,
|
|
154
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from app.core.models import EquationIR
|
|
3
|
+
|
|
4
|
+
def parse_equation(latex: str, equation_type: str="unknown", operation: str|None=None) -> EquationIR:
|
|
5
|
+
variables=sorted(set(re.findall(r"(?<![A-Za-z])[A-Za-z](?![A-Za-z])", latex)))
|
|
6
|
+
return EquationIR(latex=latex, type=equation_type, variables=variables, operation=operation)
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
from app.core.models import GraphIR, Reconstructability, VisualObject
|
|
2
|
+
|
|
3
|
+
def interpret_graph(visual_id: str, x_label: str, y_label: str,
|
|
4
|
+
objects: list[tuple[str, str, str]]) -> GraphIR:
|
|
5
|
+
parsed=[VisualObject(id=f"{visual_id}_{label.lower()}", type=kind, label=label, direction=direction)
|
|
6
|
+
for label, kind, direction in objects]
|
|
7
|
+
return GraphIR(visual_id=visual_id,
|
|
8
|
+
axes={"x":{"label":x_label,"scale":"unknown"},"y":{"label":y_label,"scale":"unknown"}},
|
|
9
|
+
objects=parsed, reconstructability=Reconstructability.reconstructable,
|
|
10
|
+
confidence=0.9 if parsed else 0.5)
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
from app.core.models import TableIR
|
|
2
|
+
|
|
3
|
+
def parse_table(columns: list[str], rows: list[list[object]]) -> TableIR:
|
|
4
|
+
if any(len(row)!=len(columns) for row in rows):
|
|
5
|
+
raise ValueError("Every table row must match the column count.")
|
|
6
|
+
return TableIR(columns=columns, rows=rows)
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: examforge
|
|
3
|
+
Version: 0.2.0.post9
|
|
4
|
+
Summary: CLI-first multimodal exam-preparation engine with a canonical learning IR
|
|
5
|
+
Requires-Python: >=3.12
|
|
6
|
+
Requires-Dist: pydantic<3,>=2.8
|
|
7
|
+
Provides-Extra: document
|
|
8
|
+
Requires-Dist: pymupdf>=1.24; extra == "document"
|
|
9
|
+
Requires-Dist: pymupdf4llm>=0.0.17; extra == "document"
|
|
10
|
+
Requires-Dist: python-docx>=1.1; extra == "document"
|
|
11
|
+
Requires-Dist: pytesseract>=0.3.13; extra == "document"
|
|
12
|
+
Requires-Dist: opencv-python>=4.10; extra == "document"
|
|
13
|
+
Provides-Extra: knowledge
|
|
14
|
+
Requires-Dist: neo4j>=5.25; extra == "knowledge"
|
|
15
|
+
Requires-Dist: qdrant-client>=1.12; extra == "knowledge"
|
|
16
|
+
Provides-Extra: generation
|
|
17
|
+
Requires-Dist: manim>=0.18; extra == "generation"
|
|
18
|
+
Provides-Extra: audio
|
|
19
|
+
Requires-Dist: openai-whisper>=20240930; extra == "audio"
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: pytest>=8.3; extra == "dev"
|
|
22
|
+
Requires-Dist: httpx>=0.27; extra == "dev"
|
|
23
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
24
|
+
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
app/__init__.py
|
|
4
|
+
app/__main__.py
|
|
5
|
+
app/cli.py
|
|
6
|
+
app/assessment/__init__.py
|
|
7
|
+
app/assessment/mastery.py
|
|
8
|
+
app/assessment/quiz.py
|
|
9
|
+
app/audio/__init__.py
|
|
10
|
+
app/audio/sync.py
|
|
11
|
+
app/blueprints/__init__.py
|
|
12
|
+
app/blueprints/engine.py
|
|
13
|
+
app/core/__init__.py
|
|
14
|
+
app/core/ids.py
|
|
15
|
+
app/core/models.py
|
|
16
|
+
app/ingestion/__init__.py
|
|
17
|
+
app/ingestion/document.py
|
|
18
|
+
app/ingestion/text.py
|
|
19
|
+
app/knowledge/__init__.py
|
|
20
|
+
app/knowledge/in_memory.py
|
|
21
|
+
app/knowledge/ports.py
|
|
22
|
+
app/llm/client.py
|
|
23
|
+
app/manim/__init__.py
|
|
24
|
+
app/manim/compiler.py
|
|
25
|
+
app/manim/validation.py
|
|
26
|
+
app/video/__init__.py
|
|
27
|
+
app/video/pipeline.py
|
|
28
|
+
app/visual/__init__.py
|
|
29
|
+
app/visual/equation.py
|
|
30
|
+
app/visual/graph.py
|
|
31
|
+
app/visual/table.py
|
|
32
|
+
examforge.egg-info/PKG-INFO
|
|
33
|
+
examforge.egg-info/SOURCES.txt
|
|
34
|
+
examforge.egg-info/dependency_links.txt
|
|
35
|
+
examforge.egg-info/entry_points.txt
|
|
36
|
+
examforge.egg-info/requires.txt
|
|
37
|
+
examforge.egg-info/top_level.txt
|
|
38
|
+
tests/test_cli.py
|
|
39
|
+
tests/test_document_video.py
|
|
40
|
+
tests/test_ids.py
|
|
41
|
+
tests/test_visual.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
pydantic<3,>=2.8
|
|
2
|
+
|
|
3
|
+
[audio]
|
|
4
|
+
openai-whisper>=20240930
|
|
5
|
+
|
|
6
|
+
[dev]
|
|
7
|
+
pytest>=8.3
|
|
8
|
+
httpx>=0.27
|
|
9
|
+
ruff>=0.6
|
|
10
|
+
mypy>=1.13
|
|
11
|
+
|
|
12
|
+
[document]
|
|
13
|
+
pymupdf>=1.24
|
|
14
|
+
pymupdf4llm>=0.0.17
|
|
15
|
+
python-docx>=1.1
|
|
16
|
+
pytesseract>=0.3.13
|
|
17
|
+
opencv-python>=4.10
|
|
18
|
+
|
|
19
|
+
[generation]
|
|
20
|
+
manim>=0.18
|
|
21
|
+
|
|
22
|
+
[knowledge]
|
|
23
|
+
neo4j>=5.25
|
|
24
|
+
qdrant-client>=1.12
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
app
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "examforge"
|
|
7
|
+
version = "0.2.0.post9"
|
|
8
|
+
description = "CLI-first multimodal exam-preparation engine with a canonical learning IR"
|
|
9
|
+
requires-python = ">=3.12"
|
|
10
|
+
dependencies = ["pydantic>=2.8,<3"]
|
|
11
|
+
|
|
12
|
+
[project.optional-dependencies]
|
|
13
|
+
document = ["pymupdf>=1.24", "pymupdf4llm>=0.0.17", "python-docx>=1.1", "pytesseract>=0.3.13", "opencv-python>=4.10"]
|
|
14
|
+
knowledge = ["neo4j>=5.25", "qdrant-client>=1.12"]
|
|
15
|
+
generation = ["manim>=0.18"]
|
|
16
|
+
audio = ["openai-whisper>=20240930"]
|
|
17
|
+
dev = ["pytest>=8.3", "httpx>=0.27", "ruff>=0.6", "mypy>=1.13"]
|
|
18
|
+
|
|
19
|
+
[project.scripts]
|
|
20
|
+
examforge = "app.cli:main"
|
|
21
|
+
|
|
22
|
+
[tool.setuptools.packages.find]
|
|
23
|
+
include = ["app*"]
|
|
24
|
+
|
|
25
|
+
[tool.pytest.ini_options]
|
|
26
|
+
testpaths = ["tests"]
|
|
27
|
+
|
|
28
|
+
[tool.ruff]
|
|
29
|
+
line-length = 100
|
|
30
|
+
target-version = "py312"
|
|
31
|
+
|
|
32
|
+
[tool.mypy]
|
|
33
|
+
python_version = "3.12"
|
|
34
|
+
strict = true
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from app.cli import main
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def test_architecture_command(capsys):
|
|
5
|
+
assert main(["architecture"]) == 0
|
|
6
|
+
assert "canonical_learning_ir" in capsys.readouterr().out
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def test_ingest_text_command(capsys):
|
|
10
|
+
assert main(["ingest-text", "--document-id", "demo", "--text", "Supply and demand"]) == 0
|
|
11
|
+
output = capsys.readouterr().out
|
|
12
|
+
assert '"document_id": "demo"' in output
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_llm_connect_ollama(capsys, monkeypatch, tmp_path):
|
|
16
|
+
import app.llm.client as client
|
|
17
|
+
|
|
18
|
+
monkeypatch.setattr(client, "_CONFIG_DIR", tmp_path)
|
|
19
|
+
monkeypatch.setattr(client, "_CONFIG_FILE", tmp_path / "llm.json")
|
|
20
|
+
assert main(["llm", "connect-ollama", "--model", "llama3.2:3b"]) == 0
|
|
21
|
+
output = capsys.readouterr().out
|
|
22
|
+
assert '"provider": "ollama"' in output
|
|
23
|
+
assert '"model": "llama3.2:3b"' in output
|
|
24
|
+
assert main(["llm", "status"]) == 0
|
|
25
|
+
assert '"connected": true' in capsys.readouterr().out
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from app.ingestion.document import ingest_document
|
|
6
|
+
from app.video.pipeline import _extract_code, _validate_source
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def test_extract_code_block():
|
|
10
|
+
source = "```python\nfrom manim import *\nclass ExamForgeScene(Scene):\n pass\n```"
|
|
11
|
+
assert _extract_code(source).startswith("from manim import *")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_validate_source_accepts_scene():
|
|
15
|
+
_validate_source("from manim import *\nclass ExamForgeScene(Scene):\n def construct(self):\n pass\n")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def test_validate_source_rejects_invalid_python():
|
|
19
|
+
with pytest.raises(RuntimeError, match="invalid Python"):
|
|
20
|
+
_validate_source("class ExamForgeScene(Scene):\n def construct(")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_validate_source_rejects_wrong_scene_name():
|
|
24
|
+
with pytest.raises(RuntimeError, match="ExamForgeScene"):
|
|
25
|
+
_validate_source("from manim import *\nclass OtherScene(Scene):\n pass\n")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_ingest_document_rejects_unknown_suffix(tmp_path: Path):
|
|
29
|
+
source = tmp_path / "notes.txt"
|
|
30
|
+
source.write_text("hello", encoding="utf-8")
|
|
31
|
+
with pytest.raises(ValueError, match="\\.pdf or \\.docx"):
|
|
32
|
+
ingest_document(source)
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from app.visual.equation import parse_equation
|
|
2
|
+
from app.visual.graph import interpret_graph
|
|
3
|
+
from app.visual.table import parse_table
|
|
4
|
+
|
|
5
|
+
def test_equation_ir():
|
|
6
|
+
eq=parse_equation(r"\frac{d}{dx}x^2 = 2x","derivative","differentiation")
|
|
7
|
+
assert eq.operation=="differentiation" and "x" in eq.variables
|
|
8
|
+
|
|
9
|
+
def test_graph_ir():
|
|
10
|
+
graph=interpret_graph("fig_1","Quantity","Price",[("D","curve","negative"),("S","curve","positive")])
|
|
11
|
+
assert graph.reconstructability.value=="RECONSTRUCTABLE" and len(graph.objects)==2
|
|
12
|
+
|
|
13
|
+
def test_table_ir():
|
|
14
|
+
assert parse_table(["Year","GDP"],[[2024,100]]).rows[0][1]==100
|