studyforge-vocab 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
studyforge/__init__.py ADDED
@@ -0,0 +1,23 @@
1
+ """Reusable StudyForge PDF vocabulary extraction library."""
2
+
3
+ __version__ = "0.2.0"
4
+
5
+ from studyforge.api import StudyForge, analyze_pdf, analyze_pdf_bytes
6
+ from studyforge.cefr import CEFR_LEVELS, CEFR_UNKNOWN, CEFRProfile
7
+ from studyforge.exporter import export_rows, export_vocabulary
8
+ from studyforge.models import AnalysisResult, PdfDocument, VocabularyItem
9
+
10
+ __all__ = [
11
+ "AnalysisResult",
12
+ "CEFR_LEVELS",
13
+ "CEFR_UNKNOWN",
14
+ "CEFRProfile",
15
+ "PdfDocument",
16
+ "StudyForge",
17
+ "VocabularyItem",
18
+ "__version__",
19
+ "analyze_pdf",
20
+ "analyze_pdf_bytes",
21
+ "export_rows",
22
+ "export_vocabulary",
23
+ ]
studyforge/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from studyforge.cli import main
2
+
3
+
4
+ raise SystemExit(main())
studyforge/api.py ADDED
@@ -0,0 +1,116 @@
1
+ from __future__ import annotations
2
+
3
+ from os import PathLike
4
+ from pathlib import Path
5
+ from typing import Any
6
+
7
+ from studyforge.cefr import CEFRProfile, default_cefr_profile
8
+ from studyforge.dictionary import DictionaryStore
9
+ from studyforge.models import AnalysisResult
10
+ from studyforge.pdf_reader import MAX_PDF_BYTES, PdfReadError, extract_pdf_text
11
+ from studyforge.resources import default_dictionary_path
12
+ from studyforge.vocabulary import LEVEL_LABELS, analyze_vocabulary
13
+
14
+
15
+ VALID_MODES = tuple(LEVEL_LABELS)
16
+
17
+
18
+ class StudyForge:
19
+ """Reusable PDF vocabulary extraction service used by Web, CLI, and Python."""
20
+
21
+ def __init__(
22
+ self,
23
+ dictionary_path: str | PathLike[str] | None = None,
24
+ cefr_profile: CEFRProfile | None = None,
25
+ ):
26
+ self.dictionary = DictionaryStore(
27
+ Path(dictionary_path) if dictionary_path else default_dictionary_path()
28
+ )
29
+ self.cefr_profile = cefr_profile or default_cefr_profile()
30
+
31
+ def analyze_bytes(
32
+ self,
33
+ file_bytes: bytes,
34
+ *,
35
+ limit: int = 30,
36
+ mode: str = "balanced",
37
+ min_occurrences: int = 1,
38
+ ) -> AnalysisResult:
39
+ _validate_options(limit, mode, min_occurrences)
40
+ document = extract_pdf_text(file_bytes)
41
+ items = analyze_vocabulary(
42
+ document,
43
+ self.dictionary,
44
+ limit=limit,
45
+ level=mode,
46
+ min_occurrences=min_occurrences,
47
+ cefr_profile=self.cefr_profile,
48
+ )
49
+ return AnalysisResult(document=document, items=tuple(items), mode=mode)
50
+
51
+ def analyze_file(
52
+ self,
53
+ path: str | PathLike[str],
54
+ *,
55
+ limit: int = 30,
56
+ mode: str = "balanced",
57
+ min_occurrences: int = 1,
58
+ ) -> AnalysisResult:
59
+ pdf_path = Path(path).expanduser()
60
+ if not pdf_path.is_file():
61
+ raise FileNotFoundError(f"PDF file not found: {pdf_path}")
62
+ if pdf_path.stat().st_size > MAX_PDF_BYTES:
63
+ raise PdfReadError(
64
+ f"PDF exceeds the {MAX_PDF_BYTES // (1024 * 1024)} MB limit."
65
+ )
66
+ return self.analyze_bytes(
67
+ pdf_path.read_bytes(),
68
+ limit=limit,
69
+ mode=mode,
70
+ min_occurrences=min_occurrences,
71
+ )
72
+
73
+
74
+ def analyze_pdf(
75
+ path: str | PathLike[str],
76
+ *,
77
+ limit: int = 30,
78
+ mode: str = "balanced",
79
+ min_occurrences: int = 1,
80
+ dictionary_path: str | PathLike[str] | None = None,
81
+ cefr_profile: CEFRProfile | None = None,
82
+ ) -> AnalysisResult:
83
+ """Analyze a PDF file with the bundled core library."""
84
+ return StudyForge(dictionary_path, cefr_profile).analyze_file(
85
+ path,
86
+ limit=limit,
87
+ mode=mode,
88
+ min_occurrences=min_occurrences,
89
+ )
90
+
91
+
92
+ def analyze_pdf_bytes(
93
+ file_bytes: bytes,
94
+ *,
95
+ limit: int = 30,
96
+ mode: str = "balanced",
97
+ min_occurrences: int = 1,
98
+ dictionary_path: str | PathLike[str] | None = None,
99
+ cefr_profile: CEFRProfile | None = None,
100
+ ) -> AnalysisResult:
101
+ """Analyze in-memory PDF bytes with the bundled core library."""
102
+ return StudyForge(dictionary_path, cefr_profile).analyze_bytes(
103
+ file_bytes,
104
+ limit=limit,
105
+ mode=mode,
106
+ min_occurrences=min_occurrences,
107
+ )
108
+
109
+
110
+ def _validate_options(limit: int, mode: str, min_occurrences: int) -> None:
111
+ if not isinstance(limit, int) or not 1 <= limit <= 500:
112
+ raise ValueError("limit must be an integer between 1 and 500.")
113
+ if mode not in VALID_MODES:
114
+ raise ValueError(f"mode must be one of: {', '.join(VALID_MODES)}.")
115
+ if not isinstance(min_occurrences, int) or not 1 <= min_occurrences <= 100:
116
+ raise ValueError("min_occurrences must be an integer between 1 and 100.")
studyforge/cefr.py ADDED
@@ -0,0 +1,62 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ from functools import lru_cache
6
+ from pathlib import Path
7
+ from typing import Mapping
8
+
9
+ from studyforge.resources import default_cefr_path
10
+
11
+
12
+ CEFR_LEVELS = ("A1", "A2", "B1", "B2", "C1", "C2")
13
+ CEFR_UNKNOWN = "unknown"
14
+ _VALID_WORD = re.compile(r"^[a-z]+(?:['-][a-z]+)*$")
15
+
16
+
17
+ def normalize_cefr_word(word: str) -> str:
18
+ return word.replace("’", "'").strip().lower()
19
+
20
+
21
+ class CEFRProfile:
22
+ """Reliable word-level CEFR lookups with explicit unknown fallbacks."""
23
+
24
+ def __init__(self, levels: Mapping[str, str] | None = None):
25
+ normalized: dict[str, str] = {}
26
+ for word, level in (levels or {}).items():
27
+ normalized_word = normalize_cefr_word(word)
28
+ normalized_level = str(level).upper()
29
+ if _VALID_WORD.fullmatch(normalized_word) and normalized_level in CEFR_LEVELS:
30
+ normalized[normalized_word] = normalized_level
31
+ self._levels = normalized
32
+
33
+ @classmethod
34
+ def from_json(cls, path: str | Path) -> "CEFRProfile":
35
+ data = json.loads(Path(path).read_text(encoding="utf-8"))
36
+ levels = data.get("levels", data)
37
+ if not isinstance(levels, dict):
38
+ raise ValueError("CEFR data must contain a 'levels' object.")
39
+ return cls(levels)
40
+
41
+ @classmethod
42
+ def from_mapping(cls, levels: Mapping[str, str]) -> "CEFRProfile":
43
+ return cls(levels)
44
+
45
+ def level_for(self, word: str) -> str:
46
+ """Return A1-C2 only for a known unambiguous entry; otherwise unknown."""
47
+ return self._levels.get(normalize_cefr_word(word), CEFR_UNKNOWN)
48
+
49
+ def classify_many(self, words: list[str] | tuple[str, ...]) -> dict[str, str]:
50
+ return {word: self.level_for(word) for word in words}
51
+
52
+ @property
53
+ def known_word_count(self) -> int:
54
+ return len(self._levels)
55
+
56
+
57
+ @lru_cache(maxsize=1)
58
+ def default_cefr_profile() -> CEFRProfile:
59
+ path = default_cefr_path()
60
+ if not path.is_file():
61
+ return CEFRProfile()
62
+ return CEFRProfile.from_json(path)
studyforge/cli.py ADDED
@@ -0,0 +1,88 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import re
5
+ import sys
6
+ from pathlib import Path
7
+ from typing import Sequence
8
+
9
+ from studyforge import __version__
10
+ from studyforge.api import StudyForge, VALID_MODES
11
+ from studyforge.exporter import EXPORT_FORMATS, export_vocabulary, output_suffix
12
+ from studyforge.pdf_reader import PdfReadError
13
+
14
+
15
+ def build_parser() -> argparse.ArgumentParser:
16
+ parser = argparse.ArgumentParser(
17
+ prog="studyforge",
18
+ description="Extract study vocabulary from English PDF files.",
19
+ )
20
+ parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
21
+ subparsers = parser.add_subparsers(dest="command", required=True)
22
+
23
+ extract = subparsers.add_parser(
24
+ "extract",
25
+ help="Extract vocabulary from a PDF.",
26
+ )
27
+ extract.add_argument("pdf", type=Path, help="Input PDF path.")
28
+ extract.add_argument("--limit", type=int, default=30, help="Maximum word count.")
29
+ extract.add_argument(
30
+ "--format",
31
+ choices=EXPORT_FORMATS,
32
+ default="anki",
33
+ help="Output format (default: anki).",
34
+ )
35
+ extract.add_argument(
36
+ "--mode",
37
+ choices=VALID_MODES,
38
+ default="balanced",
39
+ help="Vocabulary ranking mode, including ielts.",
40
+ )
41
+ extract.add_argument(
42
+ "--min-occurrences",
43
+ type=int,
44
+ default=1,
45
+ help="Minimum occurrences in the PDF.",
46
+ )
47
+ extract.add_argument(
48
+ "-o",
49
+ "--output",
50
+ type=str,
51
+ help="Output path. Use '-' to write to stdout.",
52
+ )
53
+ return parser
54
+
55
+
56
+ def main(argv: Sequence[str] | None = None) -> int:
57
+ parser = build_parser()
58
+ args = parser.parse_args(argv)
59
+ if args.command != "extract":
60
+ parser.error("A command is required.")
61
+
62
+ try:
63
+ result = StudyForge().analyze_file(
64
+ args.pdf,
65
+ limit=args.limit,
66
+ mode=args.mode,
67
+ min_occurrences=args.min_occurrences,
68
+ )
69
+ data = export_vocabulary(result.items, args.format)
70
+ output = args.output or _default_output_path(args.pdf, args.format)
71
+ if output == "-":
72
+ sys.stdout.buffer.write(data)
73
+ return 0
74
+
75
+ output_path = Path(output).expanduser()
76
+ output_path.parent.mkdir(parents=True, exist_ok=True)
77
+ output_path.write_bytes(data)
78
+ print(f"Created {output_path} with {len(result.items)} vocabulary items.")
79
+ return 0
80
+ except (FileNotFoundError, PdfReadError, ValueError, OSError) as exc:
81
+ print(f"studyforge: error: {exc}", file=sys.stderr)
82
+ return 2
83
+
84
+
85
+ def _default_output_path(pdf_path: Path, export_format: str) -> str:
86
+ safe_stem = re.sub(r"[^\w.-]+", "_", pdf_path.stem, flags=re.UNICODE).strip("._")
87
+ suffix = output_suffix(export_format)
88
+ return f"{safe_stem or 'studyforge'}_{export_format}{suffix}"
@@ -0,0 +1 @@
1
+ """Bundled, read-only StudyForge data files."""