offerprinter 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- offerprinter/__init__.py +7 -0
- offerprinter/cli.py +523 -0
- offerprinter/config.py +183 -0
- offerprinter/controllers/__init__.py +5 -0
- offerprinter/controllers/pipeline.py +135 -0
- offerprinter/llm/__init__.py +11 -0
- offerprinter/llm/anthropic_provider.py +41 -0
- offerprinter/llm/base.py +125 -0
- offerprinter/llm/factory.py +33 -0
- offerprinter/llm/gemini_provider.py +38 -0
- offerprinter/llm/kimi_provider.py +16 -0
- offerprinter/llm/ollama_provider.py +24 -0
- offerprinter/llm/openai_provider.py +42 -0
- offerprinter/models/__init__.py +21 -0
- offerprinter/models/schemas.py +202 -0
- offerprinter/pricing.py +98 -0
- offerprinter/prompts/__init__.py +32 -0
- offerprinter/prompts/templates.py +339 -0
- offerprinter/services/__init__.py +14 -0
- offerprinter/services/cv_parser.py +84 -0
- offerprinter/services/generator.py +169 -0
- offerprinter/services/jd_fetcher.py +86 -0
- offerprinter/services/pdf_writer.py +329 -0
- offerprinter/services/tracker.py +210 -0
- offerprinter/services/writer.py +142 -0
- offerprinter/ui/__init__.py +9 -0
- offerprinter/ui/printer.py +165 -0
- offerprinter-0.2.0.dist-info/METADATA +542 -0
- offerprinter-0.2.0.dist-info/RECORD +32 -0
- offerprinter-0.2.0.dist-info/WHEEL +4 -0
- offerprinter-0.2.0.dist-info/entry_points.txt +3 -0
- offerprinter-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Read a CV from .pdf, .docx, .md, or plain text into clean text.
|
|
2
|
+
|
|
3
|
+
Everything downstream works on plain text, so this is the only place that knows
|
|
4
|
+
about file formats. All parsing is local — nothing is uploaded.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import io
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from offerprinter.models.schemas import ResumeInput
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class CVParseError(RuntimeError):
|
|
16
|
+
"""Raised when a CV file cannot be read."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _from_pdf(data: bytes) -> str:
|
|
20
|
+
from pypdf import PdfReader
|
|
21
|
+
|
|
22
|
+
reader = PdfReader(io.BytesIO(data))
|
|
23
|
+
pages = [page.extract_text() or "" for page in reader.pages]
|
|
24
|
+
return "\n\n".join(pages).strip()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _from_docx(data: bytes) -> str:
|
|
28
|
+
from docx import Document
|
|
29
|
+
|
|
30
|
+
doc = Document(io.BytesIO(data))
|
|
31
|
+
parts: list[str] = [p.text for p in doc.paragraphs]
|
|
32
|
+
# Include table cell text too — some CVs put contact details in tables.
|
|
33
|
+
for table in doc.tables:
|
|
34
|
+
for row in table.rows:
|
|
35
|
+
parts.extend(cell.text for cell in row.cells)
|
|
36
|
+
return "\n".join(part for part in parts if part.strip()).strip()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _from_text(data: bytes) -> str:
|
|
40
|
+
return data.decode("utf-8", errors="replace").strip()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
_HANDLERS = {
|
|
44
|
+
".pdf": _from_pdf,
|
|
45
|
+
".docx": _from_docx,
|
|
46
|
+
".md": _from_text,
|
|
47
|
+
".markdown": _from_text,
|
|
48
|
+
".txt": _from_text,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def extract_cv_from_bytes(data: bytes, filename: str) -> ResumeInput:
|
|
53
|
+
"""Extract CV text from raw bytes, dispatching on the filename's suffix."""
|
|
54
|
+
suffix = Path(filename).suffix.lower()
|
|
55
|
+
handler = _HANDLERS.get(suffix)
|
|
56
|
+
if handler is None:
|
|
57
|
+
# Fall back to treating unknown types as UTF-8 text.
|
|
58
|
+
handler = _from_text
|
|
59
|
+
try:
|
|
60
|
+
text = handler(data)
|
|
61
|
+
except Exception as exc: # noqa: BLE001 - surface a friendly message
|
|
62
|
+
raise CVParseError(f"Could not read CV '{filename}': {exc}") from exc
|
|
63
|
+
|
|
64
|
+
if not text.strip():
|
|
65
|
+
raise CVParseError(
|
|
66
|
+
f"No text could be extracted from '{filename}'. If it is a scanned PDF, "
|
|
67
|
+
f"paste the CV text instead."
|
|
68
|
+
)
|
|
69
|
+
return ResumeInput(text=text, source=filename)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def extract_cv(path: str | Path) -> ResumeInput:
|
|
73
|
+
"""Read a CV from a filesystem path."""
|
|
74
|
+
p = Path(path)
|
|
75
|
+
if not p.is_file():
|
|
76
|
+
raise CVParseError(f"CV file not found: {p}")
|
|
77
|
+
return extract_cv_from_bytes(p.read_bytes(), p.name)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def cv_from_text(text: str) -> ResumeInput:
|
|
81
|
+
"""Wrap already-pasted CV text."""
|
|
82
|
+
if not text.strip():
|
|
83
|
+
raise CVParseError("The pasted CV is empty.")
|
|
84
|
+
return ResumeInput(text=text.strip(), source="pasted")
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""The generation service — turns (CV, JD) into the artifacts.
|
|
2
|
+
|
|
3
|
+
This is the heart of the pipeline. It:
|
|
4
|
+
1. extracts company + role (to name the output folder and the docs),
|
|
5
|
+
2. generates each enabled artifact with a single, auditable prompt,
|
|
6
|
+
3. yields results as they complete so the CLI/WebUI can stream progress.
|
|
7
|
+
|
|
8
|
+
Artifacts are independent of one another, so by default they are generated
|
|
9
|
+
**concurrently** — a full package takes about as long as its slowest single
|
|
10
|
+
document rather than the sum of all five. Set `parallel = false` in the
|
|
11
|
+
`[generation]` config block to go back to one-at-a-time.
|
|
12
|
+
|
|
13
|
+
It holds no I/O and no file-format logic — it only orchestrates prompts and the
|
|
14
|
+
LLM provider, which keeps it easy to read and test.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from collections.abc import Iterator
|
|
21
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
22
|
+
|
|
23
|
+
from offerprinter.llm.base import LLMProvider
|
|
24
|
+
from offerprinter.models.schemas import (
|
|
25
|
+
Artifact,
|
|
26
|
+
FitScore,
|
|
27
|
+
GenerationConfig,
|
|
28
|
+
JobDescription,
|
|
29
|
+
Locale,
|
|
30
|
+
ResumeInput,
|
|
31
|
+
)
|
|
32
|
+
from offerprinter.prompts import (
|
|
33
|
+
ATS_REPORT_PROMPT,
|
|
34
|
+
COVER_LETTER_PROMPT,
|
|
35
|
+
EXTRACT_META_PROMPT,
|
|
36
|
+
FIT_MEMO_PROMPT,
|
|
37
|
+
FIT_SCORE_PROMPT,
|
|
38
|
+
INTERVIEW_PREP_PROMPT,
|
|
39
|
+
ROAST_PROMPT,
|
|
40
|
+
TAILORED_CV_PROMPT,
|
|
41
|
+
build_system,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
# key -> (human title, base filename, prompt template)
|
|
45
|
+
_ARTIFACT_SPECS: dict[str, tuple[str, str, str]] = {
|
|
46
|
+
"tailored_cv": ("Tailored CV", "tailored-cv", TAILORED_CV_PROMPT),
|
|
47
|
+
"cover_letter": ("Cover Letter", "cover-letter", COVER_LETTER_PROMPT),
|
|
48
|
+
"fit_memo": ("Fit Memo", "fit-memo", FIT_MEMO_PROMPT),
|
|
49
|
+
"ats_report": ("ATS Keyword Report", "ats-keyword-report", ATS_REPORT_PROMPT),
|
|
50
|
+
"interview_prep": ("Interview Prep Pack", "interview-prep-pack", INTERVIEW_PREP_PROMPT),
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
_META_RE = re.compile(r"COMPANY:\s*(?P<company>.+?)\s*\|\s*ROLE:\s*(?P<role>.+?)\s*$")
|
|
54
|
+
_SCORE_RE = re.compile(r"SCORE:\s*(\d{1,3})", re.IGNORECASE)
|
|
55
|
+
_STRENGTHS_RE = re.compile(r"STRENGTHS:\s*(.+)", re.IGNORECASE)
|
|
56
|
+
_GAPS_RE = re.compile(r"GAPS:\s*(.+)", re.IGNORECASE)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _slugify(value: str) -> str:
|
|
60
|
+
value = value.strip().lower()
|
|
61
|
+
value = re.sub(r"[^a-z0-9]+", "-", value)
|
|
62
|
+
return value.strip("-") or "unknown"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _strip_fences(text: str) -> str:
|
|
66
|
+
"""Remove a wrapping ```markdown ... ``` fence if the model added one."""
|
|
67
|
+
stripped = text.strip()
|
|
68
|
+
if stripped.startswith("```"):
|
|
69
|
+
stripped = re.sub(r"^```[a-zA-Z]*\n", "", stripped)
|
|
70
|
+
stripped = re.sub(r"\n```$", "", stripped)
|
|
71
|
+
return stripped.strip()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _split_list(raw: str) -> list[str]:
|
|
75
|
+
"""Split a 'a ; b ; c' line into a clean list, dropping 'none'."""
|
|
76
|
+
items = [part.strip(" .-") for part in raw.split(";")]
|
|
77
|
+
return [i for i in items if i and i.lower() not in {"none", "n/a", "-"}]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def parse_fit_score(raw: str) -> FitScore:
|
|
81
|
+
"""Parse the strict SCORE/STRENGTHS/GAPS block into a FitScore.
|
|
82
|
+
|
|
83
|
+
Kept module-level and pure so it is trivial to test without a provider.
|
|
84
|
+
"""
|
|
85
|
+
score_match = _SCORE_RE.search(raw)
|
|
86
|
+
score = max(0, min(100, int(score_match.group(1)))) if score_match else 0
|
|
87
|
+
band, verdict = FitScore.band_for(score)
|
|
88
|
+
|
|
89
|
+
strengths_match = _STRENGTHS_RE.search(raw)
|
|
90
|
+
gaps_match = _GAPS_RE.search(raw)
|
|
91
|
+
return FitScore(
|
|
92
|
+
score=score,
|
|
93
|
+
band=band,
|
|
94
|
+
verdict=verdict,
|
|
95
|
+
strengths=_split_list(strengths_match.group(1)) if strengths_match else [],
|
|
96
|
+
gaps=_split_list(gaps_match.group(1)) if gaps_match else [],
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class Generator:
|
|
101
|
+
"""Orchestrates artifact generation for one (CV, JD) pair."""
|
|
102
|
+
|
|
103
|
+
def __init__(
|
|
104
|
+
self,
|
|
105
|
+
provider: LLMProvider,
|
|
106
|
+
locale: Locale = Locale.UK,
|
|
107
|
+
generation: GenerationConfig | None = None,
|
|
108
|
+
) -> None:
|
|
109
|
+
self.provider = provider
|
|
110
|
+
self.locale = locale
|
|
111
|
+
self.generation = generation or GenerationConfig()
|
|
112
|
+
self.system = build_system(locale.value)
|
|
113
|
+
|
|
114
|
+
# -- metadata -----------------------------------------------------------
|
|
115
|
+
|
|
116
|
+
def extract_meta(self, jd: JobDescription) -> tuple[str, str]:
|
|
117
|
+
"""Return (company, role), best-effort, using a cheap single call."""
|
|
118
|
+
raw = self.provider.complete(self.system, EXTRACT_META_PROMPT.format(jd=jd.text[:6000]))
|
|
119
|
+
match = _META_RE.search(raw.strip().splitlines()[-1] if raw.strip() else "")
|
|
120
|
+
if match:
|
|
121
|
+
return match.group("company").strip(), match.group("role").strip()
|
|
122
|
+
return "Company", "Role"
|
|
123
|
+
|
|
124
|
+
# -- generation ---------------------------------------------------------
|
|
125
|
+
|
|
126
|
+
def generate_one(
|
|
127
|
+
self, key: str, cv: ResumeInput, jd: JobDescription, company: str, role: str
|
|
128
|
+
) -> Artifact:
|
|
129
|
+
title, filename, template = _ARTIFACT_SPECS[key]
|
|
130
|
+
user_prompt = template.format(cv=cv.text, jd=jd.text, company=company, role=role)
|
|
131
|
+
content = _strip_fences(self.provider.complete(self.system, user_prompt))
|
|
132
|
+
return Artifact(key=key, title=title, filename=filename, content=content)
|
|
133
|
+
|
|
134
|
+
def iter_generate(
|
|
135
|
+
self, cv: ResumeInput, jd: JobDescription, company: str, role: str
|
|
136
|
+
) -> Iterator[Artifact]:
|
|
137
|
+
"""Yield each enabled artifact as it is produced (for streaming UIs).
|
|
138
|
+
|
|
139
|
+
In parallel mode artifacts arrive in completion order, not canonical
|
|
140
|
+
order; `ApplicationPackage.sort_artifacts()` restores the order before
|
|
141
|
+
anything is written to disk.
|
|
142
|
+
"""
|
|
143
|
+
keys = self.generation.enabled()
|
|
144
|
+
if not self.generation.parallel or len(keys) == 1:
|
|
145
|
+
for key in keys:
|
|
146
|
+
yield self.generate_one(key, cv, jd, company, role)
|
|
147
|
+
return
|
|
148
|
+
|
|
149
|
+
workers = max(1, min(self.generation.max_workers, len(keys)))
|
|
150
|
+
with ThreadPoolExecutor(max_workers=workers, thread_name_prefix="offerprinter") as pool:
|
|
151
|
+
futures = {
|
|
152
|
+
pool.submit(self.generate_one, key, cv, jd, company, role): key for key in keys
|
|
153
|
+
}
|
|
154
|
+
for future in as_completed(futures):
|
|
155
|
+
yield future.result()
|
|
156
|
+
|
|
157
|
+
# -- extras -------------------------------------------------------------
|
|
158
|
+
|
|
159
|
+
def score_fit(self, cv: ResumeInput, jd: JobDescription, company: str, role: str) -> FitScore:
|
|
160
|
+
"""Score the match 0-100, strictly from evidence in the CV."""
|
|
161
|
+
prompt = FIT_SCORE_PROMPT.format(cv=cv.text, jd=jd.text, company=company, role=role)
|
|
162
|
+
return parse_fit_score(self.provider.complete(self.system, prompt))
|
|
163
|
+
|
|
164
|
+
def roast(self, cv: ResumeInput) -> Artifact:
|
|
165
|
+
"""Blunt, funny, opt-in critique of the CV's writing."""
|
|
166
|
+
content = _strip_fences(
|
|
167
|
+
self.provider.complete(self.system, ROAST_PROMPT.format(cv=cv.text))
|
|
168
|
+
)
|
|
169
|
+
return Artifact(key="roast", title="CV Roast", filename="roast", content=content)
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Load a job description from a URL (fetch + extract) or from pasted text.
|
|
2
|
+
|
|
3
|
+
URL fetching is the only outbound network call besides the LLM provider, and it
|
|
4
|
+
only happens when the user actually passes a URL.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
|
|
11
|
+
import httpx
|
|
12
|
+
|
|
13
|
+
from offerprinter.models.schemas import JobDescription
|
|
14
|
+
|
|
15
|
+
_URL_RE = re.compile(r"^https?://", re.IGNORECASE)
|
|
16
|
+
|
|
17
|
+
_USER_AGENT = (
|
|
18
|
+
"Mozilla/5.0 (compatible; OfferPrinter/0.1; +https://github.com/mohitagw15856/offerprinter)"
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class JDFetchError(RuntimeError):
|
|
23
|
+
"""Raised when a job description URL cannot be fetched or parsed."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def looks_like_url(value: str) -> bool:
|
|
27
|
+
return bool(_URL_RE.match(value.strip()))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _extract_main_text(html: str) -> str:
|
|
31
|
+
"""Turn a page of HTML into readable plain text.
|
|
32
|
+
|
|
33
|
+
Uses selectolax (a fast C HTML parser). We strip script/style/nav/footer
|
|
34
|
+
noise and collapse whitespace. This is intentionally simple and dependency
|
|
35
|
+
light; for stubborn sites, users can always paste the JD text directly.
|
|
36
|
+
"""
|
|
37
|
+
from selectolax.parser import HTMLParser
|
|
38
|
+
|
|
39
|
+
tree = HTMLParser(html)
|
|
40
|
+
for tag in tree.css("script, style, noscript, nav, header, footer, svg, form"):
|
|
41
|
+
tag.decompose()
|
|
42
|
+
|
|
43
|
+
# Prefer a <main> or <article> block if present, else fall back to <body>.
|
|
44
|
+
node = tree.css_first("main") or tree.css_first("article") or tree.body
|
|
45
|
+
text = node.text(separator="\n") if node else tree.text(separator="\n")
|
|
46
|
+
|
|
47
|
+
# Collapse runs of blank lines and trailing whitespace.
|
|
48
|
+
lines = [ln.strip() for ln in text.splitlines()]
|
|
49
|
+
cleaned = "\n".join(ln for ln in lines if ln)
|
|
50
|
+
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned)
|
|
51
|
+
return cleaned.strip()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def fetch_jd_from_url(url: str, timeout: float = 30.0) -> JobDescription:
|
|
55
|
+
"""Fetch a URL and extract its job-description text."""
|
|
56
|
+
try:
|
|
57
|
+
with httpx.Client(
|
|
58
|
+
timeout=timeout, follow_redirects=True, headers={"User-Agent": _USER_AGENT}
|
|
59
|
+
) as client:
|
|
60
|
+
resp = client.get(url)
|
|
61
|
+
except httpx.HTTPError as exc:
|
|
62
|
+
raise JDFetchError(f"Could not fetch job URL: {exc}") from exc
|
|
63
|
+
|
|
64
|
+
if resp.status_code >= 400:
|
|
65
|
+
raise JDFetchError(f"Job URL returned HTTP {resp.status_code}.")
|
|
66
|
+
|
|
67
|
+
text = _extract_main_text(resp.text)
|
|
68
|
+
if len(text) < 100:
|
|
69
|
+
raise JDFetchError(
|
|
70
|
+
"Fetched the page but found very little text. The site may require "
|
|
71
|
+
"JavaScript or block bots — please paste the job description instead."
|
|
72
|
+
)
|
|
73
|
+
return JobDescription(text=text, source=url)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def jd_from_text(text: str) -> JobDescription:
|
|
77
|
+
if not text.strip():
|
|
78
|
+
raise JDFetchError("The pasted job description is empty.")
|
|
79
|
+
return JobDescription(text=text.strip(), source="pasted")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def load_job_description(value: str, timeout: float = 30.0) -> JobDescription:
|
|
83
|
+
"""Accept either a URL or pasted text and return a JobDescription."""
|
|
84
|
+
if looks_like_url(value):
|
|
85
|
+
return fetch_jd_from_url(value.strip(), timeout=timeout)
|
|
86
|
+
return jd_from_text(value)
|
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""A tiny, dependency-free PDF writer for ATS-friendly documents.
|
|
2
|
+
|
|
3
|
+
Recruiters ask for PDFs; Applicant Tracking Systems need those PDFs to contain
|
|
4
|
+
real, selectable text in a standard font — not an image, not an exotic embedded
|
|
5
|
+
typeface. That is a narrow enough target that we can write the PDF ourselves in
|
|
6
|
+
about 200 lines rather than pulling in a rendering engine.
|
|
7
|
+
|
|
8
|
+
What it supports, matching the Markdown the generator actually produces:
|
|
9
|
+
headings (`#`, `##`, `###`), bullet and numbered lists, blockquotes, horizontal
|
|
10
|
+
rules, blank lines, and `**bold**` spans inside any of them. Everything is set
|
|
11
|
+
in Helvetica — one of the 14 PDF base fonts, so nothing needs embedding and
|
|
12
|
+
every ATS on earth can read it.
|
|
13
|
+
|
|
14
|
+
Deliberately not supported: images, tables, columns, colour. Those are exactly
|
|
15
|
+
the things that break ATS parsing, so their absence is a feature.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
# --- page geometry (A4, in PostScript points) -------------------------------
|
|
24
|
+
|
|
25
|
+
PAGE_WIDTH = 595.28
|
|
26
|
+
PAGE_HEIGHT = 841.89
|
|
27
|
+
MARGIN_X = 56.0 # ~2cm
|
|
28
|
+
MARGIN_TOP = 56.0
|
|
29
|
+
MARGIN_BOTTOM = 56.0
|
|
30
|
+
|
|
31
|
+
BODY_SIZE = 10.5
|
|
32
|
+
LINE_GAP = 1.34 # line height as a multiple of font size
|
|
33
|
+
|
|
34
|
+
#: (font size, space above, space below, bold) per heading level.
|
|
35
|
+
_HEADING_STYLE = {
|
|
36
|
+
1: (18.0, 0.0, 9.0, True),
|
|
37
|
+
2: (13.5, 12.0, 5.0, True),
|
|
38
|
+
3: (11.5, 9.0, 3.0, True),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
_BOLD_RE = re.compile(r"\*\*(.+?)\*\*")
|
|
42
|
+
_ORDERED_RE = re.compile(r"^(\d+)\.\s+(.*)$")
|
|
43
|
+
_HR_RE = re.compile(r"^\s*([-*_])\s*(\1\s*){2,}$")
|
|
44
|
+
|
|
45
|
+
# Helvetica / Helvetica-Bold advance widths for ASCII 32-126, in 1/1000 em.
|
|
46
|
+
# Taken from the standard Adobe font metrics so wrapping matches what a reader
|
|
47
|
+
# will actually see.
|
|
48
|
+
_W_REGULAR = (
|
|
49
|
+
"278 278 355 556 556 889 667 191 333 333 389 584 278 333 278 278 "
|
|
50
|
+
"556 556 556 556 556 556 556 556 556 556 278 278 584 584 584 556 "
|
|
51
|
+
"1015 667 667 722 722 667 611 778 722 278 500 667 556 833 722 778 "
|
|
52
|
+
"667 778 722 667 611 722 667 944 667 667 611 278 278 278 469 556 "
|
|
53
|
+
"333 556 556 500 556 556 278 556 556 222 222 500 222 833 556 556 "
|
|
54
|
+
"556 556 333 500 278 556 500 722 500 500 500 334 260 334 584"
|
|
55
|
+
)
|
|
56
|
+
_W_BOLD = (
|
|
57
|
+
"278 333 474 556 556 889 722 238 333 333 389 584 278 333 278 278 "
|
|
58
|
+
"556 556 556 556 556 556 556 556 556 556 333 333 584 584 584 611 "
|
|
59
|
+
"975 722 722 722 722 667 611 778 722 278 556 722 611 833 722 778 "
|
|
60
|
+
"667 778 722 667 611 722 667 944 667 667 611 333 278 333 584 556 "
|
|
61
|
+
"333 556 611 556 611 556 333 611 611 278 278 556 278 889 611 611 "
|
|
62
|
+
"611 611 389 556 333 611 556 778 556 556 500 389 280 389 584"
|
|
63
|
+
)
|
|
64
|
+
_WIDTHS = {
|
|
65
|
+
False: [int(w) for w in _W_REGULAR.split()],
|
|
66
|
+
True: [int(w) for w in _W_BOLD.split()],
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
# Unicode the generator legitimately emits -> WinAnsi-safe equivalents.
|
|
70
|
+
_TRANSLITERATE = {
|
|
71
|
+
"—": "-",
|
|
72
|
+
"–": "-",
|
|
73
|
+
"‘": "'",
|
|
74
|
+
"’": "'",
|
|
75
|
+
"“": '"',
|
|
76
|
+
"”": '"',
|
|
77
|
+
"…": "...",
|
|
78
|
+
" ": " ",
|
|
79
|
+
"•": "-",
|
|
80
|
+
"→": "->",
|
|
81
|
+
"✓": "v",
|
|
82
|
+
"✗": "x",
|
|
83
|
+
"·": "-",
|
|
84
|
+
"█": "#",
|
|
85
|
+
"░": ".",
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _clean(text: str) -> str:
|
|
90
|
+
for src, dst in _TRANSLITERATE.items():
|
|
91
|
+
text = text.replace(src, dst)
|
|
92
|
+
return text
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _text_width(text: str, size: float, bold: bool) -> float:
|
|
96
|
+
"""Width of a string in points at a given size."""
|
|
97
|
+
widths = _WIDTHS[bold]
|
|
98
|
+
total = 0
|
|
99
|
+
for ch in text:
|
|
100
|
+
code = ord(ch)
|
|
101
|
+
total += widths[code - 32] if 32 <= code <= 126 else 556
|
|
102
|
+
return total * size / 1000.0
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _escape(text: str) -> bytes:
|
|
106
|
+
"""Escape a string for a PDF literal and encode as WinAnsi (cp1252)."""
|
|
107
|
+
out = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)")
|
|
108
|
+
return out.encode("cp1252", errors="replace")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
# --- inline runs -------------------------------------------------------------
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _split_bold(text: str) -> list[tuple[str, bool]]:
|
|
115
|
+
"""Split `a **b** c` into [("a ", False), ("b", True), (" c", False)]."""
|
|
116
|
+
runs: list[tuple[str, bool]] = []
|
|
117
|
+
pos = 0
|
|
118
|
+
for match in _BOLD_RE.finditer(text):
|
|
119
|
+
if match.start() > pos:
|
|
120
|
+
runs.append((text[pos : match.start()], False))
|
|
121
|
+
runs.append((match.group(1), True))
|
|
122
|
+
pos = match.end()
|
|
123
|
+
if pos < len(text):
|
|
124
|
+
runs.append((text[pos:], False))
|
|
125
|
+
return runs or [("", False)]
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _wrap_runs(
|
|
129
|
+
runs: list[tuple[str, bool]], size: float, max_width: float
|
|
130
|
+
) -> list[list[tuple[str, bool]]]:
|
|
131
|
+
"""Greedy word-wrap over styled runs, preserving each word's style."""
|
|
132
|
+
words: list[tuple[str, bool]] = []
|
|
133
|
+
for text, bold in runs:
|
|
134
|
+
for i, word in enumerate(text.split(" ")):
|
|
135
|
+
if word or i == 0:
|
|
136
|
+
words.append((word, bold))
|
|
137
|
+
|
|
138
|
+
lines: list[list[tuple[str, bool]]] = []
|
|
139
|
+
current: list[tuple[str, bool]] = []
|
|
140
|
+
width = 0.0
|
|
141
|
+
for word, bold in words:
|
|
142
|
+
if not word:
|
|
143
|
+
continue
|
|
144
|
+
space = _text_width(" ", size, bold) if current else 0.0
|
|
145
|
+
word_width = _text_width(word, size, bold)
|
|
146
|
+
if current and width + space + word_width > max_width:
|
|
147
|
+
lines.append(current)
|
|
148
|
+
current, width = [(word, bold)], word_width
|
|
149
|
+
else:
|
|
150
|
+
current.append((word, bold))
|
|
151
|
+
width += space + word_width
|
|
152
|
+
if current:
|
|
153
|
+
lines.append(current)
|
|
154
|
+
return lines or [[]]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
# --- the document builder ----------------------------------------------------
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
class _Doc:
|
|
161
|
+
"""Accumulates page content streams, breaking pages as it fills."""
|
|
162
|
+
|
|
163
|
+
def __init__(self) -> None:
|
|
164
|
+
self.pages: list[list[bytes]] = [[]]
|
|
165
|
+
self.y = PAGE_HEIGHT - MARGIN_TOP
|
|
166
|
+
|
|
167
|
+
@property
|
|
168
|
+
def _current(self) -> list[bytes]:
|
|
169
|
+
return self.pages[-1]
|
|
170
|
+
|
|
171
|
+
def _ensure_space(self, needed: float) -> None:
|
|
172
|
+
if self.y - needed < MARGIN_BOTTOM:
|
|
173
|
+
self.pages.append([])
|
|
174
|
+
self.y = PAGE_HEIGHT - MARGIN_TOP
|
|
175
|
+
|
|
176
|
+
def space(self, amount: float) -> None:
|
|
177
|
+
if amount and self.y < PAGE_HEIGHT - MARGIN_TOP:
|
|
178
|
+
self.y -= amount
|
|
179
|
+
|
|
180
|
+
def rule(self) -> None:
|
|
181
|
+
self._ensure_space(12)
|
|
182
|
+
self.y -= 6
|
|
183
|
+
self._current.append(
|
|
184
|
+
b"0.75 w 0.6 0.6 0.6 RG %.2f %.2f m %.2f %.2f l S"
|
|
185
|
+
% (MARGIN_X, self.y, PAGE_WIDTH - MARGIN_X, self.y)
|
|
186
|
+
)
|
|
187
|
+
self.y -= 8
|
|
188
|
+
|
|
189
|
+
def paragraph(
|
|
190
|
+
self,
|
|
191
|
+
runs: list[tuple[str, bool]],
|
|
192
|
+
size: float = BODY_SIZE,
|
|
193
|
+
indent: float = 0.0,
|
|
194
|
+
bullet: str = "",
|
|
195
|
+
force_bold: bool = False,
|
|
196
|
+
grey: bool = False,
|
|
197
|
+
) -> None:
|
|
198
|
+
"""Lay out one wrapped block of text."""
|
|
199
|
+
if force_bold:
|
|
200
|
+
runs = [(text, True) for text, _ in runs]
|
|
201
|
+
max_width = PAGE_WIDTH - 2 * MARGIN_X - indent
|
|
202
|
+
lines = _wrap_runs(runs, size, max_width)
|
|
203
|
+
leading = size * LINE_GAP
|
|
204
|
+
colour = b"0.35 0.35 0.35 rg" if grey else b"0 0 0 rg"
|
|
205
|
+
|
|
206
|
+
for index, line in enumerate(lines):
|
|
207
|
+
self._ensure_space(leading)
|
|
208
|
+
self.y -= leading
|
|
209
|
+
x = MARGIN_X + indent
|
|
210
|
+
parts: list[bytes] = [colour]
|
|
211
|
+
|
|
212
|
+
if bullet and index == 0:
|
|
213
|
+
parts.append(
|
|
214
|
+
b"BT /F1 %.2f Tf 1 0 0 1 %.2f %.2f Tm (%s) Tj ET"
|
|
215
|
+
% (size, MARGIN_X + indent - 14.0, self.y, _escape(bullet))
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
for word_index, (word, bold) in enumerate(line):
|
|
219
|
+
if word_index:
|
|
220
|
+
x += _text_width(" ", size, bold)
|
|
221
|
+
font = b"/F2" if bold else b"/F1"
|
|
222
|
+
parts.append(
|
|
223
|
+
b"BT %s %.2f Tf 1 0 0 1 %.2f %.2f Tm (%s) Tj ET"
|
|
224
|
+
% (font, size, x, self.y, _escape(word))
|
|
225
|
+
)
|
|
226
|
+
x += _text_width(word, size, bold)
|
|
227
|
+
self._current.append(b"\n".join(parts))
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def markdown_to_pdf_bytes(markdown: str) -> bytes:
|
|
231
|
+
"""Render OfferPrinter-flavoured Markdown to a complete PDF file."""
|
|
232
|
+
doc = _Doc()
|
|
233
|
+
|
|
234
|
+
for raw_line in _clean(markdown).splitlines():
|
|
235
|
+
line = raw_line.rstrip()
|
|
236
|
+
stripped = line.strip()
|
|
237
|
+
|
|
238
|
+
if not stripped:
|
|
239
|
+
doc.space(BODY_SIZE * 0.55)
|
|
240
|
+
continue
|
|
241
|
+
if _HR_RE.match(stripped):
|
|
242
|
+
doc.rule()
|
|
243
|
+
continue
|
|
244
|
+
if stripped.startswith("```"):
|
|
245
|
+
continue # fences never survive into the output; skip stray ones
|
|
246
|
+
|
|
247
|
+
heading = re.match(r"^(#{1,3})\s+(.*)$", stripped)
|
|
248
|
+
if heading:
|
|
249
|
+
level = len(heading.group(1))
|
|
250
|
+
size, above, below, bold = _HEADING_STYLE[level]
|
|
251
|
+
doc.space(above)
|
|
252
|
+
doc.paragraph(_split_bold(heading.group(2)), size=size, force_bold=bold)
|
|
253
|
+
doc.space(below)
|
|
254
|
+
continue
|
|
255
|
+
|
|
256
|
+
if stripped.startswith("> "):
|
|
257
|
+
doc.paragraph(_split_bold(stripped[2:]), indent=16.0, grey=True)
|
|
258
|
+
continue
|
|
259
|
+
|
|
260
|
+
if stripped.startswith(("- ", "* ")):
|
|
261
|
+
indent = 14.0 + (len(line) - len(line.lstrip())) * 0.5
|
|
262
|
+
doc.paragraph(_split_bold(stripped[2:]), indent=indent, bullet="-")
|
|
263
|
+
continue
|
|
264
|
+
|
|
265
|
+
ordered = _ORDERED_RE.match(stripped)
|
|
266
|
+
if ordered:
|
|
267
|
+
doc.paragraph(_split_bold(ordered.group(2)), indent=18.0, bullet=f"{ordered.group(1)}.")
|
|
268
|
+
continue
|
|
269
|
+
|
|
270
|
+
doc.paragraph(_split_bold(stripped))
|
|
271
|
+
|
|
272
|
+
return _assemble(doc.pages)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _assemble(pages: list[list[bytes]]) -> bytes:
|
|
276
|
+
"""Serialise pages into a valid PDF with a correct cross-reference table."""
|
|
277
|
+
pages = pages or [[]]
|
|
278
|
+
page_count = len(pages)
|
|
279
|
+
|
|
280
|
+
# Object numbering: 1 catalog, 2 pages tree, 3 + 4 fonts,
|
|
281
|
+
# then per page: a page object and its content stream.
|
|
282
|
+
first_page_obj = 5
|
|
283
|
+
objects: dict[int, bytes] = {}
|
|
284
|
+
|
|
285
|
+
kids = " ".join(f"{first_page_obj + i * 2} 0 R" for i in range(page_count))
|
|
286
|
+
objects[1] = b"<< /Type /Catalog /Pages 2 0 R >>"
|
|
287
|
+
objects[2] = f"<< /Type /Pages /Count {page_count} /Kids [{kids}] >>".encode("ascii")
|
|
288
|
+
objects[3] = (
|
|
289
|
+
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
|
|
290
|
+
)
|
|
291
|
+
objects[4] = (
|
|
292
|
+
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica-Bold /Encoding /WinAnsiEncoding >>"
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
for i, content in enumerate(pages):
|
|
296
|
+
page_obj = first_page_obj + i * 2
|
|
297
|
+
stream_obj = page_obj + 1
|
|
298
|
+
objects[page_obj] = (
|
|
299
|
+
f"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 {PAGE_WIDTH:.2f} {PAGE_HEIGHT:.2f}] "
|
|
300
|
+
f"/Resources << /Font << /F1 3 0 R /F2 4 0 R >> >> "
|
|
301
|
+
f"/Contents {stream_obj} 0 R >>"
|
|
302
|
+
).encode("ascii")
|
|
303
|
+
stream = b"\n".join(content)
|
|
304
|
+
objects[stream_obj] = (
|
|
305
|
+
f"<< /Length {len(stream)} >>\nstream\n".encode("ascii") + stream + b"\nendstream"
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
out = bytearray(b"%PDF-1.4\n%\xe2\xe3\xcf\xd3\n")
|
|
309
|
+
offsets: dict[int, int] = {}
|
|
310
|
+
for number in sorted(objects):
|
|
311
|
+
offsets[number] = len(out)
|
|
312
|
+
out += f"{number} 0 obj\n".encode("ascii") + objects[number] + b"\nendobj\n"
|
|
313
|
+
|
|
314
|
+
xref_offset = len(out)
|
|
315
|
+
total = max(objects) + 1
|
|
316
|
+
out += f"xref\n0 {total}\n".encode("ascii")
|
|
317
|
+
out += b"0000000000 65535 f \n"
|
|
318
|
+
for number in range(1, total):
|
|
319
|
+
out += f"{offsets[number]:010d} 00000 n \n".encode("ascii")
|
|
320
|
+
out += (f"trailer\n<< /Size {total} /Root 1 0 R >>\nstartxref\n{xref_offset}\n%%EOF\n").encode(
|
|
321
|
+
"ascii"
|
|
322
|
+
)
|
|
323
|
+
return bytes(out)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def write_pdf(markdown: str, path: Path) -> Path:
|
|
327
|
+
"""Render Markdown to a PDF file on disk."""
|
|
328
|
+
path.write_bytes(markdown_to_pdf_bytes(markdown))
|
|
329
|
+
return path
|