markdownizer 0.4.0__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {markdownizer-0.4.0 → markdownizer-0.4.2}/PKG-INFO +1 -1
  2. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/__init__.py +1 -1
  3. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/cli.py +3 -1
  4. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/imports.py +10 -2
  5. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/ir.py +2 -2
  6. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/rank.py +4 -0
  7. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/slice.py +87 -80
  8. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/tokens.py +1 -1
  9. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/parser.py +15 -2
  10. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/PKG-INFO +1 -1
  11. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_cli.py +24 -0
  12. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_extractor.py +12 -0
  13. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_imports.py +39 -0
  14. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_ir.py +31 -0
  15. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_optimizer.py +73 -0
  16. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_parser.py +24 -0
  17. {markdownizer-0.4.0 → markdownizer-0.4.2}/LICENSE +0 -0
  18. {markdownizer-0.4.0 → markdownizer-0.4.2}/README.md +0 -0
  19. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/__main__.py +0 -0
  20. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/__init__.py +0 -0
  21. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/base.py +0 -0
  22. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/compact.py +0 -0
  23. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/json.py +0 -0
  24. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/markdown.py +0 -0
  25. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/classifier.py +0 -0
  26. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/extractor.py +0 -0
  27. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/__init__.py +0 -0
  28. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/profiles.py +0 -0
  29. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/py.typed +0 -0
  30. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/renderer.py +0 -0
  31. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/scanner.py +0 -0
  32. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/SOURCES.txt +0 -0
  33. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/dependency_links.txt +0 -0
  34. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/entry_points.txt +0 -0
  35. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/requires.txt +0 -0
  36. {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/top_level.txt +0 -0
  37. {markdownizer-0.4.0 → markdownizer-0.4.2}/pyproject.toml +0 -0
  38. {markdownizer-0.4.0 → markdownizer-0.4.2}/setup.cfg +0 -0
  39. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_backends.py +0 -0
  40. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_classifier.py +0 -0
  41. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_packaging.py +0 -0
  42. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_ranking.py +0 -0
  43. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_renderer.py +0 -0
  44. {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_scanner.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: markdownizer
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Extract documentation from Python projects into Markdown without rewriting it
5
5
  Author-email: Mohammad Hasan Khoddami <mohammadh.khoddami@gmail.com>
6
6
  License: MIT License
@@ -8,7 +8,7 @@ from markdownizer.extractor import extract_project
8
8
  from markdownizer.ir import IR_VERSION, ProjectIR, build_project_ir
9
9
  from markdownizer.optimizer import OptimizedContext, optimize_context
10
10
 
11
- __version__ = "0.4.0"
11
+ __version__ = "0.4.2"
12
12
  __all__ = [
13
13
  "extract_project",
14
14
  "build_project_ir",
@@ -83,7 +83,9 @@ def build_parser() -> argparse.ArgumentParser:
83
83
  description=(
84
84
  "Extract existing documentation from a Python project into "
85
85
  "Markdown files. Use 'markdownizer build <project>' for format "
86
- "selection; the legacy form below is kept as a compatibility alias."
86
+ "selection; the legacy form below is kept as a compatibility "
87
+ "alias. If your project directory is named 'build', 'context', "
88
+ "or 'stats', prefix it with './'."
87
89
  ),
88
90
  )
89
91
  parser.set_defaults(command="build")
@@ -67,11 +67,17 @@ def _resolve_relative(
67
67
 
68
68
 
69
69
  def _collect_statement_imports(
70
- node: ast.AST, module_map: dict[str, str], from_rel_path: str
70
+ module: ast.Module, module_map: dict[str, str], from_rel_path: str
71
71
  ) -> list[ImportEdge]:
72
+ """Collect module-level import statements.
73
+
74
+ Only top-level statements count: imports inside function bodies, class
75
+ bodies, or ``if __name__ == "__main__":`` blocks are not structural
76
+ module dependencies. ``from __future__`` imports are filtered out.
77
+ """
72
78
  edges: list[ImportEdge] = []
73
79
 
74
- for child in ast.walk(node):
80
+ for child in module.body:
75
81
  if isinstance(child, ast.Import):
76
82
  for alias in child.names:
77
83
  target = _resolve_absolute(alias.name, module_map)
@@ -87,6 +93,8 @@ def _collect_statement_imports(
87
93
  )
88
94
  )
89
95
  elif isinstance(child, ast.ImportFrom):
96
+ if child.module == "__future__":
97
+ continue
90
98
  if child.module is None:
91
99
  # ``from . import x`` — the imported names are resolved
92
100
  # relative to the current package.
@@ -27,7 +27,7 @@ from typing import Any
27
27
 
28
28
  from markdownizer.classifier import classify
29
29
  from markdownizer.imports import ImportEdge, collect_import_edges
30
- from markdownizer.parser import DocObject, parse_file
30
+ from markdownizer.parser import DocObject, parse_file, read_python_source
31
31
  from markdownizer.scanner import scan_python_files
32
32
 
33
33
  IR_VERSION = 1
@@ -336,7 +336,7 @@ def build_project_ir(project_root: Path, exclude: list[str] | None = None) -> Pr
336
336
  for py_file in files:
337
337
  rel = py_file.relative_to(project_root).as_posix()
338
338
  dotted = _module_dotted_name(rel)
339
- source = py_file.read_text(encoding="utf-8")
339
+ source = read_python_source(py_file)
340
340
  module_obj, objects = parse_file(str(py_file), source=source)
341
341
 
342
342
  module_symbols = [_symbol_from_docobject(o, rel) for o in objects]
@@ -11,6 +11,10 @@ Methods:
11
11
  (fixed iteration count for full determinism)
12
12
  * ``fanout`` — in-degree of each module in the import graph
13
13
  * ``simple`` — uniform file scores (no graph)
14
+
15
+ Rank values are deterministic for a given platform (IEEE-754 addition
16
+ order); the IR hash excludes ranks, so ``ir.hash`` is stable across
17
+ platforms even if floating-point tie-breaking ever differs.
14
18
  """
15
19
 
16
20
  from __future__ import annotations
@@ -4,12 +4,14 @@ Turns a ranked ProjectIR into an AI-ready context artifact within a token
4
4
  budget, using deterministic layered emission:
5
5
 
6
6
  * layer 0 — project index (packages, modules, docstrings)
7
- * layer 1 — signatures + docstrings for public symbols
8
- * layer 2 — full source for ranked symbols
9
- * layer 3 — remaining symbol docstrings/comments when budget allows
10
-
11
- High-ranked symbols are placed at the start and end of the artifact
12
- (primacy/recency bias); lower-ranked content sits in the middle.
7
+ * layer 1 — per-symbol blocks: full source for the highest-ranked symbols,
8
+ signature-only for the rest (or docstrings only for ``include_source``
9
+ profiles without source)
10
+
11
+ Each symbol is emitted exactly once. High-ranked symbols are placed at the
12
+ start and end of the artifact (primacy/recency bias); lower-ranked content
13
+ sits in the middle. The reported token estimate counts the complete
14
+ artifact, header included.
13
15
  """
14
16
 
15
17
  from __future__ import annotations
@@ -102,6 +104,55 @@ def _block_tokens(text: str, prefer_tiktoken: bool) -> float:
102
104
  return count_tokens(text, prefer_tiktoken=prefer_tiktoken)
103
105
 
104
106
 
107
+ def _symbol_block(
108
+ symbol: Symbol,
109
+ profile: Profile,
110
+ prefer_tiktoken: bool,
111
+ budget: float,
112
+ ) -> tuple[str, float] | None:
113
+ """Render one symbol exactly once; ``None`` when nothing fits the budget."""
114
+ comments = profile.include_comments
115
+ if profile.include_source is True:
116
+ src_text = render_object(
117
+ symbol,
118
+ None,
119
+ include_source=True,
120
+ include_comments=comments,
121
+ )
122
+ src_cost = _block_tokens(src_text, prefer_tiktoken)
123
+ if src_cost <= budget:
124
+ return src_text, src_cost
125
+ sig_text = render_object(
126
+ symbol,
127
+ None,
128
+ include_source="signature",
129
+ include_comments=comments,
130
+ )
131
+ sig_cost = _block_tokens(sig_text, prefer_tiktoken)
132
+ return (sig_text, sig_cost) if sig_cost <= budget else None
133
+
134
+ if profile.include_source == "signature":
135
+ sig_text = render_object(
136
+ symbol,
137
+ None,
138
+ include_source="signature",
139
+ include_comments=comments,
140
+ )
141
+ sig_cost = _block_tokens(sig_text, prefer_tiktoken)
142
+ return (sig_text, sig_cost) if sig_cost <= budget else None
143
+
144
+ if symbol.docstring:
145
+ doc_text = render_object(
146
+ symbol,
147
+ None,
148
+ include_source=False,
149
+ include_comments=comments,
150
+ )
151
+ doc_cost = _block_tokens(doc_text, prefer_tiktoken)
152
+ return (doc_text, doc_cost) if doc_cost <= budget else None
153
+ return None
154
+
155
+
105
156
  def optimize_context(
106
157
  ir: ProjectIR,
107
158
  max_tokens: float,
@@ -114,19 +165,20 @@ def optimize_context(
114
165
 
115
166
  Args:
116
167
  ir: The project IR (ranked in place by this function).
117
- max_tokens: Target token budget (soft limit, ±10% slop).
168
+ max_tokens: Target token budget (soft limit, ±10% slop). Must be > 0.
118
169
  profile: Profile preset name (see :mod:`markdownizer.optimizer.profiles`).
119
170
  query: Optional space-separated keyword prefilter on symbol names,
120
171
  frameworks, and docstrings (deterministic substring match).
121
172
  rank_method: ``pagerank`` (default), ``fanout``, or ``simple``.
122
173
  prefer_tiktoken: Use ``tiktoken`` when installed for token estimates.
174
+
175
+ Raises:
176
+ ValueError: If ``max_tokens`` is not positive.
123
177
  """
178
+ if float(max_tokens) <= 0:
179
+ raise ValueError(f"max_tokens must be positive, got {max_tokens!r}")
124
180
  rank_ir(ir, method=rank_method)
125
181
  profile_obj = get_profile(profile)
126
- budget = float(max_tokens) * _BUDGET_SLOP
127
- parts: list[str] = []
128
- total: float = 0.0
129
- included: int = 0
130
182
  total_symbols = len(ir.symbols)
131
183
 
132
184
  header = (
@@ -134,90 +186,45 @@ def optimize_context(
134
186
  f"(profile={profile_obj.name}, rank={rank_method}, "
135
187
  f"max_tokens={int(max_tokens)})"
136
188
  )
189
+ header_cost = _block_tokens(header, prefer_tiktoken)
190
+ budget = float(max_tokens) * _BUDGET_SLOP - header_cost
191
+
192
+ parts: list[str] = []
193
+ total: float = 0.0
137
194
 
138
195
  index = _render_index(ir)
139
- if _block_tokens(index, prefer_tiktoken) <= budget:
196
+ index_cost = _block_tokens(index, prefer_tiktoken)
197
+ if index_cost <= budget:
140
198
  parts.append(index)
141
- total += _block_tokens(index, prefer_tiktoken)
199
+ total += index_cost
142
200
 
143
201
  module_index = _render_module_index(ir)
144
- if _block_tokens(module_index, prefer_tiktoken) <= budget:
202
+ module_cost = _block_tokens(module_index, prefer_tiktoken)
203
+ if module_cost <= budget:
145
204
  parts.append(module_index)
146
- total += _block_tokens(module_index, prefer_tiktoken)
205
+ total += module_cost
147
206
 
148
207
  symbols = _select_symbols(ir, profile_obj, query)
149
208
  ordered = _edge_place(symbols)
150
209
 
151
- signatures: list[tuple[Symbol, str, float]] = []
152
- sources: list[tuple[Symbol, str, float]] = []
153
- remainder: list[tuple[Symbol, str, float]] = []
154
-
210
+ included = 0
155
211
  for symbol in ordered:
156
- if profile_obj.include_source is not False:
157
- sig_text = render_object(
158
- symbol,
159
- None,
160
- include_source="signature",
161
- include_comments=profile_obj.include_comments,
162
- )
163
- sig_cost = _block_tokens(sig_text, prefer_tiktoken)
164
- if sig_cost <= budget:
165
- signatures.append((symbol, sig_text, sig_cost))
166
-
167
- if profile_obj.include_source is True:
168
- src_text = render_object(
169
- symbol,
170
- None,
171
- include_source=True,
172
- include_comments=profile_obj.include_comments,
173
- )
174
- src_cost = _block_tokens(src_text, prefer_tiktoken)
175
- sources.append((symbol, src_text, src_cost))
176
-
177
- if profile_obj.include_source is False and symbol.docstring:
178
- doc_text = render_object(
179
- symbol,
180
- None,
181
- include_source=False,
182
- include_comments=profile_obj.include_comments,
183
- )
184
- doc_cost = _block_tokens(doc_text, prefer_tiktoken)
185
- remainder.append((symbol, doc_text, doc_cost))
186
-
187
- if profile_obj.include_source is not False:
188
- for _, sig_text, sig_cost in signatures:
189
- if total + sig_cost > budget:
190
- break
191
- parts.append("---")
192
- parts.append("")
193
- parts.append(sig_text)
194
- total += sig_cost
195
- included += 1
196
-
197
- if profile_obj.include_source is True:
198
- for _, src_text, src_cost in sources:
199
- if total + src_cost > budget:
200
- break
201
- parts.append("---")
202
- parts.append("")
203
- parts.append(src_text)
204
- total += src_cost
205
- included += 1
206
-
207
- if profile_obj.include_source is False:
208
- for _, doc_text, doc_cost in remainder:
209
- if total + doc_cost > budget:
210
- break
211
- parts.append("---")
212
- parts.append("")
213
- parts.append(doc_text)
214
- total += doc_cost
215
- included += 1
212
+ block = _symbol_block(symbol, profile_obj, prefer_tiktoken, budget)
213
+ if block is None:
214
+ continue
215
+ text, cost = block
216
+ if total + cost > budget:
217
+ break
218
+ parts.append("---")
219
+ parts.append("")
220
+ parts.append(text)
221
+ total += cost
222
+ included += 1
216
223
 
217
224
  text = "\n".join([header, "", *parts]).rstrip() + "\n"
218
225
  return OptimizedContext(
219
226
  text=text,
220
- estimated_tokens=total,
227
+ estimated_tokens=_block_tokens(text, prefer_tiktoken),
221
228
  max_tokens=float(max_tokens),
222
229
  profile=profile_obj.name,
223
230
  rank_method=rank_method,
@@ -18,7 +18,7 @@ try:
18
18
 
19
19
  _ENCODER = _tiktoken.get_encoding("cl100k_base")
20
20
  _HAS_TIKTOKEN = True
21
- except Exception: # pragma: no cover - depends on optional dependency
21
+ except (ImportError, OSError): # pragma: no cover - depends on optional dependency
22
22
  _ENCODER = None
23
23
  _HAS_TIKTOKEN = False
24
24
 
@@ -6,6 +6,7 @@ import ast
6
6
  import io
7
7
  import tokenize
8
8
  from dataclasses import dataclass, field
9
+ from pathlib import Path
9
10
  from typing import Protocol
10
11
 
11
12
 
@@ -72,6 +73,19 @@ def _rightmost_name(name: str) -> str:
72
73
  return name.rsplit(".", 1)[-1] if name else ""
73
74
 
74
75
 
76
+ def read_python_source(path: Path) -> str:
77
+ """Read a Python source file, tolerating non-UTF-8 encodings.
78
+
79
+ Uses ``utf-8-sig`` (which also strips a UTF-8 BOM) and falls back to
80
+ latin-1, which can decode every byte sequence without raising. A single
81
+ legacy-encoded file must never abort an entire project extraction.
82
+ """
83
+ try:
84
+ return path.read_text(encoding="utf-8-sig")
85
+ except UnicodeDecodeError:
86
+ return path.read_text(encoding="latin-1")
87
+
88
+
75
89
  def _unparse_or_empty(node: ast.AST | None) -> str:
76
90
  """Render an AST node back to source text, or ``""`` if unavailable."""
77
91
  if node is None:
@@ -337,8 +351,7 @@ def parse_file(file_path: str, source: str | None = None) -> tuple[DocObject, li
337
351
  ``source`` may be passed in to avoid re-reading the file from disk.
338
352
  """
339
353
  if source is None:
340
- with open(file_path, encoding="utf-8") as fh:
341
- source = fh.read()
354
+ source = read_python_source(Path(file_path))
342
355
 
343
356
  source_lines = source.splitlines()
344
357
  all_comments = _collect_comments(source)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: markdownizer
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Extract documentation from Python projects into Markdown without rewriting it
5
5
  Author-email: Mohammad Hasan Khoddami <mohammadh.khoddami@gmail.com>
6
6
  License: MIT License
@@ -222,6 +222,30 @@ def test_stats_invalid_rank_exits(tmp_path):
222
222
  main(["stats", str(tmp_path), "--rank", "magic"])
223
223
 
224
224
 
225
+ def test_stats_empty_project(tmp_path, capsys):
226
+ assert main(["stats", str(tmp_path)]) == 0
227
+ captured = capsys.readouterr()
228
+ assert "Symbols: 0" in captured.out
229
+
230
+
231
+ def test_context_query_flag(tmp_path, capsys):
232
+ (tmp_path / "auth.py").write_text('"""Auth."""\ndef login():\n """Login."""\n pass\n')
233
+ (tmp_path / "other.py").write_text(
234
+ '"""Other."""\ndef unrelated():\n """Nope."""\n pass\n'
235
+ )
236
+ out = tmp_path / "out"
237
+ assert main(["context", str(tmp_path), "-o", str(out), "--query", "login"]) == 0
238
+ text = (out / "context.md").read_text(encoding="utf-8")
239
+ assert "login" in text
240
+ assert "unrelated" not in text
241
+
242
+
243
+ def test_context_negative_max_tokens_returns_1(tmp_path, capsys):
244
+ (tmp_path / "mod.py").write_text('"""docs"""\n')
245
+ assert main(["context", str(tmp_path), "--max-tokens", "-10"]) == 1
246
+ assert "max_tokens must be positive" in capsys.readouterr().err
247
+
248
+
225
249
  def _assert_exits_with(argv: list[str], expected: str, capsys: pytest.CaptureFixture[str]) -> None:
226
250
  with pytest.raises(SystemExit):
227
251
  main(argv)
@@ -119,3 +119,15 @@ def test_extract_error_on_unwritable_output(sample_project, tmp_path):
119
119
  pytest.skip("directory is writable; cannot simulate")
120
120
  with pytest.raises(OSError):
121
121
  extract_project(sample_project, out / "sub" / "nested")
122
+
123
+
124
+ def test_extract_survives_non_utf8_file(tmp_path):
125
+ """End-to-end: a legacy-encoded file does not abort extraction."""
126
+ (tmp_path / "good.py").write_text('"""Fine."""\ndef ok():\n pass\n')
127
+ (tmp_path / "legacy.py").write_bytes(b'"""caf\xe9"""\ndef old():\n pass\n')
128
+ out = tmp_path / "out"
129
+ written = extract_project(tmp_path, out)
130
+ assert sorted(p.name for p in written) == ["_root.md"]
131
+ md = (out / "_root.md").read_text(encoding="utf-8")
132
+ assert "# Function: old" in md
133
+ assert "café" in md
@@ -144,3 +144,42 @@ def test_dotted_import_resolution_to_init():
144
144
 
145
145
  def test_empty_source():
146
146
  assert _edges("") == []
147
+
148
+
149
+ def test_nested_function_import_not_collected():
150
+ source = "def f():\n import os\n return os.path\n"
151
+ assert _edges(source) == []
152
+
153
+
154
+ def test_main_guard_import_not_collected():
155
+ source = "import pkg.models\n\nif __name__ == '__main__':\n import sys\n"
156
+ edges = _edges(source)
157
+ assert [e.imported_module for e in edges] == ["pkg.models"]
158
+
159
+
160
+ def test_class_body_import_not_collected():
161
+ source = "class C:\n def m(self):\n import os\n"
162
+ assert _edges(source) == []
163
+
164
+
165
+ def test_future_import_filtered():
166
+ source = "from __future__ import annotations\nimport pkg.models\n"
167
+ edges = _edges(source)
168
+ assert [e.imported_module for e in edges] == ["pkg.models"]
169
+
170
+
171
+ def test_try_wrapped_import_not_collected():
172
+ """try/except-wrapped imports are conservatively treated as external."""
173
+ source = "try:\n import pkg.models\nexcept ImportError:\n pass\n"
174
+ assert _edges(source) == []
175
+
176
+
177
+ def test_latin1_source_decodes():
178
+ from markdownizer.imports import collect_import_edges
179
+
180
+ edges = collect_import_edges(
181
+ '"""caf\xe9"""\nimport os\n'.encode("latin-1").decode("latin-1"),
182
+ "mod.py",
183
+ {"mod.py": "mod.py"},
184
+ )
185
+ assert [e.imported_module for e in edges] == ["os"]
@@ -245,3 +245,34 @@ def test_unsorted_input_still_sorted(tmp_path):
245
245
  ir = build_project_ir(tmp_path)
246
246
  assert [m.path for m in ir.modules] == ["a/y.py", "b/z.py"]
247
247
  assert [s.qualified_name for s in ir.symbols] == ["y", "z"]
248
+
249
+
250
+ def test_build_survives_non_utf8_file(tmp_path):
251
+ """A single latin-1 encoded file must not abort the whole build."""
252
+ from tests.conftest import write_py
253
+
254
+ write_py(tmp_path / "good.py", '"""Fine."""\ndef ok():\n pass\n')
255
+ (tmp_path / "legacy.py").write_bytes(b'"""caf\xe9 latin-1"""\ndef old():\n pass\n')
256
+ ir = build_project_ir(tmp_path)
257
+ assert [m.path for m in ir.modules] == ["good.py", "legacy.py"]
258
+ legacy = next(m for m in ir.modules if m.path == "legacy.py")
259
+ assert legacy.docstring == "café latin-1"
260
+ assert ir.stats.symbol_count == 2
261
+
262
+
263
+ def test_build_survives_utf8_bom_file(tmp_path):
264
+ from tests.conftest import write_py
265
+
266
+ write_py(tmp_path / "a.py", "def f():\n pass\n")
267
+ (tmp_path / "bom.py").write_bytes(b'\xef\xbb\xbf"""BOM doc."""\ndef g():\n pass\n')
268
+ ir = build_project_ir(tmp_path)
269
+ assert [m.path for m in ir.modules] == ["a.py", "bom.py"]
270
+ bom = next(m for m in ir.modules if m.path == "bom.py")
271
+ assert bom.docstring == "BOM doc."
272
+
273
+
274
+ def test_non_utf8_hash_stable(tmp_path):
275
+ (tmp_path / "legacy.py").write_bytes(b'"""caf\xe9"""\ndef old():\n pass\n')
276
+ ir1 = build_project_ir(tmp_path)
277
+ ir2 = build_project_ir(tmp_path)
278
+ assert ir1.hash == ir2.hash
@@ -166,3 +166,76 @@ def test_tiny_budget_only_header(tmp_path):
166
166
  ctx = optimize_context(ir, max_tokens=1)
167
167
  assert ctx.included_symbols == 0
168
168
  assert ctx.text.startswith("# Context:")
169
+
170
+
171
+ def test_negative_budget_raises(tmp_path):
172
+ ir = _build(tmp_path)
173
+ with pytest.raises(ValueError):
174
+ optimize_context(ir, max_tokens=0)
175
+ with pytest.raises(ValueError):
176
+ optimize_context(ir, max_tokens=-50)
177
+
178
+
179
+ def test_debugging_emits_each_symbol_once(tmp_path):
180
+ """M1: full source contains the signature — never duplicate it."""
181
+ ir = _build(tmp_path)
182
+ ctx = optimize_context(ir, max_tokens=10000, profile="debugging")
183
+ assert ctx.included_symbols == ctx.total_symbols
184
+ assert "## Signature" not in ctx.text
185
+ assert ctx.text.count("## Source Code") == ctx.included_symbols
186
+
187
+
188
+ def test_debugging_falls_back_to_signature_when_source_does_not_fit(tmp_path):
189
+ """A symbol whose full source exceeds the budget still gets a signature."""
190
+ from tests.conftest import write_py
191
+
192
+ body = "\n".join(f" x{i} = {i} * 7" for i in range(200))
193
+ write_py(
194
+ tmp_path / "big.py",
195
+ f'"""Big."""\ndef huge():\n """Huge."""\n{body}\n return 1\n',
196
+ )
197
+ ir = __import__("markdownizer").build_project_ir(tmp_path)
198
+ ctx = optimize_context(ir, max_tokens=150, profile="debugging")
199
+ assert "## Signature" in ctx.text
200
+ assert "## Source Code" not in ctx.text
201
+ assert ctx.included_symbols == 1
202
+
203
+
204
+ def test_included_symbols_counts_unique_symbols(tmp_path):
205
+ """M2: each symbol counts once regardless of emission depth."""
206
+ ir = _build(tmp_path)
207
+ ctx = optimize_context(ir, max_tokens=10000, profile="debugging")
208
+ assert ctx.included_symbols == ctx.total_symbols
209
+
210
+
211
+ def test_estimated_tokens_matches_artifact(tmp_path):
212
+ """M3: the estimate counts the complete artifact, header included."""
213
+ from markdownizer.optimizer.tokens import count_tokens
214
+
215
+ ir = _build(tmp_path)
216
+ for profile in ("architecture", "api", "debugging", "onboarding"):
217
+ ctx = optimize_context(ir, max_tokens=10000, profile=profile)
218
+ assert ctx.estimated_tokens == count_tokens(ctx.text), profile
219
+
220
+
221
+ def test_edge_placement_orders_top_symbols_first(tmp_path):
222
+ """The highest-ranked symbol leads the artifact; the second leads the tail."""
223
+ ir = _build(tmp_path)
224
+ ctx = optimize_context(ir, max_tokens=10000, profile="api")
225
+ ranked = sorted(ir.symbols, key=lambda s: -s.rank)
226
+ top, second = ranked[0], ranked[1]
227
+ body = ctx.text.split("# Context:")[1]
228
+ top_index = body.find(f"# {top.framework}: {top.name}")
229
+ second_index = body.find(f"# {second.framework}: {second.name}")
230
+ assert top_index < second_index
231
+
232
+
233
+ def test_context_query_flag(tmp_path):
234
+ from tests.conftest import write_py
235
+
236
+ write_py(tmp_path / "auth.py", '"""Auth."""\ndef login():\n """Login."""\n pass\n')
237
+ write_py(tmp_path / "other.py", '"""Other."""\ndef unrelated():\n """Nope."""\n pass\n')
238
+ ir = __import__("markdownizer").build_project_ir(tmp_path)
239
+ ctx = optimize_context(ir, max_tokens=5000, profile="api", query="login")
240
+ assert "login" in ctx.text
241
+ assert "unrelated" not in ctx.text
@@ -232,3 +232,27 @@ def test_file_path_populated(write_file):
232
232
  path = write_file("mod.py", "def f():\n pass\n")
233
233
  _, objects = parse_file(str(path))
234
234
  assert Path(objects[0].file_path) == path
235
+
236
+
237
+ def test_latin1_file_parses(tmp_path):
238
+ from markdownizer.parser import read_python_source
239
+
240
+ path = tmp_path / "legacy.py"
241
+ path.write_bytes(b'"""caf\xe9"""\ndef old():\n pass\n')
242
+ source = read_python_source(path)
243
+ assert "café" in source
244
+ module, objects = parse_file(str(path))
245
+ assert module.docstring == "café"
246
+ assert objects[0].name == "old"
247
+
248
+
249
+ def test_utf8_bom_file_parses(tmp_path):
250
+ from markdownizer.parser import read_python_source
251
+
252
+ path = tmp_path / "bom.py"
253
+ path.write_bytes(b'\xef\xbb\xbf"""BOM."""\ndef g():\n pass\n')
254
+ source = read_python_source(path)
255
+ assert not source.startswith("\ufeff")
256
+ module, objects = parse_file(str(path))
257
+ assert module.docstring == "BOM."
258
+ assert objects[0].name == "g"
File without changes
File without changes
File without changes