markdownizer 0.4.0__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {markdownizer-0.4.0 → markdownizer-0.4.2}/PKG-INFO +1 -1
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/__init__.py +1 -1
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/cli.py +3 -1
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/imports.py +10 -2
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/ir.py +2 -2
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/rank.py +4 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/slice.py +87 -80
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/tokens.py +1 -1
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/parser.py +15 -2
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/PKG-INFO +1 -1
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_cli.py +24 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_extractor.py +12 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_imports.py +39 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_ir.py +31 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_optimizer.py +73 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_parser.py +24 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/LICENSE +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/README.md +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/__main__.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/__init__.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/base.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/compact.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/json.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/backends/markdown.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/classifier.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/extractor.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/__init__.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/optimizer/profiles.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/py.typed +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/renderer.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer/scanner.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/SOURCES.txt +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/dependency_links.txt +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/entry_points.txt +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/requires.txt +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/markdownizer.egg-info/top_level.txt +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/pyproject.toml +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/setup.cfg +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_backends.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_classifier.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_packaging.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_ranking.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_renderer.py +0 -0
- {markdownizer-0.4.0 → markdownizer-0.4.2}/tests/test_scanner.py +0 -0
|
@@ -8,7 +8,7 @@ from markdownizer.extractor import extract_project
|
|
|
8
8
|
from markdownizer.ir import IR_VERSION, ProjectIR, build_project_ir
|
|
9
9
|
from markdownizer.optimizer import OptimizedContext, optimize_context
|
|
10
10
|
|
|
11
|
-
__version__ = "0.4.
|
|
11
|
+
__version__ = "0.4.2"
|
|
12
12
|
__all__ = [
|
|
13
13
|
"extract_project",
|
|
14
14
|
"build_project_ir",
|
|
@@ -83,7 +83,9 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
83
83
|
description=(
|
|
84
84
|
"Extract existing documentation from a Python project into "
|
|
85
85
|
"Markdown files. Use 'markdownizer build <project>' for format "
|
|
86
|
-
"selection; the legacy form below is kept as a compatibility
|
|
86
|
+
"selection; the legacy form below is kept as a compatibility "
|
|
87
|
+
"alias. If your project directory is named 'build', 'context', "
|
|
88
|
+
"or 'stats', prefix it with './'."
|
|
87
89
|
),
|
|
88
90
|
)
|
|
89
91
|
parser.set_defaults(command="build")
|
|
@@ -67,11 +67,17 @@ def _resolve_relative(
|
|
|
67
67
|
|
|
68
68
|
|
|
69
69
|
def _collect_statement_imports(
|
|
70
|
-
|
|
70
|
+
module: ast.Module, module_map: dict[str, str], from_rel_path: str
|
|
71
71
|
) -> list[ImportEdge]:
|
|
72
|
+
"""Collect module-level import statements.
|
|
73
|
+
|
|
74
|
+
Only top-level statements count: imports inside function bodies, class
|
|
75
|
+
bodies, or ``if __name__ == "__main__":`` blocks are not structural
|
|
76
|
+
module dependencies. ``from __future__`` imports are filtered out.
|
|
77
|
+
"""
|
|
72
78
|
edges: list[ImportEdge] = []
|
|
73
79
|
|
|
74
|
-
for child in
|
|
80
|
+
for child in module.body:
|
|
75
81
|
if isinstance(child, ast.Import):
|
|
76
82
|
for alias in child.names:
|
|
77
83
|
target = _resolve_absolute(alias.name, module_map)
|
|
@@ -87,6 +93,8 @@ def _collect_statement_imports(
|
|
|
87
93
|
)
|
|
88
94
|
)
|
|
89
95
|
elif isinstance(child, ast.ImportFrom):
|
|
96
|
+
if child.module == "__future__":
|
|
97
|
+
continue
|
|
90
98
|
if child.module is None:
|
|
91
99
|
# ``from . import x`` — the imported names are resolved
|
|
92
100
|
# relative to the current package.
|
|
@@ -27,7 +27,7 @@ from typing import Any
|
|
|
27
27
|
|
|
28
28
|
from markdownizer.classifier import classify
|
|
29
29
|
from markdownizer.imports import ImportEdge, collect_import_edges
|
|
30
|
-
from markdownizer.parser import DocObject, parse_file
|
|
30
|
+
from markdownizer.parser import DocObject, parse_file, read_python_source
|
|
31
31
|
from markdownizer.scanner import scan_python_files
|
|
32
32
|
|
|
33
33
|
IR_VERSION = 1
|
|
@@ -336,7 +336,7 @@ def build_project_ir(project_root: Path, exclude: list[str] | None = None) -> Pr
|
|
|
336
336
|
for py_file in files:
|
|
337
337
|
rel = py_file.relative_to(project_root).as_posix()
|
|
338
338
|
dotted = _module_dotted_name(rel)
|
|
339
|
-
source = py_file
|
|
339
|
+
source = read_python_source(py_file)
|
|
340
340
|
module_obj, objects = parse_file(str(py_file), source=source)
|
|
341
341
|
|
|
342
342
|
module_symbols = [_symbol_from_docobject(o, rel) for o in objects]
|
|
@@ -11,6 +11,10 @@ Methods:
|
|
|
11
11
|
(fixed iteration count for full determinism)
|
|
12
12
|
* ``fanout`` — in-degree of each module in the import graph
|
|
13
13
|
* ``simple`` — uniform file scores (no graph)
|
|
14
|
+
|
|
15
|
+
Rank values are deterministic for a given platform (IEEE-754 addition
|
|
16
|
+
order); the IR hash excludes ranks, so ``ir.hash`` is stable across
|
|
17
|
+
platforms even if floating-point tie-breaking ever differs.
|
|
14
18
|
"""
|
|
15
19
|
|
|
16
20
|
from __future__ import annotations
|
|
@@ -4,12 +4,14 @@ Turns a ranked ProjectIR into an AI-ready context artifact within a token
|
|
|
4
4
|
budget, using deterministic layered emission:
|
|
5
5
|
|
|
6
6
|
* layer 0 — project index (packages, modules, docstrings)
|
|
7
|
-
* layer 1 —
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
High-ranked symbols are placed at the
|
|
12
|
-
(primacy/recency bias); lower-ranked content
|
|
7
|
+
* layer 1 — per-symbol blocks: full source for the highest-ranked symbols,
|
|
8
|
+
signature-only for the rest (or docstrings only for ``include_source``
|
|
9
|
+
profiles without source)
|
|
10
|
+
|
|
11
|
+
Each symbol is emitted exactly once. High-ranked symbols are placed at the
|
|
12
|
+
start and end of the artifact (primacy/recency bias); lower-ranked content
|
|
13
|
+
sits in the middle. The reported token estimate counts the complete
|
|
14
|
+
artifact, header included.
|
|
13
15
|
"""
|
|
14
16
|
|
|
15
17
|
from __future__ import annotations
|
|
@@ -102,6 +104,55 @@ def _block_tokens(text: str, prefer_tiktoken: bool) -> float:
|
|
|
102
104
|
return count_tokens(text, prefer_tiktoken=prefer_tiktoken)
|
|
103
105
|
|
|
104
106
|
|
|
107
|
+
def _symbol_block(
|
|
108
|
+
symbol: Symbol,
|
|
109
|
+
profile: Profile,
|
|
110
|
+
prefer_tiktoken: bool,
|
|
111
|
+
budget: float,
|
|
112
|
+
) -> tuple[str, float] | None:
|
|
113
|
+
"""Render one symbol exactly once; ``None`` when nothing fits the budget."""
|
|
114
|
+
comments = profile.include_comments
|
|
115
|
+
if profile.include_source is True:
|
|
116
|
+
src_text = render_object(
|
|
117
|
+
symbol,
|
|
118
|
+
None,
|
|
119
|
+
include_source=True,
|
|
120
|
+
include_comments=comments,
|
|
121
|
+
)
|
|
122
|
+
src_cost = _block_tokens(src_text, prefer_tiktoken)
|
|
123
|
+
if src_cost <= budget:
|
|
124
|
+
return src_text, src_cost
|
|
125
|
+
sig_text = render_object(
|
|
126
|
+
symbol,
|
|
127
|
+
None,
|
|
128
|
+
include_source="signature",
|
|
129
|
+
include_comments=comments,
|
|
130
|
+
)
|
|
131
|
+
sig_cost = _block_tokens(sig_text, prefer_tiktoken)
|
|
132
|
+
return (sig_text, sig_cost) if sig_cost <= budget else None
|
|
133
|
+
|
|
134
|
+
if profile.include_source == "signature":
|
|
135
|
+
sig_text = render_object(
|
|
136
|
+
symbol,
|
|
137
|
+
None,
|
|
138
|
+
include_source="signature",
|
|
139
|
+
include_comments=comments,
|
|
140
|
+
)
|
|
141
|
+
sig_cost = _block_tokens(sig_text, prefer_tiktoken)
|
|
142
|
+
return (sig_text, sig_cost) if sig_cost <= budget else None
|
|
143
|
+
|
|
144
|
+
if symbol.docstring:
|
|
145
|
+
doc_text = render_object(
|
|
146
|
+
symbol,
|
|
147
|
+
None,
|
|
148
|
+
include_source=False,
|
|
149
|
+
include_comments=comments,
|
|
150
|
+
)
|
|
151
|
+
doc_cost = _block_tokens(doc_text, prefer_tiktoken)
|
|
152
|
+
return (doc_text, doc_cost) if doc_cost <= budget else None
|
|
153
|
+
return None
|
|
154
|
+
|
|
155
|
+
|
|
105
156
|
def optimize_context(
|
|
106
157
|
ir: ProjectIR,
|
|
107
158
|
max_tokens: float,
|
|
@@ -114,19 +165,20 @@ def optimize_context(
|
|
|
114
165
|
|
|
115
166
|
Args:
|
|
116
167
|
ir: The project IR (ranked in place by this function).
|
|
117
|
-
max_tokens: Target token budget (soft limit, ±10% slop).
|
|
168
|
+
max_tokens: Target token budget (soft limit, ±10% slop). Must be > 0.
|
|
118
169
|
profile: Profile preset name (see :mod:`markdownizer.optimizer.profiles`).
|
|
119
170
|
query: Optional space-separated keyword prefilter on symbol names,
|
|
120
171
|
frameworks, and docstrings (deterministic substring match).
|
|
121
172
|
rank_method: ``pagerank`` (default), ``fanout``, or ``simple``.
|
|
122
173
|
prefer_tiktoken: Use ``tiktoken`` when installed for token estimates.
|
|
174
|
+
|
|
175
|
+
Raises:
|
|
176
|
+
ValueError: If ``max_tokens`` is not positive.
|
|
123
177
|
"""
|
|
178
|
+
if float(max_tokens) <= 0:
|
|
179
|
+
raise ValueError(f"max_tokens must be positive, got {max_tokens!r}")
|
|
124
180
|
rank_ir(ir, method=rank_method)
|
|
125
181
|
profile_obj = get_profile(profile)
|
|
126
|
-
budget = float(max_tokens) * _BUDGET_SLOP
|
|
127
|
-
parts: list[str] = []
|
|
128
|
-
total: float = 0.0
|
|
129
|
-
included: int = 0
|
|
130
182
|
total_symbols = len(ir.symbols)
|
|
131
183
|
|
|
132
184
|
header = (
|
|
@@ -134,90 +186,45 @@ def optimize_context(
|
|
|
134
186
|
f"(profile={profile_obj.name}, rank={rank_method}, "
|
|
135
187
|
f"max_tokens={int(max_tokens)})"
|
|
136
188
|
)
|
|
189
|
+
header_cost = _block_tokens(header, prefer_tiktoken)
|
|
190
|
+
budget = float(max_tokens) * _BUDGET_SLOP - header_cost
|
|
191
|
+
|
|
192
|
+
parts: list[str] = []
|
|
193
|
+
total: float = 0.0
|
|
137
194
|
|
|
138
195
|
index = _render_index(ir)
|
|
139
|
-
|
|
196
|
+
index_cost = _block_tokens(index, prefer_tiktoken)
|
|
197
|
+
if index_cost <= budget:
|
|
140
198
|
parts.append(index)
|
|
141
|
-
total +=
|
|
199
|
+
total += index_cost
|
|
142
200
|
|
|
143
201
|
module_index = _render_module_index(ir)
|
|
144
|
-
|
|
202
|
+
module_cost = _block_tokens(module_index, prefer_tiktoken)
|
|
203
|
+
if module_cost <= budget:
|
|
145
204
|
parts.append(module_index)
|
|
146
|
-
total +=
|
|
205
|
+
total += module_cost
|
|
147
206
|
|
|
148
207
|
symbols = _select_symbols(ir, profile_obj, query)
|
|
149
208
|
ordered = _edge_place(symbols)
|
|
150
209
|
|
|
151
|
-
|
|
152
|
-
sources: list[tuple[Symbol, str, float]] = []
|
|
153
|
-
remainder: list[tuple[Symbol, str, float]] = []
|
|
154
|
-
|
|
210
|
+
included = 0
|
|
155
211
|
for symbol in ordered:
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
if profile_obj.include_source is True:
|
|
168
|
-
src_text = render_object(
|
|
169
|
-
symbol,
|
|
170
|
-
None,
|
|
171
|
-
include_source=True,
|
|
172
|
-
include_comments=profile_obj.include_comments,
|
|
173
|
-
)
|
|
174
|
-
src_cost = _block_tokens(src_text, prefer_tiktoken)
|
|
175
|
-
sources.append((symbol, src_text, src_cost))
|
|
176
|
-
|
|
177
|
-
if profile_obj.include_source is False and symbol.docstring:
|
|
178
|
-
doc_text = render_object(
|
|
179
|
-
symbol,
|
|
180
|
-
None,
|
|
181
|
-
include_source=False,
|
|
182
|
-
include_comments=profile_obj.include_comments,
|
|
183
|
-
)
|
|
184
|
-
doc_cost = _block_tokens(doc_text, prefer_tiktoken)
|
|
185
|
-
remainder.append((symbol, doc_text, doc_cost))
|
|
186
|
-
|
|
187
|
-
if profile_obj.include_source is not False:
|
|
188
|
-
for _, sig_text, sig_cost in signatures:
|
|
189
|
-
if total + sig_cost > budget:
|
|
190
|
-
break
|
|
191
|
-
parts.append("---")
|
|
192
|
-
parts.append("")
|
|
193
|
-
parts.append(sig_text)
|
|
194
|
-
total += sig_cost
|
|
195
|
-
included += 1
|
|
196
|
-
|
|
197
|
-
if profile_obj.include_source is True:
|
|
198
|
-
for _, src_text, src_cost in sources:
|
|
199
|
-
if total + src_cost > budget:
|
|
200
|
-
break
|
|
201
|
-
parts.append("---")
|
|
202
|
-
parts.append("")
|
|
203
|
-
parts.append(src_text)
|
|
204
|
-
total += src_cost
|
|
205
|
-
included += 1
|
|
206
|
-
|
|
207
|
-
if profile_obj.include_source is False:
|
|
208
|
-
for _, doc_text, doc_cost in remainder:
|
|
209
|
-
if total + doc_cost > budget:
|
|
210
|
-
break
|
|
211
|
-
parts.append("---")
|
|
212
|
-
parts.append("")
|
|
213
|
-
parts.append(doc_text)
|
|
214
|
-
total += doc_cost
|
|
215
|
-
included += 1
|
|
212
|
+
block = _symbol_block(symbol, profile_obj, prefer_tiktoken, budget)
|
|
213
|
+
if block is None:
|
|
214
|
+
continue
|
|
215
|
+
text, cost = block
|
|
216
|
+
if total + cost > budget:
|
|
217
|
+
break
|
|
218
|
+
parts.append("---")
|
|
219
|
+
parts.append("")
|
|
220
|
+
parts.append(text)
|
|
221
|
+
total += cost
|
|
222
|
+
included += 1
|
|
216
223
|
|
|
217
224
|
text = "\n".join([header, "", *parts]).rstrip() + "\n"
|
|
218
225
|
return OptimizedContext(
|
|
219
226
|
text=text,
|
|
220
|
-
estimated_tokens=
|
|
227
|
+
estimated_tokens=_block_tokens(text, prefer_tiktoken),
|
|
221
228
|
max_tokens=float(max_tokens),
|
|
222
229
|
profile=profile_obj.name,
|
|
223
230
|
rank_method=rank_method,
|
|
@@ -18,7 +18,7 @@ try:
|
|
|
18
18
|
|
|
19
19
|
_ENCODER = _tiktoken.get_encoding("cl100k_base")
|
|
20
20
|
_HAS_TIKTOKEN = True
|
|
21
|
-
except
|
|
21
|
+
except (ImportError, OSError): # pragma: no cover - depends on optional dependency
|
|
22
22
|
_ENCODER = None
|
|
23
23
|
_HAS_TIKTOKEN = False
|
|
24
24
|
|
|
@@ -6,6 +6,7 @@ import ast
|
|
|
6
6
|
import io
|
|
7
7
|
import tokenize
|
|
8
8
|
from dataclasses import dataclass, field
|
|
9
|
+
from pathlib import Path
|
|
9
10
|
from typing import Protocol
|
|
10
11
|
|
|
11
12
|
|
|
@@ -72,6 +73,19 @@ def _rightmost_name(name: str) -> str:
|
|
|
72
73
|
return name.rsplit(".", 1)[-1] if name else ""
|
|
73
74
|
|
|
74
75
|
|
|
76
|
+
def read_python_source(path: Path) -> str:
|
|
77
|
+
"""Read a Python source file, tolerating non-UTF-8 encodings.
|
|
78
|
+
|
|
79
|
+
Uses ``utf-8-sig`` (which also strips a UTF-8 BOM) and falls back to
|
|
80
|
+
latin-1, which can decode every byte sequence without raising. A single
|
|
81
|
+
legacy-encoded file must never abort an entire project extraction.
|
|
82
|
+
"""
|
|
83
|
+
try:
|
|
84
|
+
return path.read_text(encoding="utf-8-sig")
|
|
85
|
+
except UnicodeDecodeError:
|
|
86
|
+
return path.read_text(encoding="latin-1")
|
|
87
|
+
|
|
88
|
+
|
|
75
89
|
def _unparse_or_empty(node: ast.AST | None) -> str:
|
|
76
90
|
"""Render an AST node back to source text, or ``""`` if unavailable."""
|
|
77
91
|
if node is None:
|
|
@@ -337,8 +351,7 @@ def parse_file(file_path: str, source: str | None = None) -> tuple[DocObject, li
|
|
|
337
351
|
``source`` may be passed in to avoid re-reading the file from disk.
|
|
338
352
|
"""
|
|
339
353
|
if source is None:
|
|
340
|
-
|
|
341
|
-
source = fh.read()
|
|
354
|
+
source = read_python_source(Path(file_path))
|
|
342
355
|
|
|
343
356
|
source_lines = source.splitlines()
|
|
344
357
|
all_comments = _collect_comments(source)
|
|
@@ -222,6 +222,30 @@ def test_stats_invalid_rank_exits(tmp_path):
|
|
|
222
222
|
main(["stats", str(tmp_path), "--rank", "magic"])
|
|
223
223
|
|
|
224
224
|
|
|
225
|
+
def test_stats_empty_project(tmp_path, capsys):
|
|
226
|
+
assert main(["stats", str(tmp_path)]) == 0
|
|
227
|
+
captured = capsys.readouterr()
|
|
228
|
+
assert "Symbols: 0" in captured.out
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def test_context_query_flag(tmp_path, capsys):
|
|
232
|
+
(tmp_path / "auth.py").write_text('"""Auth."""\ndef login():\n """Login."""\n pass\n')
|
|
233
|
+
(tmp_path / "other.py").write_text(
|
|
234
|
+
'"""Other."""\ndef unrelated():\n """Nope."""\n pass\n'
|
|
235
|
+
)
|
|
236
|
+
out = tmp_path / "out"
|
|
237
|
+
assert main(["context", str(tmp_path), "-o", str(out), "--query", "login"]) == 0
|
|
238
|
+
text = (out / "context.md").read_text(encoding="utf-8")
|
|
239
|
+
assert "login" in text
|
|
240
|
+
assert "unrelated" not in text
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def test_context_negative_max_tokens_returns_1(tmp_path, capsys):
|
|
244
|
+
(tmp_path / "mod.py").write_text('"""docs"""\n')
|
|
245
|
+
assert main(["context", str(tmp_path), "--max-tokens", "-10"]) == 1
|
|
246
|
+
assert "max_tokens must be positive" in capsys.readouterr().err
|
|
247
|
+
|
|
248
|
+
|
|
225
249
|
def _assert_exits_with(argv: list[str], expected: str, capsys: pytest.CaptureFixture[str]) -> None:
|
|
226
250
|
with pytest.raises(SystemExit):
|
|
227
251
|
main(argv)
|
|
@@ -119,3 +119,15 @@ def test_extract_error_on_unwritable_output(sample_project, tmp_path):
|
|
|
119
119
|
pytest.skip("directory is writable; cannot simulate")
|
|
120
120
|
with pytest.raises(OSError):
|
|
121
121
|
extract_project(sample_project, out / "sub" / "nested")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_extract_survives_non_utf8_file(tmp_path):
|
|
125
|
+
"""End-to-end: a legacy-encoded file does not abort extraction."""
|
|
126
|
+
(tmp_path / "good.py").write_text('"""Fine."""\ndef ok():\n pass\n')
|
|
127
|
+
(tmp_path / "legacy.py").write_bytes(b'"""caf\xe9"""\ndef old():\n pass\n')
|
|
128
|
+
out = tmp_path / "out"
|
|
129
|
+
written = extract_project(tmp_path, out)
|
|
130
|
+
assert sorted(p.name for p in written) == ["_root.md"]
|
|
131
|
+
md = (out / "_root.md").read_text(encoding="utf-8")
|
|
132
|
+
assert "# Function: old" in md
|
|
133
|
+
assert "café" in md
|
|
@@ -144,3 +144,42 @@ def test_dotted_import_resolution_to_init():
|
|
|
144
144
|
|
|
145
145
|
def test_empty_source():
|
|
146
146
|
assert _edges("") == []
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_nested_function_import_not_collected():
|
|
150
|
+
source = "def f():\n import os\n return os.path\n"
|
|
151
|
+
assert _edges(source) == []
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_main_guard_import_not_collected():
|
|
155
|
+
source = "import pkg.models\n\nif __name__ == '__main__':\n import sys\n"
|
|
156
|
+
edges = _edges(source)
|
|
157
|
+
assert [e.imported_module for e in edges] == ["pkg.models"]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_class_body_import_not_collected():
|
|
161
|
+
source = "class C:\n def m(self):\n import os\n"
|
|
162
|
+
assert _edges(source) == []
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_future_import_filtered():
|
|
166
|
+
source = "from __future__ import annotations\nimport pkg.models\n"
|
|
167
|
+
edges = _edges(source)
|
|
168
|
+
assert [e.imported_module for e in edges] == ["pkg.models"]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def test_try_wrapped_import_not_collected():
|
|
172
|
+
"""try/except-wrapped imports are conservatively treated as external."""
|
|
173
|
+
source = "try:\n import pkg.models\nexcept ImportError:\n pass\n"
|
|
174
|
+
assert _edges(source) == []
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def test_latin1_source_decodes():
|
|
178
|
+
from markdownizer.imports import collect_import_edges
|
|
179
|
+
|
|
180
|
+
edges = collect_import_edges(
|
|
181
|
+
'"""caf\xe9"""\nimport os\n'.encode("latin-1").decode("latin-1"),
|
|
182
|
+
"mod.py",
|
|
183
|
+
{"mod.py": "mod.py"},
|
|
184
|
+
)
|
|
185
|
+
assert [e.imported_module for e in edges] == ["os"]
|
|
@@ -245,3 +245,34 @@ def test_unsorted_input_still_sorted(tmp_path):
|
|
|
245
245
|
ir = build_project_ir(tmp_path)
|
|
246
246
|
assert [m.path for m in ir.modules] == ["a/y.py", "b/z.py"]
|
|
247
247
|
assert [s.qualified_name for s in ir.symbols] == ["y", "z"]
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def test_build_survives_non_utf8_file(tmp_path):
|
|
251
|
+
"""A single latin-1 encoded file must not abort the whole build."""
|
|
252
|
+
from tests.conftest import write_py
|
|
253
|
+
|
|
254
|
+
write_py(tmp_path / "good.py", '"""Fine."""\ndef ok():\n pass\n')
|
|
255
|
+
(tmp_path / "legacy.py").write_bytes(b'"""caf\xe9 latin-1"""\ndef old():\n pass\n')
|
|
256
|
+
ir = build_project_ir(tmp_path)
|
|
257
|
+
assert [m.path for m in ir.modules] == ["good.py", "legacy.py"]
|
|
258
|
+
legacy = next(m for m in ir.modules if m.path == "legacy.py")
|
|
259
|
+
assert legacy.docstring == "café latin-1"
|
|
260
|
+
assert ir.stats.symbol_count == 2
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def test_build_survives_utf8_bom_file(tmp_path):
|
|
264
|
+
from tests.conftest import write_py
|
|
265
|
+
|
|
266
|
+
write_py(tmp_path / "a.py", "def f():\n pass\n")
|
|
267
|
+
(tmp_path / "bom.py").write_bytes(b'\xef\xbb\xbf"""BOM doc."""\ndef g():\n pass\n')
|
|
268
|
+
ir = build_project_ir(tmp_path)
|
|
269
|
+
assert [m.path for m in ir.modules] == ["a.py", "bom.py"]
|
|
270
|
+
bom = next(m for m in ir.modules if m.path == "bom.py")
|
|
271
|
+
assert bom.docstring == "BOM doc."
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def test_non_utf8_hash_stable(tmp_path):
|
|
275
|
+
(tmp_path / "legacy.py").write_bytes(b'"""caf\xe9"""\ndef old():\n pass\n')
|
|
276
|
+
ir1 = build_project_ir(tmp_path)
|
|
277
|
+
ir2 = build_project_ir(tmp_path)
|
|
278
|
+
assert ir1.hash == ir2.hash
|
|
@@ -166,3 +166,76 @@ def test_tiny_budget_only_header(tmp_path):
|
|
|
166
166
|
ctx = optimize_context(ir, max_tokens=1)
|
|
167
167
|
assert ctx.included_symbols == 0
|
|
168
168
|
assert ctx.text.startswith("# Context:")
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def test_negative_budget_raises(tmp_path):
|
|
172
|
+
ir = _build(tmp_path)
|
|
173
|
+
with pytest.raises(ValueError):
|
|
174
|
+
optimize_context(ir, max_tokens=0)
|
|
175
|
+
with pytest.raises(ValueError):
|
|
176
|
+
optimize_context(ir, max_tokens=-50)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def test_debugging_emits_each_symbol_once(tmp_path):
|
|
180
|
+
"""M1: full source contains the signature — never duplicate it."""
|
|
181
|
+
ir = _build(tmp_path)
|
|
182
|
+
ctx = optimize_context(ir, max_tokens=10000, profile="debugging")
|
|
183
|
+
assert ctx.included_symbols == ctx.total_symbols
|
|
184
|
+
assert "## Signature" not in ctx.text
|
|
185
|
+
assert ctx.text.count("## Source Code") == ctx.included_symbols
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def test_debugging_falls_back_to_signature_when_source_does_not_fit(tmp_path):
|
|
189
|
+
"""A symbol whose full source exceeds the budget still gets a signature."""
|
|
190
|
+
from tests.conftest import write_py
|
|
191
|
+
|
|
192
|
+
body = "\n".join(f" x{i} = {i} * 7" for i in range(200))
|
|
193
|
+
write_py(
|
|
194
|
+
tmp_path / "big.py",
|
|
195
|
+
f'"""Big."""\ndef huge():\n """Huge."""\n{body}\n return 1\n',
|
|
196
|
+
)
|
|
197
|
+
ir = __import__("markdownizer").build_project_ir(tmp_path)
|
|
198
|
+
ctx = optimize_context(ir, max_tokens=150, profile="debugging")
|
|
199
|
+
assert "## Signature" in ctx.text
|
|
200
|
+
assert "## Source Code" not in ctx.text
|
|
201
|
+
assert ctx.included_symbols == 1
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def test_included_symbols_counts_unique_symbols(tmp_path):
|
|
205
|
+
"""M2: each symbol counts once regardless of emission depth."""
|
|
206
|
+
ir = _build(tmp_path)
|
|
207
|
+
ctx = optimize_context(ir, max_tokens=10000, profile="debugging")
|
|
208
|
+
assert ctx.included_symbols == ctx.total_symbols
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def test_estimated_tokens_matches_artifact(tmp_path):
|
|
212
|
+
"""M3: the estimate counts the complete artifact, header included."""
|
|
213
|
+
from markdownizer.optimizer.tokens import count_tokens
|
|
214
|
+
|
|
215
|
+
ir = _build(tmp_path)
|
|
216
|
+
for profile in ("architecture", "api", "debugging", "onboarding"):
|
|
217
|
+
ctx = optimize_context(ir, max_tokens=10000, profile=profile)
|
|
218
|
+
assert ctx.estimated_tokens == count_tokens(ctx.text), profile
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def test_edge_placement_orders_top_symbols_first(tmp_path):
|
|
222
|
+
"""The highest-ranked symbol leads the artifact; the second leads the tail."""
|
|
223
|
+
ir = _build(tmp_path)
|
|
224
|
+
ctx = optimize_context(ir, max_tokens=10000, profile="api")
|
|
225
|
+
ranked = sorted(ir.symbols, key=lambda s: -s.rank)
|
|
226
|
+
top, second = ranked[0], ranked[1]
|
|
227
|
+
body = ctx.text.split("# Context:")[1]
|
|
228
|
+
top_index = body.find(f"# {top.framework}: {top.name}")
|
|
229
|
+
second_index = body.find(f"# {second.framework}: {second.name}")
|
|
230
|
+
assert top_index < second_index
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def test_context_query_flag(tmp_path):
|
|
234
|
+
from tests.conftest import write_py
|
|
235
|
+
|
|
236
|
+
write_py(tmp_path / "auth.py", '"""Auth."""\ndef login():\n """Login."""\n pass\n')
|
|
237
|
+
write_py(tmp_path / "other.py", '"""Other."""\ndef unrelated():\n """Nope."""\n pass\n')
|
|
238
|
+
ir = __import__("markdownizer").build_project_ir(tmp_path)
|
|
239
|
+
ctx = optimize_context(ir, max_tokens=5000, profile="api", query="login")
|
|
240
|
+
assert "login" in ctx.text
|
|
241
|
+
assert "unrelated" not in ctx.text
|
|
@@ -232,3 +232,27 @@ def test_file_path_populated(write_file):
|
|
|
232
232
|
path = write_file("mod.py", "def f():\n pass\n")
|
|
233
233
|
_, objects = parse_file(str(path))
|
|
234
234
|
assert Path(objects[0].file_path) == path
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def test_latin1_file_parses(tmp_path):
|
|
238
|
+
from markdownizer.parser import read_python_source
|
|
239
|
+
|
|
240
|
+
path = tmp_path / "legacy.py"
|
|
241
|
+
path.write_bytes(b'"""caf\xe9"""\ndef old():\n pass\n')
|
|
242
|
+
source = read_python_source(path)
|
|
243
|
+
assert "café" in source
|
|
244
|
+
module, objects = parse_file(str(path))
|
|
245
|
+
assert module.docstring == "café"
|
|
246
|
+
assert objects[0].name == "old"
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def test_utf8_bom_file_parses(tmp_path):
|
|
250
|
+
from markdownizer.parser import read_python_source
|
|
251
|
+
|
|
252
|
+
path = tmp_path / "bom.py"
|
|
253
|
+
path.write_bytes(b'\xef\xbb\xbf"""BOM."""\ndef g():\n pass\n')
|
|
254
|
+
source = read_python_source(path)
|
|
255
|
+
assert not source.startswith("\ufeff")
|
|
256
|
+
module, objects = parse_file(str(path))
|
|
257
|
+
assert module.docstring == "BOM."
|
|
258
|
+
assert objects[0].name == "g"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|