context-loader 0.1.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_loader/__init__.py +3 -0
- context_loader/application.py +197 -0
- context_loader/cli.py +75 -0
- context_loader/collect.py +993 -0
- context_loader/git.py +348 -0
- context_loader/render.py +377 -0
- context_loader-0.1.8.dist-info/METADATA +555 -0
- context_loader-0.1.8.dist-info/RECORD +10 -0
- context_loader-0.1.8.dist-info/WHEEL +4 -0
- context_loader-0.1.8.dist-info/entry_points.txt +3 -0
context_loader/render.py
ADDED
|
@@ -0,0 +1,377 @@
|
|
|
1
|
+
"""Render collected Git and project context as bounded deterministic Markdown."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
from .collect import (
|
|
10
|
+
AGENTS_LIMIT_BYTES,
|
|
11
|
+
DECLARED_COMMANDS_LIMIT_BYTES,
|
|
12
|
+
DIRECTORY_TREE_LIMIT_BYTES,
|
|
13
|
+
ENTRY_FILE_LIMIT_BYTES,
|
|
14
|
+
ENTRY_FILES_TOTAL_LIMIT_BYTES,
|
|
15
|
+
README_LIMIT_BYTES,
|
|
16
|
+
TRUNCATION_MARKER,
|
|
17
|
+
CollectedFile,
|
|
18
|
+
DeclaredCommand,
|
|
19
|
+
DirectoryTree,
|
|
20
|
+
ProjectContext,
|
|
21
|
+
TreeEntry,
|
|
22
|
+
render_agents_selection_audit,
|
|
23
|
+
)
|
|
24
|
+
from .git import RecentCommit, RepositoryState
|
|
25
|
+
|
|
26
|
+
SCHEMA = "context-loader/v0.1"
|
|
27
|
+
GLOBAL_OUTPUT_LIMIT_BYTES = 98_304
|
|
28
|
+
MAX_CHANGE_ENTRIES = 100
|
|
29
|
+
MAX_CHANGE_MARKDOWN_BYTES = 4 * 1024
|
|
30
|
+
GLOBAL_OMISSION = "Omitted: global output limit reached."
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True, slots=True)
|
|
34
|
+
class MarkdownRender:
|
|
35
|
+
output: bytes
|
|
36
|
+
included_sections: tuple[str, ...]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _display(value: str) -> str:
|
|
40
|
+
rendered: list[str] = []
|
|
41
|
+
for character in value:
|
|
42
|
+
codepoint = ord(character)
|
|
43
|
+
if character == "`":
|
|
44
|
+
rendered.append(r"\x60")
|
|
45
|
+
elif 0xDC80 <= codepoint <= 0xDCFF:
|
|
46
|
+
rendered.append(f"\\x{codepoint - 0xDC00:02x}")
|
|
47
|
+
elif character.isprintable():
|
|
48
|
+
rendered.append(character)
|
|
49
|
+
elif codepoint <= 0xFF:
|
|
50
|
+
rendered.append(f"\\x{codepoint:02x}")
|
|
51
|
+
elif codepoint <= 0xFFFF:
|
|
52
|
+
rendered.append(f"\\u{codepoint:04x}")
|
|
53
|
+
else:
|
|
54
|
+
rendered.append(f"\\U{codepoint:08x}")
|
|
55
|
+
return "".join(rendered)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _display_limited(value: str, limit: int) -> str | None:
|
|
59
|
+
rendered: list[str] = []
|
|
60
|
+
used = 0
|
|
61
|
+
for character in value:
|
|
62
|
+
piece = _display(character)
|
|
63
|
+
size = len(piece.encode())
|
|
64
|
+
if used + size > limit:
|
|
65
|
+
return None
|
|
66
|
+
rendered.append(piece)
|
|
67
|
+
used += size
|
|
68
|
+
return "".join(rendered)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _truncate_at_line_boundary(content: str, limit: int) -> tuple[str, bool]:
|
|
72
|
+
encoded = content.encode()
|
|
73
|
+
if len(encoded) <= limit:
|
|
74
|
+
return content, False
|
|
75
|
+
if limit <= 0:
|
|
76
|
+
return "", True
|
|
77
|
+
boundary = encoded.rfind(b"\n", 0, limit + 1)
|
|
78
|
+
if boundary < 0:
|
|
79
|
+
return "", True
|
|
80
|
+
return encoded[: boundary + 1].decode(), True
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _content_with_marker(content: str, limit: int, truncated: bool) -> tuple[str, bool]:
|
|
84
|
+
if not truncated and len(content.encode()) <= limit:
|
|
85
|
+
return content, False
|
|
86
|
+
marker_size = len(TRUNCATION_MARKER.encode())
|
|
87
|
+
prefix, _was_cut = _truncate_at_line_boundary(content, max(0, limit - marker_size))
|
|
88
|
+
return f"{prefix}{TRUNCATION_MARKER}", True
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _fenced(language: str, body: str) -> str:
|
|
92
|
+
runs = (len(match.group(0)) for match in re.finditer(r"`+", body))
|
|
93
|
+
fence = "`" * max(3, max(runs, default=0) + 1)
|
|
94
|
+
separator = "" if not body or body.endswith("\n") else "\n"
|
|
95
|
+
return f"{fence}{language}\n{body}{separator}{fence}"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _file_body(source: CollectedFile, limit: int) -> str | None:
|
|
99
|
+
if source.status is not None:
|
|
100
|
+
return None
|
|
101
|
+
if source.selection is not None:
|
|
102
|
+
return source.content
|
|
103
|
+
body, _truncated = _content_with_marker(source.content, limit, source.truncated)
|
|
104
|
+
return body
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _file_payload(source: CollectedFile, limit: int) -> str:
|
|
108
|
+
body = _file_body(source, limit)
|
|
109
|
+
if body is None:
|
|
110
|
+
assert source.status is not None
|
|
111
|
+
return source.status
|
|
112
|
+
return _fenced(source.language, body)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _change_lines(state: RepositoryState) -> tuple[list[str], bool]:
|
|
116
|
+
lines: list[str] = []
|
|
117
|
+
used_bytes = 0
|
|
118
|
+
for change in state.changes:
|
|
119
|
+
if len(lines) == MAX_CHANGE_ENTRIES:
|
|
120
|
+
return lines, True
|
|
121
|
+
line = f"- `{change.status} {_display(change.path)}`"
|
|
122
|
+
line_bytes = len(f"{line}\n".encode())
|
|
123
|
+
if used_bytes + line_bytes > MAX_CHANGE_MARKDOWN_BYTES:
|
|
124
|
+
return lines, True
|
|
125
|
+
lines.append(line)
|
|
126
|
+
used_bytes += line_bytes
|
|
127
|
+
return lines, False
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _render_git_state(state: RepositoryState) -> str:
|
|
131
|
+
lines = [
|
|
132
|
+
"## Git State",
|
|
133
|
+
"",
|
|
134
|
+
f"- Branch: `{_display(state.branch)}`",
|
|
135
|
+
f"- HEAD: `{_display(state.head)}`",
|
|
136
|
+
f"- Upstream: `{_display(state.upstream)}`",
|
|
137
|
+
f"- Ahead / behind: `{_display(state.ahead_behind)}`",
|
|
138
|
+
f"- Worktree: `{state.worktree}`",
|
|
139
|
+
"",
|
|
140
|
+
"### Working Tree Changes",
|
|
141
|
+
"",
|
|
142
|
+
]
|
|
143
|
+
changes, truncated = _change_lines(state)
|
|
144
|
+
if changes:
|
|
145
|
+
lines.extend(changes)
|
|
146
|
+
elif not truncated:
|
|
147
|
+
lines.append("No changes.")
|
|
148
|
+
if truncated:
|
|
149
|
+
lines.append("- Additional changes omitted by the 100-entry / 4-KiB limit.")
|
|
150
|
+
return "\n".join(lines)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _render_source_section(title: str, source: CollectedFile, limit: int) -> str:
|
|
154
|
+
parts = [
|
|
155
|
+
f"## {title}",
|
|
156
|
+
"",
|
|
157
|
+
f"Source: `{source.name}`",
|
|
158
|
+
"",
|
|
159
|
+
_file_payload(source, limit),
|
|
160
|
+
]
|
|
161
|
+
if source.selection is not None:
|
|
162
|
+
parts.extend(("", render_agents_selection_audit(source.selection)))
|
|
163
|
+
return "\n".join(parts)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _command_line(command: DeclaredCommand) -> str:
|
|
167
|
+
if command.parse_error:
|
|
168
|
+
return f"- `{command.source}`: unable to parse command declarations."
|
|
169
|
+
assert command.invocation is not None
|
|
170
|
+
invocation = _display(command.invocation)
|
|
171
|
+
if command.target is None:
|
|
172
|
+
return f"- `{invocation}`"
|
|
173
|
+
return f"- `{invocation}` → `{_display(command.target)}`"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _bounded_lines(lines: list[str], limit: int, *, already_truncated: bool = False) -> str:
|
|
177
|
+
complete = "\n".join(lines)
|
|
178
|
+
if not already_truncated and len(complete.encode()) <= limit:
|
|
179
|
+
return complete
|
|
180
|
+
marker_size = len(TRUNCATION_MARKER.encode())
|
|
181
|
+
budget = max(0, limit - marker_size - 1)
|
|
182
|
+
selected: list[str] = []
|
|
183
|
+
used = 0
|
|
184
|
+
for line in lines:
|
|
185
|
+
size = len(line.encode()) + (1 if selected else 0)
|
|
186
|
+
if used + size > budget:
|
|
187
|
+
break
|
|
188
|
+
selected.append(line)
|
|
189
|
+
used += size
|
|
190
|
+
selected.append(TRUNCATION_MARKER)
|
|
191
|
+
return "\n".join(selected)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _render_commands(commands: tuple[DeclaredCommand, ...]) -> str:
|
|
195
|
+
lines = [_command_line(command) for command in commands]
|
|
196
|
+
if not lines:
|
|
197
|
+
lines.append("No supported command declarations found.")
|
|
198
|
+
return "\n".join(
|
|
199
|
+
(
|
|
200
|
+
"## Declared Commands",
|
|
201
|
+
"",
|
|
202
|
+
_bounded_lines(lines, DECLARED_COMMANDS_LIMIT_BYTES),
|
|
203
|
+
)
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _entry_body(source: CollectedFile, remaining: int) -> tuple[str | None, int]:
|
|
208
|
+
if source.status is not None:
|
|
209
|
+
return None, 0
|
|
210
|
+
encoded_size = len(source.content.encode())
|
|
211
|
+
aggregate_truncated = encoded_size > remaining
|
|
212
|
+
needs_marker = source.truncated or aggregate_truncated
|
|
213
|
+
marker_size = len(TRUNCATION_MARKER.encode()) if needs_marker else 0
|
|
214
|
+
content_limit = min(remaining, ENTRY_FILE_LIMIT_BYTES - marker_size)
|
|
215
|
+
visible, was_cut = _truncate_at_line_boundary(source.content, max(0, content_limit))
|
|
216
|
+
needs_marker = needs_marker or was_cut
|
|
217
|
+
body = f"{visible}{TRUNCATION_MARKER}" if needs_marker else visible
|
|
218
|
+
return body, len(visible.encode())
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _entry_payload(source: CollectedFile, remaining: int) -> tuple[str, int]:
|
|
222
|
+
body, consumed = _entry_body(source, remaining)
|
|
223
|
+
if body is None:
|
|
224
|
+
assert source.status is not None
|
|
225
|
+
return source.status, consumed
|
|
226
|
+
return _fenced(source.language, body), consumed
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _render_entry_files(entry_files: tuple[CollectedFile, ...]) -> str:
|
|
230
|
+
parts = ["## Project Entry Files"]
|
|
231
|
+
used_content = 0
|
|
232
|
+
for source in entry_files:
|
|
233
|
+
remaining = max(0, ENTRY_FILES_TOTAL_LIMIT_BYTES - used_content)
|
|
234
|
+
payload, consumed = _entry_payload(source, remaining)
|
|
235
|
+
used_content += consumed
|
|
236
|
+
parts.extend(("", f"### `{source.name}`", "", payload))
|
|
237
|
+
return "\n".join(parts)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _render_recent_commits(commits: tuple[RecentCommit, ...]) -> str | None:
|
|
241
|
+
lines = ["## Recent Commits", ""]
|
|
242
|
+
if not commits:
|
|
243
|
+
lines.append("No commits.")
|
|
244
|
+
return "\n".join(lines)
|
|
245
|
+
used = len(b"## Recent Commits\n\n")
|
|
246
|
+
for commit in commits:
|
|
247
|
+
prefix = f"- `{commit.object_id[:12]}` · "
|
|
248
|
+
date = _display_limited(commit.author_date, GLOBAL_OUTPUT_LIMIT_BYTES)
|
|
249
|
+
if date is None:
|
|
250
|
+
return None
|
|
251
|
+
prefix = f"{prefix}{date} · "
|
|
252
|
+
remaining = GLOBAL_OUTPUT_LIMIT_BYTES - used - len(prefix.encode()) - 1
|
|
253
|
+
subject = _display_limited(commit.subject, max(0, remaining))
|
|
254
|
+
if subject is None:
|
|
255
|
+
return None
|
|
256
|
+
line = f"{prefix}{subject}"
|
|
257
|
+
used += len(line.encode()) + 1
|
|
258
|
+
if used > GLOBAL_OUTPUT_LIMIT_BYTES:
|
|
259
|
+
return None
|
|
260
|
+
lines.append(line)
|
|
261
|
+
return "\n".join(lines)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _tree_line(entry: TreeEntry) -> str:
|
|
265
|
+
path = _display(entry.path)
|
|
266
|
+
if entry.kind == "directory":
|
|
267
|
+
return f"{path}/"
|
|
268
|
+
if entry.kind == "symlink":
|
|
269
|
+
return f"{path}@"
|
|
270
|
+
if entry.kind == "unreadable_directory":
|
|
271
|
+
return f"{path}/ [Skipped: unreadable.]" if path else "Skipped: unreadable directory."
|
|
272
|
+
return path
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _render_directory_tree(tree: DirectoryTree) -> str:
|
|
276
|
+
lines = [_tree_line(entry) for entry in tree.entries]
|
|
277
|
+
if not lines:
|
|
278
|
+
lines.append("No entries.")
|
|
279
|
+
body = _bounded_lines(
|
|
280
|
+
lines,
|
|
281
|
+
DIRECTORY_TREE_LIMIT_BYTES,
|
|
282
|
+
already_truncated=tree.truncated,
|
|
283
|
+
)
|
|
284
|
+
return "\n".join(("## Directory Tree", "", _fenced("text", body)))
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _omitted_section(title: str) -> str:
|
|
288
|
+
return f"## {title}\n\n{GLOBAL_OMISSION}"
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def render_markdown_with_details(state: RepositoryState, project: ProjectContext) -> MarkdownRender:
|
|
292
|
+
"""Render all sections in fixed order without exceeding the global byte limit."""
|
|
293
|
+
header = "\n".join(
|
|
294
|
+
(
|
|
295
|
+
"# Codex Project Context",
|
|
296
|
+
"",
|
|
297
|
+
f"- Schema: `{SCHEMA}`",
|
|
298
|
+
f"- Repository: `{_display(os.fspath(state.repository))}`",
|
|
299
|
+
)
|
|
300
|
+
)
|
|
301
|
+
sections: list[tuple[str, str | None]] = [
|
|
302
|
+
("Git State", _render_git_state(state)),
|
|
303
|
+
(
|
|
304
|
+
"Development Instructions",
|
|
305
|
+
_render_source_section(
|
|
306
|
+
"Development Instructions",
|
|
307
|
+
project.instructions,
|
|
308
|
+
AGENTS_LIMIT_BYTES,
|
|
309
|
+
),
|
|
310
|
+
),
|
|
311
|
+
(
|
|
312
|
+
"Project Overview",
|
|
313
|
+
_render_source_section("Project Overview", project.overview, README_LIMIT_BYTES),
|
|
314
|
+
),
|
|
315
|
+
("Declared Commands", _render_commands(project.commands)),
|
|
316
|
+
("Project Entry Files", _render_entry_files(project.entry_files)),
|
|
317
|
+
("Recent Commits", _render_recent_commits(state.commits)),
|
|
318
|
+
("Directory Tree", _render_directory_tree(project.directory_tree)),
|
|
319
|
+
]
|
|
320
|
+
|
|
321
|
+
rendered_parts = [header.encode()]
|
|
322
|
+
included_sections: list[str] = []
|
|
323
|
+
omission_started = False
|
|
324
|
+
for index, (title, full_section) in enumerate(sections):
|
|
325
|
+
omission = _omitted_section(title).encode()
|
|
326
|
+
future_omissions = [
|
|
327
|
+
_omitted_section(future_title).encode()
|
|
328
|
+
for future_title, _future_section in sections[index + 1 :]
|
|
329
|
+
]
|
|
330
|
+
if omission_started or full_section is None:
|
|
331
|
+
selected = omission
|
|
332
|
+
omission_started = True
|
|
333
|
+
else:
|
|
334
|
+
candidate = full_section.encode()
|
|
335
|
+
tentative = b"\n\n".join((*rendered_parts, candidate, *future_omissions)) + b"\n"
|
|
336
|
+
if len(tentative) <= GLOBAL_OUTPUT_LIMIT_BYTES:
|
|
337
|
+
selected = candidate
|
|
338
|
+
included_sections.append(title)
|
|
339
|
+
else:
|
|
340
|
+
selected = omission
|
|
341
|
+
omission_started = True
|
|
342
|
+
rendered_parts.append(selected)
|
|
343
|
+
|
|
344
|
+
output = b"\n\n".join(rendered_parts) + b"\n"
|
|
345
|
+
if len(output) > GLOBAL_OUTPUT_LIMIT_BYTES:
|
|
346
|
+
raise RuntimeError("global context output limit could not be satisfied")
|
|
347
|
+
return MarkdownRender(output=output, included_sections=tuple(included_sections))
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def rendered_source_contents(
|
|
351
|
+
project: ProjectContext, included_sections: tuple[str, ...]
|
|
352
|
+
) -> tuple[tuple[CollectedFile, str], ...]:
|
|
353
|
+
"""Return source-file bodies that actually appear in the final Markdown context."""
|
|
354
|
+
included = frozenset(included_sections)
|
|
355
|
+
rendered: list[tuple[CollectedFile, str]] = []
|
|
356
|
+
if "Development Instructions" in included:
|
|
357
|
+
body = _file_body(project.instructions, AGENTS_LIMIT_BYTES)
|
|
358
|
+
if body is not None:
|
|
359
|
+
rendered.append((project.instructions, body))
|
|
360
|
+
if "Project Overview" in included:
|
|
361
|
+
body = _file_body(project.overview, README_LIMIT_BYTES)
|
|
362
|
+
if body is not None:
|
|
363
|
+
rendered.append((project.overview, body))
|
|
364
|
+
if "Project Entry Files" in included:
|
|
365
|
+
used_content = 0
|
|
366
|
+
for source in project.entry_files:
|
|
367
|
+
remaining = max(0, ENTRY_FILES_TOTAL_LIMIT_BYTES - used_content)
|
|
368
|
+
body, consumed = _entry_body(source, remaining)
|
|
369
|
+
used_content += consumed
|
|
370
|
+
if body is not None:
|
|
371
|
+
rendered.append((source, body))
|
|
372
|
+
return tuple(rendered)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def render_markdown(state: RepositoryState, project: ProjectContext) -> bytes:
|
|
376
|
+
"""Render all sections in fixed order without exceeding the global byte limit."""
|
|
377
|
+
return render_markdown_with_details(state, project).output
|