context-loader 0.1.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,377 @@
1
+ """Render collected Git and project context as bounded deterministic Markdown."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ from dataclasses import dataclass
8
+
9
+ from .collect import (
10
+ AGENTS_LIMIT_BYTES,
11
+ DECLARED_COMMANDS_LIMIT_BYTES,
12
+ DIRECTORY_TREE_LIMIT_BYTES,
13
+ ENTRY_FILE_LIMIT_BYTES,
14
+ ENTRY_FILES_TOTAL_LIMIT_BYTES,
15
+ README_LIMIT_BYTES,
16
+ TRUNCATION_MARKER,
17
+ CollectedFile,
18
+ DeclaredCommand,
19
+ DirectoryTree,
20
+ ProjectContext,
21
+ TreeEntry,
22
+ render_agents_selection_audit,
23
+ )
24
+ from .git import RecentCommit, RepositoryState
25
+
26
+ SCHEMA = "context-loader/v0.1"
27
+ GLOBAL_OUTPUT_LIMIT_BYTES = 98_304
28
+ MAX_CHANGE_ENTRIES = 100
29
+ MAX_CHANGE_MARKDOWN_BYTES = 4 * 1024
30
+ GLOBAL_OMISSION = "Omitted: global output limit reached."
31
+
32
+
33
+ @dataclass(frozen=True, slots=True)
34
+ class MarkdownRender:
35
+ output: bytes
36
+ included_sections: tuple[str, ...]
37
+
38
+
39
+ def _display(value: str) -> str:
40
+ rendered: list[str] = []
41
+ for character in value:
42
+ codepoint = ord(character)
43
+ if character == "`":
44
+ rendered.append(r"\x60")
45
+ elif 0xDC80 <= codepoint <= 0xDCFF:
46
+ rendered.append(f"\\x{codepoint - 0xDC00:02x}")
47
+ elif character.isprintable():
48
+ rendered.append(character)
49
+ elif codepoint <= 0xFF:
50
+ rendered.append(f"\\x{codepoint:02x}")
51
+ elif codepoint <= 0xFFFF:
52
+ rendered.append(f"\\u{codepoint:04x}")
53
+ else:
54
+ rendered.append(f"\\U{codepoint:08x}")
55
+ return "".join(rendered)
56
+
57
+
58
+ def _display_limited(value: str, limit: int) -> str | None:
59
+ rendered: list[str] = []
60
+ used = 0
61
+ for character in value:
62
+ piece = _display(character)
63
+ size = len(piece.encode())
64
+ if used + size > limit:
65
+ return None
66
+ rendered.append(piece)
67
+ used += size
68
+ return "".join(rendered)
69
+
70
+
71
+ def _truncate_at_line_boundary(content: str, limit: int) -> tuple[str, bool]:
72
+ encoded = content.encode()
73
+ if len(encoded) <= limit:
74
+ return content, False
75
+ if limit <= 0:
76
+ return "", True
77
+ boundary = encoded.rfind(b"\n", 0, limit + 1)
78
+ if boundary < 0:
79
+ return "", True
80
+ return encoded[: boundary + 1].decode(), True
81
+
82
+
83
+ def _content_with_marker(content: str, limit: int, truncated: bool) -> tuple[str, bool]:
84
+ if not truncated and len(content.encode()) <= limit:
85
+ return content, False
86
+ marker_size = len(TRUNCATION_MARKER.encode())
87
+ prefix, _was_cut = _truncate_at_line_boundary(content, max(0, limit - marker_size))
88
+ return f"{prefix}{TRUNCATION_MARKER}", True
89
+
90
+
91
+ def _fenced(language: str, body: str) -> str:
92
+ runs = (len(match.group(0)) for match in re.finditer(r"`+", body))
93
+ fence = "`" * max(3, max(runs, default=0) + 1)
94
+ separator = "" if not body or body.endswith("\n") else "\n"
95
+ return f"{fence}{language}\n{body}{separator}{fence}"
96
+
97
+
98
+ def _file_body(source: CollectedFile, limit: int) -> str | None:
99
+ if source.status is not None:
100
+ return None
101
+ if source.selection is not None:
102
+ return source.content
103
+ body, _truncated = _content_with_marker(source.content, limit, source.truncated)
104
+ return body
105
+
106
+
107
+ def _file_payload(source: CollectedFile, limit: int) -> str:
108
+ body = _file_body(source, limit)
109
+ if body is None:
110
+ assert source.status is not None
111
+ return source.status
112
+ return _fenced(source.language, body)
113
+
114
+
115
+ def _change_lines(state: RepositoryState) -> tuple[list[str], bool]:
116
+ lines: list[str] = []
117
+ used_bytes = 0
118
+ for change in state.changes:
119
+ if len(lines) == MAX_CHANGE_ENTRIES:
120
+ return lines, True
121
+ line = f"- `{change.status} {_display(change.path)}`"
122
+ line_bytes = len(f"{line}\n".encode())
123
+ if used_bytes + line_bytes > MAX_CHANGE_MARKDOWN_BYTES:
124
+ return lines, True
125
+ lines.append(line)
126
+ used_bytes += line_bytes
127
+ return lines, False
128
+
129
+
130
+ def _render_git_state(state: RepositoryState) -> str:
131
+ lines = [
132
+ "## Git State",
133
+ "",
134
+ f"- Branch: `{_display(state.branch)}`",
135
+ f"- HEAD: `{_display(state.head)}`",
136
+ f"- Upstream: `{_display(state.upstream)}`",
137
+ f"- Ahead / behind: `{_display(state.ahead_behind)}`",
138
+ f"- Worktree: `{state.worktree}`",
139
+ "",
140
+ "### Working Tree Changes",
141
+ "",
142
+ ]
143
+ changes, truncated = _change_lines(state)
144
+ if changes:
145
+ lines.extend(changes)
146
+ elif not truncated:
147
+ lines.append("No changes.")
148
+ if truncated:
149
+ lines.append("- Additional changes omitted by the 100-entry / 4-KiB limit.")
150
+ return "\n".join(lines)
151
+
152
+
153
+ def _render_source_section(title: str, source: CollectedFile, limit: int) -> str:
154
+ parts = [
155
+ f"## {title}",
156
+ "",
157
+ f"Source: `{source.name}`",
158
+ "",
159
+ _file_payload(source, limit),
160
+ ]
161
+ if source.selection is not None:
162
+ parts.extend(("", render_agents_selection_audit(source.selection)))
163
+ return "\n".join(parts)
164
+
165
+
166
+ def _command_line(command: DeclaredCommand) -> str:
167
+ if command.parse_error:
168
+ return f"- `{command.source}`: unable to parse command declarations."
169
+ assert command.invocation is not None
170
+ invocation = _display(command.invocation)
171
+ if command.target is None:
172
+ return f"- `{invocation}`"
173
+ return f"- `{invocation}` → `{_display(command.target)}`"
174
+
175
+
176
+ def _bounded_lines(lines: list[str], limit: int, *, already_truncated: bool = False) -> str:
177
+ complete = "\n".join(lines)
178
+ if not already_truncated and len(complete.encode()) <= limit:
179
+ return complete
180
+ marker_size = len(TRUNCATION_MARKER.encode())
181
+ budget = max(0, limit - marker_size - 1)
182
+ selected: list[str] = []
183
+ used = 0
184
+ for line in lines:
185
+ size = len(line.encode()) + (1 if selected else 0)
186
+ if used + size > budget:
187
+ break
188
+ selected.append(line)
189
+ used += size
190
+ selected.append(TRUNCATION_MARKER)
191
+ return "\n".join(selected)
192
+
193
+
194
+ def _render_commands(commands: tuple[DeclaredCommand, ...]) -> str:
195
+ lines = [_command_line(command) for command in commands]
196
+ if not lines:
197
+ lines.append("No supported command declarations found.")
198
+ return "\n".join(
199
+ (
200
+ "## Declared Commands",
201
+ "",
202
+ _bounded_lines(lines, DECLARED_COMMANDS_LIMIT_BYTES),
203
+ )
204
+ )
205
+
206
+
207
+ def _entry_body(source: CollectedFile, remaining: int) -> tuple[str | None, int]:
208
+ if source.status is not None:
209
+ return None, 0
210
+ encoded_size = len(source.content.encode())
211
+ aggregate_truncated = encoded_size > remaining
212
+ needs_marker = source.truncated or aggregate_truncated
213
+ marker_size = len(TRUNCATION_MARKER.encode()) if needs_marker else 0
214
+ content_limit = min(remaining, ENTRY_FILE_LIMIT_BYTES - marker_size)
215
+ visible, was_cut = _truncate_at_line_boundary(source.content, max(0, content_limit))
216
+ needs_marker = needs_marker or was_cut
217
+ body = f"{visible}{TRUNCATION_MARKER}" if needs_marker else visible
218
+ return body, len(visible.encode())
219
+
220
+
221
+ def _entry_payload(source: CollectedFile, remaining: int) -> tuple[str, int]:
222
+ body, consumed = _entry_body(source, remaining)
223
+ if body is None:
224
+ assert source.status is not None
225
+ return source.status, consumed
226
+ return _fenced(source.language, body), consumed
227
+
228
+
229
+ def _render_entry_files(entry_files: tuple[CollectedFile, ...]) -> str:
230
+ parts = ["## Project Entry Files"]
231
+ used_content = 0
232
+ for source in entry_files:
233
+ remaining = max(0, ENTRY_FILES_TOTAL_LIMIT_BYTES - used_content)
234
+ payload, consumed = _entry_payload(source, remaining)
235
+ used_content += consumed
236
+ parts.extend(("", f"### `{source.name}`", "", payload))
237
+ return "\n".join(parts)
238
+
239
+
240
+ def _render_recent_commits(commits: tuple[RecentCommit, ...]) -> str | None:
241
+ lines = ["## Recent Commits", ""]
242
+ if not commits:
243
+ lines.append("No commits.")
244
+ return "\n".join(lines)
245
+ used = len(b"## Recent Commits\n\n")
246
+ for commit in commits:
247
+ prefix = f"- `{commit.object_id[:12]}` · "
248
+ date = _display_limited(commit.author_date, GLOBAL_OUTPUT_LIMIT_BYTES)
249
+ if date is None:
250
+ return None
251
+ prefix = f"{prefix}{date} · "
252
+ remaining = GLOBAL_OUTPUT_LIMIT_BYTES - used - len(prefix.encode()) - 1
253
+ subject = _display_limited(commit.subject, max(0, remaining))
254
+ if subject is None:
255
+ return None
256
+ line = f"{prefix}{subject}"
257
+ used += len(line.encode()) + 1
258
+ if used > GLOBAL_OUTPUT_LIMIT_BYTES:
259
+ return None
260
+ lines.append(line)
261
+ return "\n".join(lines)
262
+
263
+
264
+ def _tree_line(entry: TreeEntry) -> str:
265
+ path = _display(entry.path)
266
+ if entry.kind == "directory":
267
+ return f"{path}/"
268
+ if entry.kind == "symlink":
269
+ return f"{path}@"
270
+ if entry.kind == "unreadable_directory":
271
+ return f"{path}/ [Skipped: unreadable.]" if path else "Skipped: unreadable directory."
272
+ return path
273
+
274
+
275
+ def _render_directory_tree(tree: DirectoryTree) -> str:
276
+ lines = [_tree_line(entry) for entry in tree.entries]
277
+ if not lines:
278
+ lines.append("No entries.")
279
+ body = _bounded_lines(
280
+ lines,
281
+ DIRECTORY_TREE_LIMIT_BYTES,
282
+ already_truncated=tree.truncated,
283
+ )
284
+ return "\n".join(("## Directory Tree", "", _fenced("text", body)))
285
+
286
+
287
+ def _omitted_section(title: str) -> str:
288
+ return f"## {title}\n\n{GLOBAL_OMISSION}"
289
+
290
+
291
+ def render_markdown_with_details(state: RepositoryState, project: ProjectContext) -> MarkdownRender:
292
+ """Render all sections in fixed order without exceeding the global byte limit."""
293
+ header = "\n".join(
294
+ (
295
+ "# Codex Project Context",
296
+ "",
297
+ f"- Schema: `{SCHEMA}`",
298
+ f"- Repository: `{_display(os.fspath(state.repository))}`",
299
+ )
300
+ )
301
+ sections: list[tuple[str, str | None]] = [
302
+ ("Git State", _render_git_state(state)),
303
+ (
304
+ "Development Instructions",
305
+ _render_source_section(
306
+ "Development Instructions",
307
+ project.instructions,
308
+ AGENTS_LIMIT_BYTES,
309
+ ),
310
+ ),
311
+ (
312
+ "Project Overview",
313
+ _render_source_section("Project Overview", project.overview, README_LIMIT_BYTES),
314
+ ),
315
+ ("Declared Commands", _render_commands(project.commands)),
316
+ ("Project Entry Files", _render_entry_files(project.entry_files)),
317
+ ("Recent Commits", _render_recent_commits(state.commits)),
318
+ ("Directory Tree", _render_directory_tree(project.directory_tree)),
319
+ ]
320
+
321
+ rendered_parts = [header.encode()]
322
+ included_sections: list[str] = []
323
+ omission_started = False
324
+ for index, (title, full_section) in enumerate(sections):
325
+ omission = _omitted_section(title).encode()
326
+ future_omissions = [
327
+ _omitted_section(future_title).encode()
328
+ for future_title, _future_section in sections[index + 1 :]
329
+ ]
330
+ if omission_started or full_section is None:
331
+ selected = omission
332
+ omission_started = True
333
+ else:
334
+ candidate = full_section.encode()
335
+ tentative = b"\n\n".join((*rendered_parts, candidate, *future_omissions)) + b"\n"
336
+ if len(tentative) <= GLOBAL_OUTPUT_LIMIT_BYTES:
337
+ selected = candidate
338
+ included_sections.append(title)
339
+ else:
340
+ selected = omission
341
+ omission_started = True
342
+ rendered_parts.append(selected)
343
+
344
+ output = b"\n\n".join(rendered_parts) + b"\n"
345
+ if len(output) > GLOBAL_OUTPUT_LIMIT_BYTES:
346
+ raise RuntimeError("global context output limit could not be satisfied")
347
+ return MarkdownRender(output=output, included_sections=tuple(included_sections))
348
+
349
+
350
+ def rendered_source_contents(
351
+ project: ProjectContext, included_sections: tuple[str, ...]
352
+ ) -> tuple[tuple[CollectedFile, str], ...]:
353
+ """Return source-file bodies that actually appear in the final Markdown context."""
354
+ included = frozenset(included_sections)
355
+ rendered: list[tuple[CollectedFile, str]] = []
356
+ if "Development Instructions" in included:
357
+ body = _file_body(project.instructions, AGENTS_LIMIT_BYTES)
358
+ if body is not None:
359
+ rendered.append((project.instructions, body))
360
+ if "Project Overview" in included:
361
+ body = _file_body(project.overview, README_LIMIT_BYTES)
362
+ if body is not None:
363
+ rendered.append((project.overview, body))
364
+ if "Project Entry Files" in included:
365
+ used_content = 0
366
+ for source in project.entry_files:
367
+ remaining = max(0, ENTRY_FILES_TOTAL_LIMIT_BYTES - used_content)
368
+ body, consumed = _entry_body(source, remaining)
369
+ used_content += consumed
370
+ if body is not None:
371
+ rendered.append((source, body))
372
+ return tuple(rendered)
373
+
374
+
375
+ def render_markdown(state: RepositoryState, project: ProjectContext) -> bytes:
376
+ """Render all sections in fixed order without exceeding the global byte limit."""
377
+ return render_markdown_with_details(state, project).output