@pennixrv/trellis 0.6.31 → 0.6.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,508 @@
1
+ #!/usr/bin/env python3
2
+ """Shared materialization for Python sub-agent context.
3
+
4
+ The shared Python Hook renders this projection for implement/check agents and
5
+ ``task.py validate`` reads the same result before a task starts. A notice is
6
+ still useful in a live prompt, but it is never a complete projection.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import os
13
+ from dataclasses import dataclass, field
14
+ from pathlib import Path
15
+
16
+ from .paths import DIR_ARCHIVE, DIR_TASKS, DIR_WORKFLOW
17
+
18
+
19
+ DIRECTORY_MAX_FILES = 20
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class ContextEntry:
24
+ """A normalised JSONL entry consumed by the Python Hook."""
25
+
26
+ path: str
27
+ entry_type: str
28
+ reason: str
29
+ line: int
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class ManifestProblem:
34
+ """A malformed manifest row that the CLI must reject."""
35
+
36
+ line: int
37
+ message: str
38
+
39
+
40
+ @dataclass(frozen=True)
41
+ class ProjectionIssue:
42
+ """Why one declared source was not represented as complete body text."""
43
+
44
+ role: str
45
+ category: str
46
+ path: str
47
+ reason: str
48
+ line: int | None = None
49
+
50
+
51
+ @dataclass
52
+ class ContextProjection:
53
+ """The Hook text and the facts the validator must fail closed on."""
54
+
55
+ role: str
56
+ text: str
57
+ entries: list[ContextEntry] = field(default_factory=list)
58
+ manifest_problems: list[ManifestProblem] = field(default_factory=list)
59
+ issues: list[ProjectionIssue] = field(default_factory=list)
60
+ manifest_exists: bool = False
61
+
62
+
63
+ class _Budget:
64
+ def __init__(self, max_total_bytes: int) -> None:
65
+ self.max_total_bytes = max_total_bytes
66
+ self.used = 0
67
+
68
+ def has_room(self, size: int) -> bool:
69
+ return self.max_total_bytes <= 0 or self.used + size <= self.max_total_bytes
70
+
71
+ def add(self, size: int) -> None:
72
+ self.used += size
73
+
74
+
75
+ def truncate_utf8(data: bytes, cap: int) -> bytes:
76
+ """Truncate at a valid UTF-8 boundary; zero means unlimited."""
77
+ if cap <= 0 or len(data) <= cap:
78
+ return data
79
+
80
+ truncated = data[:cap]
81
+ index = len(truncated)
82
+ while index > 0 and (truncated[index - 1] & 0xC0) == 0x80:
83
+ index -= 1
84
+ if index == 0:
85
+ return b""
86
+
87
+ lead = truncated[index - 1]
88
+ if lead & 0x80:
89
+ if (lead & 0xE0) == 0xC0:
90
+ sequence_length = 2
91
+ elif (lead & 0xF0) == 0xE0:
92
+ sequence_length = 3
93
+ elif (lead & 0xF8) == 0xF0:
94
+ sequence_length = 4
95
+ else:
96
+ sequence_length = 1
97
+ if (index - 1) + sequence_length > len(truncated):
98
+ index -= 1
99
+ return truncated[:index]
100
+
101
+
102
+ def is_binary_content(data: bytes) -> bool:
103
+ if b"\x00" in data:
104
+ return True
105
+ try:
106
+ data.decode("utf-8", errors="strict")
107
+ except UnicodeDecodeError:
108
+ return True
109
+ return False
110
+
111
+
112
+ def parse_context_manifest(jsonl_file: Path) -> tuple[list[ContextEntry], list[ManifestProblem], bool]:
113
+ """Parse the JSONL schema shared by the Hook and validation command."""
114
+ if not jsonl_file.is_file():
115
+ return [], [], False
116
+ try:
117
+ lines = jsonl_file.read_text(encoding="utf-8").splitlines()
118
+ except (OSError, UnicodeDecodeError):
119
+ return [], [ManifestProblem(0, "Manifest is unreadable or is not valid UTF-8")], True
120
+
121
+ entries: list[ContextEntry] = []
122
+ problems: list[ManifestProblem] = []
123
+ for line_number, line in enumerate(lines, start=1):
124
+ if not line.strip():
125
+ continue
126
+ try:
127
+ data = json.loads(line)
128
+ except json.JSONDecodeError:
129
+ problems.append(ManifestProblem(line_number, "Invalid JSON"))
130
+ continue
131
+ if not isinstance(data, dict):
132
+ problems.append(ManifestProblem(line_number, "Expected a JSON object"))
133
+ continue
134
+ if "_example" in data:
135
+ problems.append(
136
+ ManifestProblem(
137
+ line_number,
138
+ "Placeholder `_example` row left by an older task.py create — delete this line, or replace it with "
139
+ '{"file": "<path>", "reason": "<why>"}',
140
+ )
141
+ )
142
+ continue
143
+
144
+ file_value = data.get("file")
145
+ path_value = data.get("path")
146
+ if file_value not in (None, "") and not isinstance(file_value, str):
147
+ problems.append(
148
+ ManifestProblem(line_number, "`file` or legacy `path` must be a string path")
149
+ )
150
+ continue
151
+ if path_value not in (None, "") and not isinstance(path_value, str):
152
+ problems.append(
153
+ ManifestProblem(line_number, "`file` or legacy `path` must be a string path")
154
+ )
155
+ continue
156
+ value = file_value or path_value
157
+ if not value:
158
+ continue
159
+ if not isinstance(value, str):
160
+ problems.append(ManifestProblem(line_number, "`file` or legacy `path` must be a string path"))
161
+ continue
162
+
163
+ entry_type = data.get("type", "file")
164
+ reason = data.get("reason") or "-"
165
+ entries.append(
166
+ ContextEntry(
167
+ path=value,
168
+ entry_type=entry_type if isinstance(entry_type, str) else "file",
169
+ reason=reason if isinstance(reason, str) else str(reason),
170
+ line=line_number,
171
+ )
172
+ )
173
+ return entries, problems, True
174
+
175
+
176
+ def _real_path_contained(base_real: str, target_real: str) -> bool:
177
+ try:
178
+ return os.path.commonpath([base_real, target_real]) == base_real
179
+ except ValueError:
180
+ return False
181
+
182
+
183
+ def _is_allowed_path(repo_root: Path, candidate: Path) -> bool:
184
+ try:
185
+ root_real = os.path.realpath(repo_root)
186
+ workflow_real = os.path.realpath(repo_root / DIR_WORKFLOW)
187
+ candidate_real = os.path.realpath(candidate)
188
+ except OSError:
189
+ return False
190
+ return _real_path_contained(root_real, candidate_real) or _real_path_contained(
191
+ workflow_real, candidate_real
192
+ )
193
+
194
+
195
+ def resolve_context_entry_path(file_path: str, repo_root: Path, task_dir: Path | None) -> Path | None:
196
+ """Resolve an entry and preserve archived task self-reference semantics."""
197
+ repo_path = repo_root / file_path
198
+ if task_dir is None:
199
+ return repo_path
200
+
201
+ try:
202
+ task_parts = task_dir.resolve().relative_to(repo_root.resolve()).parts
203
+ except ValueError:
204
+ return repo_path
205
+
206
+ archive_prefix = (DIR_WORKFLOW, DIR_TASKS, DIR_ARCHIVE)
207
+ if len(task_parts) != 5 or task_parts[:3] != archive_prefix:
208
+ return repo_path
209
+
210
+ year_month = task_parts[3]
211
+ if (
212
+ len(year_month) != 7
213
+ or year_month[4] != "-"
214
+ or not year_month[:4].isdigit()
215
+ or not year_month[5:].isdigit()
216
+ ):
217
+ return repo_path
218
+
219
+ historical_root = f"{DIR_WORKFLOW}/{DIR_TASKS}/{task_dir.name}"
220
+ posix_path = file_path.replace("\\", "/")
221
+ if posix_path == historical_root:
222
+ relative_parts: tuple[str, ...] = ()
223
+ elif posix_path.startswith(f"{historical_root}/"):
224
+ relative_path = posix_path[len(historical_root) + 1 :].rstrip("/")
225
+ relative_parts = tuple(relative_path.split("/")) if relative_path else ()
226
+ if any(part in ("", ".", "..") for part in relative_parts):
227
+ return None
228
+ else:
229
+ return repo_path
230
+
231
+ try:
232
+ archive_root = task_dir.resolve()
233
+ resolved_path = task_dir.joinpath(*relative_parts).resolve()
234
+ resolved_path.relative_to(archive_root)
235
+ except (OSError, RuntimeError, ValueError):
236
+ return None
237
+ return resolved_path
238
+
239
+
240
+ def _display_task_path(repo_root: Path, task_dir: Path, file_name: str) -> str:
241
+ try:
242
+ return (task_dir.relative_to(repo_root) / file_name).as_posix()
243
+ except ValueError:
244
+ return f"{task_dir.as_posix().rstrip('/')}/{file_name}"
245
+
246
+
247
+ def _truncate_notice(path: str, cap: int) -> str:
248
+ return f"\n[Trellis: truncated at {cap} bytes — read {path} for the full content]"
249
+
250
+
251
+ def _binary_notice(path: str, size: int, reason: str) -> str:
252
+ return f"[Trellis: not inlined (binary file) — {path} ({size} bytes): {reason}]"
253
+
254
+
255
+ def _index_notice(path: str, size: int, reason: str) -> str:
256
+ return f"[Trellis: not inlined (total context limit reached) — {path} ({size} bytes): {reason}]"
257
+
258
+
259
+ def _append_block(
260
+ projection: ContextProjection,
261
+ budget: _Budget,
262
+ *,
263
+ header: str,
264
+ path: str,
265
+ content: str,
266
+ size: int,
267
+ reason: str,
268
+ category: str,
269
+ line: int | None,
270
+ ) -> str:
271
+ block = f"=== {header} ===\n{content}"
272
+ block_bytes = len(block.encode("utf-8"))
273
+ if budget.has_room(block_bytes):
274
+ budget.add(block_bytes)
275
+ return block
276
+
277
+ notice = _index_notice(path, size, reason)
278
+ budget.add(len(notice.encode("utf-8")))
279
+ limit = budget.max_total_bytes
280
+ projection.issues.append(
281
+ ProjectionIssue(
282
+ role=projection.role,
283
+ category="total",
284
+ path=path,
285
+ reason=f"exceeds context_injection.max_total_bytes ({limit}); injection uses an index notice",
286
+ line=line,
287
+ )
288
+ )
289
+ return notice
290
+
291
+
292
+ def _new_projection(role: str, entries: list[ContextEntry], problems: list[ManifestProblem], manifest_exists: bool) -> ContextProjection:
293
+ projection = ContextProjection(
294
+ role=role,
295
+ text="",
296
+ entries=entries,
297
+ manifest_problems=problems,
298
+ manifest_exists=manifest_exists,
299
+ )
300
+ return projection
301
+
302
+
303
+ def _read_material(
304
+ repo_root: Path, target: Path | None, category: str, path: str, projection: ContextProjection, line: int | None
305
+ ) -> bytes | None:
306
+ if target is None:
307
+ projection.issues.append(
308
+ ProjectionIssue(projection.role, category, path, "path cannot be resolved safely", line)
309
+ )
310
+ return None
311
+ if not _is_allowed_path(repo_root, target):
312
+ projection.issues.append(
313
+ ProjectionIssue(projection.role, category, path, "path escapes the allowed project roots", line)
314
+ )
315
+ return None
316
+ try:
317
+ if not target.is_file():
318
+ raise OSError
319
+ return target.read_bytes()
320
+ except OSError:
321
+ projection.issues.append(
322
+ ProjectionIssue(projection.role, category, path, "file is missing or unreadable", line)
323
+ )
324
+ return None
325
+
326
+
327
+ def _materialize_file(
328
+ repo_root: Path,
329
+ target: Path | None,
330
+ path: str,
331
+ reason: str,
332
+ limits: dict[str, int],
333
+ budget: _Budget,
334
+ projection: ContextProjection,
335
+ *,
336
+ category: str,
337
+ line: int | None,
338
+ header: str | None = None,
339
+ cap_key: str = "max_file_bytes",
340
+ ) -> str | None:
341
+ data = _read_material(repo_root, target, category, path, projection, line)
342
+ if data is None:
343
+ return None
344
+
345
+ size = len(data)
346
+ if is_binary_content(data):
347
+ notice = _binary_notice(path, size, reason)
348
+ budget.add(len(notice.encode("utf-8")))
349
+ projection.issues.append(
350
+ ProjectionIssue(projection.role, category, path, "binary content is not inlined", line)
351
+ )
352
+ return notice
353
+
354
+ cap = limits[cap_key]
355
+ materialized = truncate_utf8(data, cap)
356
+ content = materialized.decode("utf-8")
357
+ if len(materialized) < size:
358
+ content += _truncate_notice(path, cap)
359
+ projection.issues.append(
360
+ ProjectionIssue(
361
+ projection.role,
362
+ category,
363
+ path,
364
+ f"exceeds context_injection.{cap_key} ({cap}); injection truncates it",
365
+ line,
366
+ )
367
+ )
368
+ return _append_block(
369
+ projection,
370
+ budget,
371
+ header=header or path,
372
+ path=path,
373
+ content=content,
374
+ size=size,
375
+ reason=reason,
376
+ category=category,
377
+ line=line,
378
+ )
379
+
380
+
381
+ def _materialize_directory(
382
+ repo_root: Path,
383
+ entry: ContextEntry,
384
+ task_dir: Path,
385
+ limits: dict[str, int],
386
+ budget: _Budget,
387
+ projection: ContextProjection,
388
+ ) -> list[str]:
389
+ target = resolve_context_entry_path(entry.path, repo_root, task_dir)
390
+ if target is None or not _is_allowed_path(repo_root, target) or not target.is_dir():
391
+ reason = "directory is missing, unreadable, or escapes the allowed project roots"
392
+ projection.issues.append(
393
+ ProjectionIssue(projection.role, "directory", entry.path, reason, entry.line)
394
+ )
395
+ return []
396
+ try:
397
+ files = sorted(
398
+ child for child in target.iterdir() if child.suffix == ".md" and child.is_file()
399
+ )
400
+ except OSError:
401
+ projection.issues.append(
402
+ ProjectionIssue(projection.role, "directory", entry.path, "directory is unreadable", entry.line)
403
+ )
404
+ return []
405
+
406
+ if not files:
407
+ projection.issues.append(
408
+ ProjectionIssue(
409
+ projection.role, "directory", entry.path, "directory has no direct Markdown files to inject", entry.line
410
+ )
411
+ )
412
+ return []
413
+ if len(files) > DIRECTORY_MAX_FILES:
414
+ projection.issues.append(
415
+ ProjectionIssue(
416
+ projection.role,
417
+ "directory",
418
+ entry.path,
419
+ f"directory contains {len(files)} Markdown files; injection selects only the first {DIRECTORY_MAX_FILES}",
420
+ entry.line,
421
+ )
422
+ )
423
+
424
+ blocks: list[str] = []
425
+ for child in files[:DIRECTORY_MAX_FILES]:
426
+ display = f"{entry.path.rstrip('/')}/{child.name}"
427
+ block = _materialize_file(
428
+ repo_root,
429
+ child,
430
+ display,
431
+ entry.reason,
432
+ limits,
433
+ budget,
434
+ projection,
435
+ category="file",
436
+ line=entry.line,
437
+ )
438
+ if block:
439
+ blocks.append(block)
440
+ return blocks
441
+
442
+
443
+ def project_agent_context(
444
+ repo_root: Path, task_dir: Path, agent_type: str, limits: dict[str, int]
445
+ ) -> ContextProjection:
446
+ """Render the exact Python Hook context and record every loss of body text."""
447
+ jsonl_name = f"{agent_type}.jsonl"
448
+ entries, manifest_problems, manifest_exists = parse_context_manifest(task_dir / jsonl_name)
449
+ projection = _new_projection(agent_type, entries, manifest_problems, manifest_exists)
450
+ budget = _Budget(limits["max_total_bytes"])
451
+ blocks: list[str] = []
452
+
453
+ for entry in entries:
454
+ if entry.entry_type == "directory":
455
+ blocks.extend(_materialize_directory(repo_root, entry, task_dir, limits, budget, projection))
456
+ continue
457
+ target = resolve_context_entry_path(entry.path, repo_root, task_dir)
458
+ block = _materialize_file(
459
+ repo_root,
460
+ target,
461
+ entry.path,
462
+ entry.reason,
463
+ limits,
464
+ budget,
465
+ projection,
466
+ category="file",
467
+ line=entry.line,
468
+ )
469
+ if block:
470
+ blocks.append(block)
471
+
472
+ if blocks:
473
+ context_parts = ["\n\n".join(blocks)]
474
+ else:
475
+ context_parts = [
476
+ f"[Trellis] {_display_task_path(repo_root, task_dir, jsonl_name)} has no curated entries, so no spec/research "
477
+ "context was injected. Before working, read the guidelines relevant to the code you will touch under "
478
+ ".trellis/spec/, and treat the task artifacts below as the only prepared context."
479
+ ]
480
+
481
+ artifact_specs = (
482
+ ("prd.md", "Requirements", "Requirements document"),
483
+ ("design.md", "Technical Design", "Technical design document"),
484
+ ("implement.md", "Execution Plan", "Execution plan document"),
485
+ )
486
+ for file_name, label, reason in artifact_specs:
487
+ target = task_dir / file_name
488
+ if not target.exists():
489
+ continue
490
+ display = _display_task_path(repo_root, task_dir, file_name)
491
+ block = _materialize_file(
492
+ repo_root,
493
+ target,
494
+ display,
495
+ reason,
496
+ limits,
497
+ budget,
498
+ projection,
499
+ category="artifact",
500
+ line=None,
501
+ header=f"{display} ({label})",
502
+ cap_key="max_artifact_bytes",
503
+ )
504
+ if block:
505
+ context_parts.append(block)
506
+
507
+ projection.text = "\n\n".join(context_parts)
508
+ return projection