open-code-review-toolkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. ocr_toolkit/__init__.py +1 -0
  2. ocr_toolkit/_version.py +24 -0
  3. ocr_toolkit/cli.py +56 -0
  4. ocr_toolkit/common/__init__.py +1 -0
  5. ocr_toolkit/common/language.py +60 -0
  6. ocr_toolkit/common/markdown.py +214 -0
  7. ocr_toolkit/common/redaction.py +324 -0
  8. ocr_toolkit/config_writer.py +106 -0
  9. ocr_toolkit/configure.py +194 -0
  10. ocr_toolkit/context/__init__.py +1 -0
  11. ocr_toolkit/context/__main__.py +8 -0
  12. ocr_toolkit/context/ansible.py +561 -0
  13. ocr_toolkit/context/categorize.py +162 -0
  14. ocr_toolkit/context/instructions.py +269 -0
  15. ocr_toolkit/context/manifests.py +459 -0
  16. ocr_toolkit/context/planner.py +227 -0
  17. ocr_toolkit/context/render.py +955 -0
  18. ocr_toolkit/context/repo.py +672 -0
  19. ocr_toolkit/context/settings.py +97 -0
  20. ocr_toolkit/mcp_config.py +257 -0
  21. ocr_toolkit/posting/__init__.py +1 -0
  22. ocr_toolkit/posting/__main__.py +8 -0
  23. ocr_toolkit/posting/comments.py +87 -0
  24. ocr_toolkit/posting/formatting.py +755 -0
  25. ocr_toolkit/posting/gitlab.py +853 -0
  26. ocr_toolkit/posting/markers.py +284 -0
  27. ocr_toolkit/posting/payloads.py +141 -0
  28. ocr_toolkit/posting/result.py +116 -0
  29. ocr_toolkit/posting/settings.py +181 -0
  30. ocr_toolkit/posting/snapshot.py +468 -0
  31. ocr_toolkit/posting/workflow.py +873 -0
  32. ocr_toolkit/preflight.py +395 -0
  33. ocr_toolkit/py.typed +1 -0
  34. open_code_review_toolkit-0.1.0.dist-info/METADATA +283 -0
  35. open_code_review_toolkit-0.1.0.dist-info/RECORD +38 -0
  36. open_code_review_toolkit-0.1.0.dist-info/WHEEL +4 -0
  37. open_code_review_toolkit-0.1.0.dist-info/entry_points.txt +2 -0
  38. open_code_review_toolkit-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,955 @@
1
+ """Build and summarize the OCR review background Markdown."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import os
7
+ import re
8
+ import sys
9
+ from collections.abc import Sequence
10
+ from typing import Any
11
+
12
+ from ocr_toolkit.common.language import resolve_review_language
13
+ from ocr_toolkit.common.markdown import (
14
+ markdown_code_block,
15
+ markdown_fence_transition,
16
+ open_markdown_fence,
17
+ )
18
+ from ocr_toolkit.common.redaction import (
19
+ SENSITIVE_NAMED_KEY_PATTERN,
20
+ redact_env_secret_values,
21
+ redact_sensitive,
22
+ redact_url_userinfo,
23
+ redact_url_userinfo_only,
24
+ strip_path_controls,
25
+ )
26
+ from ocr_toolkit.context import repo as context_repo
27
+ from ocr_toolkit.context.ansible import (
28
+ detect_root_ansible_playbooks,
29
+ extract_application_versions,
30
+ extract_inventory_topology,
31
+ is_inventory_topology_file,
32
+ parse_ansible_requirement_version_pins,
33
+ parse_ansible_requirements,
34
+ )
35
+ from ocr_toolkit.context.categorize import (
36
+ PYTHON_MANIFEST_PATTERN,
37
+ active_context_providers,
38
+ categorize_files,
39
+ )
40
+ from ocr_toolkit.context.instructions import (
41
+ read_accepted_decisions,
42
+ read_project_instructions,
43
+ )
44
+ from ocr_toolkit.context.manifests import (
45
+ discover_package_json_paths,
46
+ discover_pyproject_paths,
47
+ parse_composer_json,
48
+ parse_composer_lock,
49
+ parse_go_mod,
50
+ parse_package_json,
51
+ parse_pyproject,
52
+ parse_requirements_txt,
53
+ )
54
+ from ocr_toolkit.context.planner import ContextSection, render_context, split_markdown_sections
55
+ from ocr_toolkit.context.settings import (
56
+ DEFAULT_BACKGROUND_MAX_BYTES,
57
+ DEFAULT_BACKGROUND_MAX_CHARS,
58
+ MAX_BACKGROUND_MAX_BYTES,
59
+ MAX_BACKGROUND_MAX_CHARS,
60
+ getenv_int,
61
+ inline_code,
62
+ string_list_value,
63
+ string_value,
64
+ )
65
+
66
+
67
+ def safe_text(value: str) -> str:
68
+ """Redact repository-controlled text before rendering it to context."""
69
+
70
+ return redact_sensitive(redact_url_userinfo(value))
71
+
72
+
73
+ def safe_inline_value(value: str) -> str:
74
+ """Return a redacted inline-code value for context output."""
75
+
76
+ return inline_code(safe_text(value))
77
+
78
+
79
+ def safe_inline_path(value: str) -> str:
80
+ """Return a Markdown-safe repository path without generic key rewrites."""
81
+
82
+ cleaned = redact_env_secret_values(redact_url_userinfo_only(strip_path_controls(value)))
83
+ cleaned = re.sub(
84
+ rf"(?i)(^|[/?&;]|\b)((?:{SENSITIVE_NAMED_KEY_PATTERN})(?:=|[-_:]))"
85
+ r"[^/?&;`\s]+",
86
+ r"\1\2***",
87
+ cleaned,
88
+ )
89
+ cleaned = re.sub(
90
+ r"(?i)(^|/)(?:password|passwd|secret|secrets|token|api[_-]?key|x-api-key|"
91
+ r"auth[_-]?token|access[_-]?token|refresh[_-]?token|private[_-]?token|"
92
+ r"client[_-]?secret|secret[_-]?key|aws[_-]?secret[_-]?access[_-]?key)(?=\.|/|$)",
93
+ r"\1***",
94
+ cleaned,
95
+ )
96
+ cleaned = re.sub(
97
+ r"(?i)(^|/)(?:id_)?(?:rsa|dsa|ecdsa|ed25519)(?=\.|/|$)",
98
+ r"\1***",
99
+ cleaned,
100
+ )
101
+ cleaned = re.sub(
102
+ r"(?i)(^|/)(?:private[_-]?key)(?=\.|/|$)",
103
+ r"\1***",
104
+ cleaned,
105
+ )
106
+ return inline_code(cleaned)
107
+
108
+
109
+ def format_items(items: Sequence[str], limit: int = 80) -> str:
110
+ """Format a bounded markdown bullet list.
111
+
112
+ Most dependency/runtime metadata reaches the review background through
113
+ this boundary. Redact URL userinfo here so future manifest parsers do not
114
+ need to remember the same credential-safety rule individually.
115
+ """
116
+
117
+ if not items:
118
+ return "- none detected"
119
+
120
+ clipped = list(items[:limit])
121
+ lines = [f"- {safe_inline_value(item)}" for item in clipped]
122
+ if len(items) > limit:
123
+ lines.append(f"- ... and {len(items) - limit} more")
124
+ return "\n".join(lines)
125
+
126
+
127
+ def format_paths(items: Sequence[str], limit: int = 80) -> str:
128
+ """Format repository paths without generic secret-key rewrites."""
129
+
130
+ if not items:
131
+ return "- none detected"
132
+
133
+ clipped = list(items[:limit])
134
+ lines = [f"- {safe_inline_path(item)}" for item in clipped]
135
+ if len(items) > limit:
136
+ lines.append(f"- ... and {len(items) - limit} more")
137
+ return "\n".join(lines)
138
+
139
+
140
+ def format_manifest_items(items: Sequence[str], omitted: int = 0, limit: int = 80) -> str:
141
+ """Format manifest items while preserving parser-level omitted counts."""
142
+
143
+ if not items:
144
+ if omitted > 0:
145
+ return f"- ... and {omitted} more"
146
+ return "- none detected"
147
+
148
+ clipped = list(items[:limit])
149
+ lines = [f"- {safe_inline_value(item)}" for item in clipped]
150
+ total_omitted = max(0, len(items) - len(clipped)) + max(0, omitted)
151
+ if total_omitted:
152
+ lines.append(f"- ... and {total_omitted} more")
153
+ return "\n".join(lines)
154
+
155
+
156
+ def add_section(
157
+ lines: list[str],
158
+ title: str,
159
+ items: Sequence[str],
160
+ limit: int = 80,
161
+ *,
162
+ paths: bool = True,
163
+ ) -> None:
164
+ """Append a markdown section containing a bounded list."""
165
+
166
+ lines.append(f"## {title}")
167
+ formatter = format_paths if paths else format_items
168
+ lines.append(formatter(items, limit=limit))
169
+ lines.append("")
170
+
171
+
172
+ def format_category_summary(categories: dict[str, list[str]], limit: int = 12) -> str:
173
+ """Render changed-file categories as dense, bounded path samples."""
174
+
175
+ if not categories:
176
+ return "- none detected"
177
+
178
+ lines: list[str] = []
179
+ shown: set[str] = set()
180
+ for category, files in categories.items():
181
+ unique_files = [path for path in files if path not in shown]
182
+ samples = [safe_inline_path(path) for path in unique_files[:limit]]
183
+ omitted = len(unique_files) - len(samples)
184
+ if samples:
185
+ suffix = f"; +{omitted} more" if omitted else ""
186
+ lines.append(f"- {category} ({len(files)}): {', '.join(samples)}{suffix}")
187
+ else:
188
+ lines.append(f"- {category} ({len(files)}): overlaps prior categories")
189
+ shown.update(files)
190
+ return "\n".join(lines)
191
+
192
+
193
+ def limit_text_bytes(text: str, max_bytes: int) -> str:
194
+ """Limit text to a strict UTF-8 byte budget without raising decode errors."""
195
+
196
+ if max_bytes <= 0:
197
+ return ""
198
+
199
+ encoded = text.encode("utf-8")
200
+ if len(encoded) <= max_bytes:
201
+ return text
202
+
203
+ notice = (
204
+ "\n\n## Context truncation notice\n"
205
+ f"- Review background was truncated to {max_bytes} bytes.\n"
206
+ )
207
+ notice_bytes = notice.encode("utf-8")
208
+
209
+ if len(notice_bytes) >= max_bytes:
210
+ return notice_bytes[:max_bytes].decode("utf-8", errors="ignore")
211
+
212
+ clip_budget = max_bytes - len(notice_bytes)
213
+ clipped = encoded[:clip_budget].decode("utf-8", errors="ignore").rstrip()
214
+
215
+ closing_fence = ""
216
+ while True:
217
+ open_fence = open_markdown_fence(clipped)
218
+ next_closing_fence = f"\n{open_fence}" if open_fence is not None else ""
219
+ closing_fence_bytes = next_closing_fence.encode("utf-8")
220
+ if not next_closing_fence:
221
+ closing_fence = ""
222
+ break
223
+ if len(clipped.encode("utf-8")) + len(closing_fence_bytes) + len(notice_bytes) <= max_bytes:
224
+ closing_fence = next_closing_fence
225
+ break
226
+
227
+ clip_budget = max(
228
+ 0,
229
+ max_bytes - len(notice_bytes) - len(closing_fence_bytes),
230
+ )
231
+ shortened = encoded[:clip_budget].decode("utf-8", errors="ignore").rstrip()
232
+ if shortened == clipped:
233
+ closing_fence = next_closing_fence
234
+ break
235
+ clipped = shortened
236
+
237
+ result = clipped + closing_fence + notice
238
+
239
+ # Final guard: never exceed the declared byte budget.
240
+ result_bytes = result.encode("utf-8")
241
+ if len(result_bytes) <= max_bytes:
242
+ return result
243
+
244
+ return result_bytes[:max_bytes].decode("utf-8", errors="ignore")
245
+
246
+
247
+ def build_context() -> str:
248
+ """Build the complete Markdown review background."""
249
+
250
+ response_language = resolve_review_language()
251
+ raw_changed = context_repo.changed_files()
252
+ changed_unavailable = raw_changed is None
253
+ changed: list[str] = list(raw_changed) if raw_changed is not None else []
254
+ categories = categorize_files(changed)
255
+ active_categories = set(categories)
256
+ active_providers = active_context_providers(changed, categories)
257
+ ansible_active = "ansible" in active_providers
258
+ python_active = "python" in active_providers
259
+ go_active = "go" in active_providers
260
+ php_active = "php" in active_providers
261
+ javascript_active = "javascript" in active_providers
262
+ versions_active = bool(
263
+ active_categories
264
+ & {
265
+ "ansible_playbooks",
266
+ "ansible_roles",
267
+ "ci",
268
+ "containers",
269
+ "dependency_manifests",
270
+ "go",
271
+ "javascript_typescript",
272
+ "php",
273
+ "python",
274
+ "terraform_hcl",
275
+ "templates",
276
+ }
277
+ )
278
+
279
+ ansible_core_paths = (
280
+ context_repo.rel_glob(
281
+ [
282
+ "ansible.cfg",
283
+ "requirements.yml",
284
+ "requirements.yaml",
285
+ "collections/requirements.yml",
286
+ "collections/requirements.yaml",
287
+ ],
288
+ limit=20,
289
+ )
290
+ if ansible_active
291
+ else []
292
+ )
293
+ role_metadata_paths = (
294
+ context_repo.rel_glob(
295
+ [
296
+ "roles/**/meta/main.yml",
297
+ "roles/**/defaults/main.yml",
298
+ "roles/**/vars/main.yml",
299
+ ],
300
+ limit=120,
301
+ )
302
+ if ansible_active
303
+ else []
304
+ )
305
+
306
+ ansible_requirements: list[str] = []
307
+ ansible_requirement_versions: list[str] = []
308
+ for req_path in (
309
+ context_repo.rel_glob(
310
+ [
311
+ "requirements.yml",
312
+ "requirements.yaml",
313
+ "collections/requirements.yml",
314
+ "collections/requirements.yaml",
315
+ ],
316
+ limit=20,
317
+ )
318
+ if ansible_active
319
+ else []
320
+ ):
321
+ ansible_requirements.extend(parse_ansible_requirements(context_repo.ROOT / req_path))
322
+ ansible_requirement_versions.extend(
323
+ parse_ansible_requirement_version_pins(context_repo.ROOT / req_path)
324
+ )
325
+
326
+ root_playbooks = detect_root_ansible_playbooks() if ansible_active else []
327
+ pyproject_discovery = discover_pyproject_paths(changed) if python_active else None
328
+ pyproject_paths = pyproject_discovery.paths if pyproject_discovery else []
329
+ pyprojects: list[tuple[str, dict[str, Any]]] = [
330
+ (rel, parse_pyproject(context_repo.ROOT / rel)) for rel in pyproject_paths
331
+ ]
332
+ requirements: list[tuple[str, dict[str, Any]]] = []
333
+ root_requirement_paths = [
334
+ path
335
+ for path in (
336
+ "requirements.txt",
337
+ "requirements.in",
338
+ "constraints.txt",
339
+ "constraints.in",
340
+ )
341
+ if python_active and context_repo.path_exists(path)
342
+ ]
343
+ changed_requirements = [
344
+ path
345
+ for path in categories.get("dependency_manifests", [])
346
+ if PYTHON_MANIFEST_PATTERN.search(path) and path.lower().endswith((".in", ".txt"))
347
+ ]
348
+ all_requirement_paths = list(dict.fromkeys([*root_requirement_paths, *changed_requirements]))
349
+ requirement_paths = all_requirement_paths[:30]
350
+ omitted_requirement_paths = max(0, len(all_requirement_paths) - len(requirement_paths))
351
+ for req_path in requirement_paths if python_active else []:
352
+ requirements.append((req_path, parse_requirements_txt(context_repo.ROOT / req_path)))
353
+
354
+ go_mod = (
355
+ parse_go_mod(context_repo.ROOT / "go.mod")
356
+ if go_active and context_repo.path_exists("go.mod")
357
+ else {}
358
+ )
359
+ composer_json = (
360
+ parse_composer_json(context_repo.ROOT / "composer.json")
361
+ if php_active and context_repo.path_exists("composer.json")
362
+ else {}
363
+ )
364
+ composer_lock = (
365
+ parse_composer_lock(context_repo.ROOT / "composer.lock")
366
+ if php_active and context_repo.path_exists("composer.lock")
367
+ else {}
368
+ )
369
+ package_json_discovery = discover_package_json_paths(changed) if javascript_active else None
370
+ package_json_paths = package_json_discovery.paths if package_json_discovery else []
371
+
372
+ app_versions = (
373
+ extract_application_versions(changed, include_discovered=True)
374
+ + ansible_requirement_versions
375
+ if versions_active
376
+ else []
377
+ )
378
+ manifest_paths = sorted(categories.get("dependency_manifests", []))
379
+ inventory_topology_paths = (
380
+ [
381
+ rel_path
382
+ for rel_path in context_repo.rel_glob_files(
383
+ [
384
+ "inventory",
385
+ "inventory.*",
386
+ "inventory/**/*",
387
+ "inventories",
388
+ "inventories/**/*",
389
+ "**/hosts",
390
+ "**/hosts.*",
391
+ "**/inventory",
392
+ "**/inventory.*",
393
+ ],
394
+ limit=240,
395
+ exclude_dirs=context_repo.DEFAULT_EXCLUDE_DIRS | {".review-context"},
396
+ )
397
+ if is_inventory_topology_file(rel_path)
398
+ ]
399
+ if ansible_active
400
+ else []
401
+ )
402
+ inventory_groups = extract_inventory_topology(inventory_topology_paths)
403
+ # When context_repo.changed_files() failed, do NOT include reviewer guidance or
404
+ # accepted-decisions: we cannot tell whether the current MR is
405
+ # editing them, which would let an MR self-whitelist its own
406
+ # findings. The MR will simply review without those constraints.
407
+ instructions = [] if changed_unavailable else read_project_instructions(changed_paths=changed)
408
+
409
+ lines: list[str] = []
410
+ lines.append("# Review Background")
411
+ lines.append("")
412
+ lines.append(f"Response language: {response_language}.")
413
+ lines.append(
414
+ "All user-visible review comments, summaries, warnings and recommendations MUST be written in this response "
415
+ "language. Keep code identifiers, file paths, function names, package names and error names unchanged."
416
+ )
417
+ lines.append("")
418
+ lines.append("Use this context as hard constraints for the review.")
419
+ lines.append(
420
+ "If a suggested API or behavior conflicts with detected runtime, dependency, provider or application versions, "
421
+ "do not suggest it."
422
+ )
423
+ lines.append(
424
+ "If version-specific knowledge is unavailable, state the uncertainty explicitly instead of recommending an "
425
+ "obsolete API."
426
+ )
427
+ lines.append("Review only changed code and directly related context.")
428
+ lines.append("Post only high-confidence findings.")
429
+ lines.append("Do not block the merge request.")
430
+ lines.append("")
431
+
432
+ lines.append("## Merge Request")
433
+ lines.append(
434
+ "- Target branch: "
435
+ f"{context_repo.inline_ci_value('CI_MERGE_REQUEST_TARGET_BRANCH_NAME', use_default_branch=True)}"
436
+ )
437
+ lines.append(
438
+ "- Source branch: "
439
+ f"{context_repo.inline_ci_value('CI_MERGE_REQUEST_SOURCE_BRANCH_NAME', fallback_git_args=['branch', '--show-current'])}"
440
+ )
441
+ pipeline_sha = os.environ.get("CI_COMMIT_SHA", "").strip()
442
+ source_sha = os.environ.get("CI_MERGE_REQUEST_SOURCE_BRANCH_SHA", "").strip()
443
+ source_sha_available = bool(source_sha) and set(source_sha) != {"0"}
444
+ if source_sha_available:
445
+ lines.append(f"- Source commit SHA: {inline_code(source_sha)}")
446
+ else:
447
+ lines.append(
448
+ "- Commit SHA: "
449
+ f"{context_repo.inline_ci_value('CI_COMMIT_SHA', fallback_git_args=['rev-parse', 'HEAD'])}"
450
+ )
451
+ if source_sha_available and pipeline_sha and pipeline_sha != source_sha:
452
+ lines.append(
453
+ "- Pipeline commit SHA: "
454
+ f"{inline_code(pipeline_sha)} _(pipeline ref; may be synthetic in merged-result pipelines)_"
455
+ )
456
+ if changed_unavailable:
457
+ lines.append(
458
+ "- Changed files: unavailable; git diff failed, so project instruction files and accepted decisions are "
459
+ "intentionally omitted to avoid MR self-whitelisting."
460
+ )
461
+ else:
462
+ lines.append(f"- Changed files: {len(changed)} detected")
463
+ lines.append("")
464
+
465
+ accepted_decisions = (
466
+ "" if changed_unavailable else read_accepted_decisions(changed_paths=changed)
467
+ )
468
+ if accepted_decisions:
469
+ lines.append("## Accepted project decisions")
470
+ lines.append(
471
+ "The following items are intentionally accepted by the project. "
472
+ "Do not raise them as new review findings. If you would normally "
473
+ "comment on one, skip it; if you must acknowledge it, mark it as "
474
+ "an accepted decision and move on."
475
+ )
476
+ lines.append("")
477
+ # Demote the embedded document below `##` so its own headings do not
478
+ # pollute the top-level section count in the stdout summary.
479
+ safe_accepted_decisions = redact_sensitive(redact_url_userinfo(accepted_decisions))
480
+ demoted = re.sub(
481
+ r"(?m)^(#{1,5}) ",
482
+ lambda match: "#" * min(6, len(match.group(1)) + 2) + " ",
483
+ safe_accepted_decisions,
484
+ )
485
+ lines.append(demoted)
486
+ lines.append("")
487
+
488
+ lines.append("## Changed files by category")
489
+ if changed_unavailable:
490
+ lines.append(
491
+ "- unavailable: changed-file discovery failed; review should rely on OCR diff input and directly related "
492
+ "repository context instead of treating this as an empty MR."
493
+ )
494
+ lines.append("")
495
+ elif categories:
496
+ lines.append(format_category_summary(categories))
497
+ lines.append("")
498
+ else:
499
+ lines.append("- none detected")
500
+ lines.append("")
501
+
502
+ if instructions:
503
+ lines.append("## Project instruction files")
504
+ lines.append(
505
+ "The following bounded excerpts were found in repository instruction files. "
506
+ "Treat them as project guidance, but prefer concrete diff evidence over generic style preferences."
507
+ )
508
+ lines.append("")
509
+ for rel_path, excerpt in instructions:
510
+ lines.append(f"### {safe_inline_path(rel_path)}")
511
+ safe_excerpt = redact_sensitive(redact_url_userinfo(excerpt[:8_000]))
512
+ lines.append(markdown_code_block("markdown", safe_excerpt))
513
+ lines.append("")
514
+
515
+ add_section(lines, "Ansible core manifests", ansible_core_paths, limit=20)
516
+ add_section(lines, "Role defaults and metadata files", role_metadata_paths, limit=60)
517
+ add_section(lines, "Detected root Ansible playbook entrypoints", root_playbooks, limit=120)
518
+ add_section(lines, "Ansible inventory topology files", inventory_topology_paths, limit=120)
519
+ add_section(lines, "Ansible inventory groups", inventory_groups, limit=120, paths=False)
520
+ add_section(lines, "Ansible Galaxy requirements", ansible_requirements, limit=100, paths=False)
521
+
522
+ ansible_version = context_repo.tool_version("ansible", ["--version"]) if ansible_active else []
523
+ ansible_lint_version = (
524
+ context_repo.tool_version("ansible-lint", ["--version"]) if ansible_active else []
525
+ )
526
+ add_section(lines, "Detected ansible --version", ansible_version, limit=8, paths=False)
527
+ add_section(
528
+ lines,
529
+ "Detected ansible-lint --version",
530
+ ansible_lint_version,
531
+ limit=8,
532
+ paths=False,
533
+ )
534
+
535
+ lines.append("## Python context")
536
+ emitted_python_context = False
537
+ for rel_path, pyproject in pyprojects:
538
+ if not pyproject:
539
+ continue
540
+ header_suffix = "" if rel_path == "pyproject.toml" else f" ({safe_inline_path(rel_path)})"
541
+ parse_error = string_value(pyproject.get("parse_error"))
542
+ if parse_error:
543
+ lines.append(
544
+ f"- pyproject.toml parse error{header_suffix}: {safe_inline_value(parse_error)}"
545
+ )
546
+ emitted_python_context = True
547
+ continue
548
+
549
+ requires_python = string_value(pyproject.get("requires_python"))
550
+ dependencies = string_list_value(pyproject.get("dependencies"))
551
+ if requires_python:
552
+ lines.append(f"- requires-python{header_suffix}: {safe_inline_value(requires_python)}")
553
+ emitted_python_context = True
554
+ if dependencies:
555
+ lines.append(f"### pyproject dependencies{header_suffix}")
556
+ lines.append(
557
+ format_manifest_items(
558
+ dependencies,
559
+ int(pyproject.get("dependencies_omitted") or 0),
560
+ limit=100,
561
+ )
562
+ )
563
+ emitted_python_context = True
564
+ if not requires_python and not dependencies:
565
+ lines.append(
566
+ f"- pyproject.toml detected{header_suffix}, but no "
567
+ "requires-python/dependencies found"
568
+ )
569
+ emitted_python_context = True
570
+ remaining_requirement_items = 100
571
+ omitted_requirement_groups = omitted_requirement_paths
572
+ for req_path, requirement_data in requirements:
573
+ parse_error = string_value(requirement_data.get("parse_error"))
574
+ if parse_error:
575
+ lines.append(f"### requirements-style dependencies ({safe_inline_path(req_path)})")
576
+ lines.append(f"- parse error: {safe_inline_value(parse_error)}")
577
+ else:
578
+ dependencies = string_list_value(requirement_data.get("dependencies"))
579
+ if remaining_requirement_items <= 0:
580
+ omitted_requirement_groups += 1
581
+ emitted_python_context = True
582
+ continue
583
+
584
+ lines.append(f"### requirements-style dependencies ({safe_inline_path(req_path)})")
585
+ visible_dependencies = dependencies[:remaining_requirement_items]
586
+ parser_omitted = int(requirement_data.get("dependencies_omitted") or 0)
587
+ budget_omitted = max(0, len(dependencies) - len(visible_dependencies))
588
+ remaining_requirement_items -= len(visible_dependencies)
589
+ lines.append(
590
+ format_manifest_items(
591
+ visible_dependencies,
592
+ parser_omitted + budget_omitted,
593
+ limit=len(visible_dependencies),
594
+ )
595
+ )
596
+ emitted_python_context = True
597
+ if omitted_requirement_groups:
598
+ lines.append(f"- ... and {omitted_requirement_groups} requirements file(s) omitted")
599
+ if pyproject_discovery and pyproject_discovery.omitted:
600
+ lines.append(f"- ... and {pyproject_discovery.omitted} pyproject manifest(s) omitted")
601
+ if not emitted_python_context:
602
+ lines.append("- none detected")
603
+ lines.append("")
604
+
605
+ lines.append("## Go context")
606
+ if go_mod:
607
+ emitted_go_context = False
608
+ parse_error = string_value(go_mod.get("parse_error"))
609
+ go_version = string_value(go_mod.get("go"))
610
+ toolchain = string_value(go_mod.get("toolchain"))
611
+ modules = string_list_value(go_mod.get("modules"))
612
+ if parse_error:
613
+ lines.append(f"- go.mod parse error: {safe_inline_value(parse_error)}")
614
+ emitted_go_context = True
615
+ if go_version:
616
+ lines.append(f"- go version: {safe_inline_value(go_version)}")
617
+ emitted_go_context = True
618
+ if toolchain:
619
+ lines.append(f"- toolchain: {safe_inline_value(toolchain)}")
620
+ emitted_go_context = True
621
+ if modules:
622
+ lines.append("### go.mod modules")
623
+ lines.append(
624
+ format_manifest_items(
625
+ modules,
626
+ int(go_mod.get("modules_omitted") or 0),
627
+ limit=100,
628
+ )
629
+ )
630
+ emitted_go_context = True
631
+ if not emitted_go_context:
632
+ lines.append("- go.mod detected, but no go/toolchain/require entries found")
633
+ else:
634
+ lines.append("- none detected")
635
+ lines.append("")
636
+
637
+ lines.append("## PHP context")
638
+ emitted_php_context = False
639
+ if composer_json:
640
+ parse_error = string_value(composer_json.get("parse_error"))
641
+ if parse_error:
642
+ lines.append(f"- composer.json parse error: {safe_inline_value(parse_error)}")
643
+ emitted_php_context = True
644
+ else:
645
+ emitted_composer_json_context = False
646
+ platform = string_list_value(composer_json.get("platform"))
647
+ require = string_list_value(composer_json.get("require"))
648
+ require_dev = string_list_value(composer_json.get("require_dev"))
649
+ if platform:
650
+ lines.append("### composer platform")
651
+ lines.append(
652
+ format_manifest_items(
653
+ platform,
654
+ int(composer_json.get("platform_omitted") or 0),
655
+ limit=50,
656
+ )
657
+ )
658
+ emitted_composer_json_context = True
659
+ if require:
660
+ lines.append("### composer require")
661
+ lines.append(
662
+ format_manifest_items(
663
+ require,
664
+ int(composer_json.get("require_omitted") or 0),
665
+ limit=100,
666
+ )
667
+ )
668
+ emitted_composer_json_context = True
669
+ if require_dev:
670
+ lines.append("### composer require-dev")
671
+ lines.append(
672
+ format_manifest_items(
673
+ require_dev,
674
+ int(composer_json.get("require_dev_omitted") or 0),
675
+ limit=100,
676
+ )
677
+ )
678
+ emitted_composer_json_context = True
679
+ if not emitted_composer_json_context:
680
+ lines.append(
681
+ "- composer.json detected, but no platform/require/require-dev entries found"
682
+ )
683
+ emitted_composer_json_context = True
684
+ emitted_php_context = emitted_php_context or emitted_composer_json_context
685
+ if composer_lock:
686
+ parse_error = string_value(composer_lock.get("parse_error"))
687
+ packages = string_list_value(composer_lock.get("packages"))
688
+ if parse_error:
689
+ lines.append(f"- composer.lock parse error: {safe_inline_value(parse_error)}")
690
+ emitted_php_context = True
691
+ elif packages:
692
+ lines.append("### composer.lock packages")
693
+ lines.append(
694
+ format_manifest_items(
695
+ packages,
696
+ int(composer_lock.get("packages_omitted") or 0),
697
+ limit=100,
698
+ )
699
+ )
700
+ emitted_php_context = True
701
+ else:
702
+ lines.append("- composer.lock detected, but no package entries found")
703
+ emitted_php_context = True
704
+ if not emitted_php_context:
705
+ lines.append("- none detected")
706
+ lines.append("")
707
+
708
+ lines.append("## JavaScript/TypeScript context")
709
+ if package_json_paths:
710
+ for package_json_path in package_json_paths:
711
+ package_json = parse_package_json(context_repo.ROOT / package_json_path)
712
+ lines.append(f"### {safe_inline_path(package_json_path)}")
713
+ parse_error = string_value(package_json.get("parse_error"))
714
+ if parse_error:
715
+ lines.append(f"- parse error: {safe_inline_value(parse_error)}")
716
+ continue
717
+
718
+ emitted_package_json_context = False
719
+ engines = string_list_value(package_json.get("engines"))
720
+ dependencies = string_list_value(package_json.get("dependencies"))
721
+ dev_dependencies = string_list_value(package_json.get("dev_dependencies"))
722
+ if engines:
723
+ lines.append("#### engines")
724
+ lines.append(
725
+ format_manifest_items(
726
+ engines,
727
+ int(package_json.get("engines_omitted") or 0),
728
+ limit=40,
729
+ )
730
+ )
731
+ emitted_package_json_context = True
732
+ if dependencies:
733
+ lines.append("#### dependencies")
734
+ lines.append(
735
+ format_manifest_items(
736
+ dependencies,
737
+ int(package_json.get("dependencies_omitted") or 0),
738
+ limit=100,
739
+ )
740
+ )
741
+ emitted_package_json_context = True
742
+ if dev_dependencies:
743
+ lines.append("#### devDependencies")
744
+ lines.append(
745
+ format_manifest_items(
746
+ dev_dependencies,
747
+ int(package_json.get("dev_dependencies_omitted") or 0),
748
+ limit=100,
749
+ )
750
+ )
751
+ emitted_package_json_context = True
752
+ if not emitted_package_json_context:
753
+ lines.append(
754
+ "- package.json detected, but no engines/dependencies/devDependencies found"
755
+ )
756
+ if package_json_discovery and package_json_discovery.omitted:
757
+ lines.append(
758
+ f"- package.json manifests omitted due to context limit: {package_json_discovery.omitted}"
759
+ )
760
+ else:
761
+ lines.append("- none detected")
762
+ lines.append("")
763
+
764
+ if app_versions:
765
+ lines.append("## Application and infrastructure version pins")
766
+ lines.append(
767
+ "Each item identifies its source path and a detected dependency, image, version or ref; URLs are sanitized."
768
+ )
769
+ lines.append(format_items(app_versions, limit=160))
770
+ lines.append("")
771
+ if manifest_paths:
772
+ add_section(
773
+ lines,
774
+ "Detected dependency/runtime manifest files",
775
+ manifest_paths,
776
+ limit=160,
777
+ )
778
+
779
+ context = "\n".join(lines).rstrip() + "\n"
780
+ max_bytes = getenv_int(
781
+ "OCR_BACKGROUND_MAX_BYTES",
782
+ DEFAULT_BACKGROUND_MAX_BYTES,
783
+ max_value=MAX_BACKGROUND_MAX_BYTES,
784
+ )
785
+ max_chars = getenv_int(
786
+ "OCR_BACKGROUND_MAX_CHARS",
787
+ DEFAULT_BACKGROUND_MAX_CHARS,
788
+ max_value=MAX_BACKGROUND_MAX_CHARS,
789
+ )
790
+ preamble, sections = split_markdown_sections(context)
791
+ ansible_titles = {
792
+ "Ansible core manifests",
793
+ "Role defaults and metadata files",
794
+ "Detected root Ansible playbook entrypoints",
795
+ "Ansible inventory topology files",
796
+ "Ansible inventory groups",
797
+ "Ansible Galaxy requirements",
798
+ "Detected ansible --version",
799
+ "Detected ansible-lint --version",
800
+ }
801
+ inactive_titles: set[str] = set()
802
+ if not ansible_active:
803
+ inactive_titles.update(ansible_titles)
804
+ if not python_active:
805
+ inactive_titles.add("Python context")
806
+ if not go_active:
807
+ inactive_titles.add("Go context")
808
+ if not php_active:
809
+ inactive_titles.add("PHP context")
810
+ if not javascript_active:
811
+ inactive_titles.add("JavaScript/TypeScript context")
812
+ if not versions_active:
813
+ inactive_titles.update(
814
+ {
815
+ "Application and infrastructure version pins",
816
+ "Detected dependency/runtime manifest files",
817
+ }
818
+ )
819
+
820
+ priorities = {
821
+ "Merge Request": 150,
822
+ "Review instructions": 150,
823
+ "Changed files by category": 140,
824
+ "Accepted project decisions": 130,
825
+ "Project instruction files": 120,
826
+ "Python context": 110,
827
+ "Go context": 110,
828
+ "PHP context": 110,
829
+ "JavaScript/TypeScript context": 110,
830
+ "Application and infrastructure version pins": 100,
831
+ "Detected dependency/runtime manifest files": 90,
832
+ }
833
+ selected = [
834
+ ContextSection(
835
+ title=section.title,
836
+ body=section.body,
837
+ priority=priorities.get(section.title, 60),
838
+ minimum_bytes=220 if section.title in priorities else 140,
839
+ )
840
+ for section in sections
841
+ if section.title not in inactive_titles
842
+ and section.body.strip()
843
+ and section.body.strip() != "- none detected"
844
+ ]
845
+ return render_context(preamble, selected, max_bytes, max_chars=max_chars)
846
+
847
+
848
+ def summarize_context(markdown: str, max_bytes: int, max_chars: int | None = None) -> str:
849
+ """Render a one-screen summary of the generated review background.
850
+
851
+ The summary is printed to stdout so it shows up in the GitLab job log
852
+ next to `Wrote ...`. It deliberately reports per-section *lines*, not
853
+ contents, so the log stays compact even on large repositories.
854
+ """
855
+
856
+ byte_len = len(markdown.encode("utf-8"))
857
+ char_len = len(markdown)
858
+ truncated = (
859
+ "- ... section truncated" in markdown
860
+ or "## Context coverage" in markdown
861
+ or bool(
862
+ re.search(
863
+ r"(?m)\n\n## Context truncation notice\n- Review background was truncated to \d+ bytes\.\n?$",
864
+ markdown,
865
+ )
866
+ )
867
+ )
868
+ byte_pct = (byte_len / max_bytes * 100.0) if max_bytes > 0 else 0.0
869
+ char_pct = (char_len / max_chars * 100.0) if max_chars else None
870
+ est_tokens = char_len // 4
871
+
872
+ section_headers: list[tuple[str, int, int]] = []
873
+ open_fence: str | None = None
874
+ offset = 0
875
+ for line in markdown.splitlines(keepends=True):
876
+ open_fence, _ = markdown_fence_transition(line.rstrip("\r\n"), open_fence)
877
+
878
+ if open_fence is None:
879
+ header_match = re.match(r"^##\s+([^#].*?)\s*$", line.rstrip("\n"))
880
+ if header_match:
881
+ section_headers.append((header_match.group(1).strip(), offset, offset + len(line)))
882
+
883
+ offset += len(line)
884
+
885
+ section_lines: list[tuple[str, int]] = []
886
+ for index, (title, header_start, body_start) in enumerate(section_headers):
887
+ # Skip the truncation notice — it is metadata about size, not a
888
+ # content block.
889
+ if title == "Context truncation notice":
890
+ continue
891
+ end = section_headers[index + 1][1] if index + 1 < len(section_headers) else len(markdown)
892
+ body = markdown[body_start:end]
893
+ # Count non-empty lines so a section padded with blanks does not
894
+ # look larger than it is.
895
+ line_count = sum(1 for line in body.splitlines() if line.strip())
896
+ section_lines.append((title, line_count))
897
+
898
+ lines = [
899
+ "Review background summary:",
900
+ (
901
+ f" size: {char_len} chars ({char_pct:.1f} % of {max_chars} limit) / "
902
+ f"{byte_len} bytes ({byte_pct:.1f} % of {max_bytes} limit)"
903
+ if char_pct is not None
904
+ else f" size: {byte_len} bytes ({byte_pct:.1f} % of {max_bytes} budget) / {char_len} chars"
905
+ ),
906
+ f" est. tokens: ~{est_tokens} (rough, chars/4)",
907
+ f" truncated: {'yes' if truncated else 'no'}",
908
+ f" sections ({len(section_lines)} total, lines):",
909
+ ]
910
+ for title, count in section_lines:
911
+ lines.append(f" - {title}: {count}")
912
+ return "\n".join(lines)
913
+
914
+
915
+ def main(argv: Sequence[str] | None = None) -> int:
916
+ """CLI entrypoint."""
917
+
918
+ parser = argparse.ArgumentParser(description=__doc__)
919
+ parser.add_argument(
920
+ "--output",
921
+ default=".review-context/dependencies.md",
922
+ help="Output markdown path.",
923
+ )
924
+ args = parser.parse_args(argv)
925
+
926
+ output = context_repo.resolve_output_path(args.output)
927
+ if output is None:
928
+ print(
929
+ f"Refusing to write review context outside repository or temp dir: {args.output}",
930
+ file=sys.stderr,
931
+ )
932
+ return 1
933
+
934
+ markdown = build_context()
935
+
936
+ try:
937
+ output.parent.mkdir(parents=True, exist_ok=True)
938
+ output.write_text(markdown, encoding="utf-8")
939
+ except OSError as exc:
940
+ print(f"Failed to write review context to {output}: {exc}", file=sys.stderr)
941
+ return 1
942
+
943
+ print(f"Wrote {output}")
944
+ max_bytes = getenv_int(
945
+ "OCR_BACKGROUND_MAX_BYTES",
946
+ DEFAULT_BACKGROUND_MAX_BYTES,
947
+ max_value=MAX_BACKGROUND_MAX_BYTES,
948
+ )
949
+ max_chars = getenv_int(
950
+ "OCR_BACKGROUND_MAX_CHARS",
951
+ DEFAULT_BACKGROUND_MAX_CHARS,
952
+ max_value=MAX_BACKGROUND_MAX_CHARS,
953
+ )
954
+ print(summarize_context(markdown, max_bytes, max_chars))
955
+ return 0