syncade 0.6.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. syncade/__init__.py +3 -0
  2. syncade/__main__.py +6 -0
  3. syncade/adapters/__init__.py +0 -0
  4. syncade/adapters/anthropic.py +457 -0
  5. syncade/adapters/base.py +221 -0
  6. syncade/adapters/fake.py +73 -0
  7. syncade/adapters/fake_common.py +29 -0
  8. syncade/adapters/fake_producer_audit_draft.py +460 -0
  9. syncade/adapters/fake_reviewer_synth.py +310 -0
  10. syncade/adapters/openai.py +484 -0
  11. syncade/adapters/openai_parsing.py +119 -0
  12. syncade/adapters/producer.py +221 -0
  13. syncade/adapters/producer_anthropic.py +300 -0
  14. syncade/adapters/producer_openai.py +226 -0
  15. syncade/adapters/registry.py +81 -0
  16. syncade/auth_check.py +554 -0
  17. syncade/auth_preflight.py +342 -0
  18. syncade/base_resolution.py +214 -0
  19. syncade/billing.py +141 -0
  20. syncade/checks_config.py +113 -0
  21. syncade/cli/__init__.py +546 -0
  22. syncade/cli/auth_gate.py +59 -0
  23. syncade/cli/config_keys.py +135 -0
  24. syncade/cli/config_list.py +82 -0
  25. syncade/cli/config_menu_rows.py +166 -0
  26. syncade/cli/config_mode.py +609 -0
  27. syncade/cli/config_overrides.py +122 -0
  28. syncade/cli/config_tui.py +476 -0
  29. syncade/cli/doctor_mode.py +72 -0
  30. syncade/cli/gc_mode.py +109 -0
  31. syncade/cli/install_skill.py +514 -0
  32. syncade/cli/metrics_mode.py +363 -0
  33. syncade/cli/modes.py +573 -0
  34. syncade/cli/parser.py +450 -0
  35. syncade/cli/parser_types.py +137 -0
  36. syncade/cli/paths.py +38 -0
  37. syncade/cli/preflight_paths.py +90 -0
  38. syncade/cli/resolve.py +116 -0
  39. syncade/cli/resume_mode.py +324 -0
  40. syncade/cli/toml_writer.py +410 -0
  41. syncade/cli/validate.py +421 -0
  42. syncade/config.py +478 -0
  43. syncade/config_auth.py +310 -0
  44. syncade/config_cold.py +209 -0
  45. syncade/config_gc.py +55 -0
  46. syncade/config_loader.py +182 -0
  47. syncade/config_loop.py +282 -0
  48. syncade/config_producer.py +222 -0
  49. syncade/config_retry.py +49 -0
  50. syncade/config_types.py +59 -0
  51. syncade/diff_filter.py +437 -0
  52. syncade/dispatcher.py +571 -0
  53. syncade/doctor.py +425 -0
  54. syncade/doctor_env.py +218 -0
  55. syncade/doctor_preview.py +524 -0
  56. syncade/doctor_types.py +28 -0
  57. syncade/exit_codes.py +82 -0
  58. syncade/findings.py +242 -0
  59. syncade/findings_json.py +456 -0
  60. syncade/gc.py +211 -0
  61. syncade/gc_execute.py +372 -0
  62. syncade/gc_protection.py +129 -0
  63. syncade/gc_types.py +50 -0
  64. syncade/gc_worktrees.py +200 -0
  65. syncade/git_object_id.py +12 -0
  66. syncade/git_preconditions.py +389 -0
  67. syncade/logging.py +289 -0
  68. syncade/metrics/__init__.py +32 -0
  69. syncade/metrics/aggregate.py +550 -0
  70. syncade/metrics/schema.py +221 -0
  71. syncade/orchestrator/__init__.py +61 -0
  72. syncade/orchestrator/_runs_dir.py +24 -0
  73. syncade/orchestrator/branch_advance.py +165 -0
  74. syncade/orchestrator/branch_guard.py +98 -0
  75. syncade/orchestrator/budget.py +107 -0
  76. syncade/orchestrator/escalation_coverage.py +81 -0
  77. syncade/orchestrator/loop.py +611 -0
  78. syncade/orchestrator/loop_dispatch_check.py +112 -0
  79. syncade/orchestrator/loop_finalize.py +404 -0
  80. syncade/orchestrator/loop_preflight.py +131 -0
  81. syncade/orchestrator/loop_resume.py +91 -0
  82. syncade/orchestrator/loop_rmtree.py +70 -0
  83. syncade/orchestrator/loop_round_step.py +599 -0
  84. syncade/orchestrator/prior_round.py +336 -0
  85. syncade/orchestrator/producer_phase.py +169 -0
  86. syncade/orchestrator/results.py +306 -0
  87. syncade/orchestrator/resume.py +96 -0
  88. syncade/orchestrator/resume_load.py +483 -0
  89. syncade/orchestrator/resume_plan.py +554 -0
  90. syncade/orchestrator/resume_target.py +215 -0
  91. syncade/orchestrator/resume_types.py +182 -0
  92. syncade/orchestrator/reviewer_template_failure.py +99 -0
  93. syncade/orchestrator/round.py +573 -0
  94. syncade/orchestrator/round_checks.py +91 -0
  95. syncade/orchestrator/round_no_changes.py +369 -0
  96. syncade/orchestrator/round_predispatch.py +212 -0
  97. syncade/orchestrator/verdict.py +279 -0
  98. syncade/persistence/__init__.py +189 -0
  99. syncade/persistence/_atomic.py +33 -0
  100. syncade/persistence/_clusters.py +70 -0
  101. syncade/persistence/_findings_verdict.py +201 -0
  102. syncade/persistence/_markdown.py +286 -0
  103. syncade/persistence/_validation.py +37 -0
  104. syncade/persistence/checks.py +249 -0
  105. syncade/persistence/decision_needed.py +289 -0
  106. syncade/persistence/findings_md.py +389 -0
  107. syncade/persistence/handoff.py +389 -0
  108. syncade/persistence/handoff_classify.py +196 -0
  109. syncade/persistence/last_reviewed.py +67 -0
  110. syncade/persistence/loop_manifest.py +165 -0
  111. syncade/persistence/loop_summary.py +352 -0
  112. syncade/persistence/loop_summary_text.py +428 -0
  113. syncade/persistence/producer.py +250 -0
  114. syncade/persistence/reviewer.py +198 -0
  115. syncade/persistence/round_manifest.py +238 -0
  116. syncade/persistence/run_init.py +153 -0
  117. syncade/persistence/run_summary.py +585 -0
  118. syncade/persistence/run_summary_next_steps.py +443 -0
  119. syncade/persistence/synth.py +242 -0
  120. syncade/persistence/test_run.py +152 -0
  121. syncade/presets.py +36 -0
  122. syncade/pricing_config.py +72 -0
  123. syncade/process.py +600 -0
  124. syncade/producer.py +189 -0
  125. syncade/producer_attempt.py +463 -0
  126. syncade/producer_escalation.py +146 -0
  127. syncade/producer_git.py +199 -0
  128. syncade/producer_result.py +205 -0
  129. syncade/prompts.py +448 -0
  130. syncade/prompts_loader.py +238 -0
  131. syncade/retry.py +159 -0
  132. syncade/run_inputs.py +40 -0
  133. syncade/run_status.py +198 -0
  134. syncade/selfcheck.py +471 -0
  135. syncade/skills/claude/README.md +221 -0
  136. syncade/skills/claude/SKILL.md +625 -0
  137. syncade/skills/codex/README.md +116 -0
  138. syncade/skills/codex/SKILL.md +574 -0
  139. syncade/snapshot.py +598 -0
  140. syncade/spec_audit.py +437 -0
  141. syncade/spec_audit_schema.py +190 -0
  142. syncade/spec_draft.py +423 -0
  143. syncade/spec_source.py +135 -0
  144. syncade/synthesis.py +428 -0
  145. syncade/synthesis_clusters.py +203 -0
  146. syncade/synthesis_repair.py +230 -0
  147. syncade/synthesis_schema.py +65 -0
  148. syncade/synthesizer/__init__.py +38 -0
  149. syncade/synthesizer/constants.py +33 -0
  150. syncade/synthesizer/driver.py +531 -0
  151. syncade/synthesizer/rendering.py +63 -0
  152. syncade/synthesizer/result.py +73 -0
  153. syncade/synthesizer/validation.py +421 -0
  154. syncade/synthesizer/workspace.py +208 -0
  155. syncade/templates/presets/balanced.toml +13 -0
  156. syncade/templates/presets/cheap.toml +12 -0
  157. syncade/templates/presets/thorough.toml +9 -0
  158. syncade/templates/producer.md +231 -0
  159. syncade/templates/reviewer.md +279 -0
  160. syncade/templates/reviewer_adversarial.md +164 -0
  161. syncade/templates/reviewer_codex.md +165 -0
  162. syncade/templates/spec_audit.md +168 -0
  163. syncade/templates/spec_draft.md +62 -0
  164. syncade/templates/synthesizer.md +204 -0
  165. syncade/test_runner.py +476 -0
  166. syncade/test_runner_classify.py +98 -0
  167. syncade/transcript.py +150 -0
  168. syncade/usage.py +407 -0
  169. syncade/worktree.py +497 -0
  170. syncade/worktree_env.py +133 -0
  171. syncade/worktree_paths.py +139 -0
  172. syncade-0.6.2.dist-info/METADATA +314 -0
  173. syncade-0.6.2.dist-info/RECORD +177 -0
  174. syncade-0.6.2.dist-info/WHEEL +5 -0
  175. syncade-0.6.2.dist-info/entry_points.txt +2 -0
  176. syncade-0.6.2.dist-info/licenses/LICENSE +202 -0
  177. syncade-0.6.2.dist-info/top_level.txt +1 -0
syncade/diff_filter.py ADDED
@@ -0,0 +1,437 @@
1
+ """Reviewer-facing diff filtering.
2
+
3
+ Reviewer worktrees have ``CLAUDE.md`` / ``AGENTS.md`` removed (the
4
+ architectural invariant that reviewers must not see project memory).
5
+ But ``git diff <base>..HEAD`` — captured in the operator's repo, where
6
+ those files still exist — contains hunks for them. A reviewer whose
7
+ worktree lacks ``CLAUDE.md`` but whose diff shows it changed will flag
8
+ a "tracked deletion" / missing-file finding. That false positive
9
+ persisted across all three rounds of the validation and drove the
10
+ loop to exit 20.
11
+
12
+ :func:`filter_diff_for_reviewer` removes those hunks at the orchestrator
13
+ boundary, before the diff reaches ``render_reviewer_prompt``. The
14
+ worktree strip is unchanged; the diff strip complements it so the file
15
+ is invisible across BOTH surfaces.
16
+
17
+ :data:`REVIEWER_STRIP_FILES` is the single source of truth for which
18
+ files are stripped: it feeds the default of
19
+ :attr:`syncade.config.ReviewConfig.strip_repo_context_files`, which the
20
+ orchestrator passes to BOTH the worktree-create call AND this filter, so a
21
+ customized ``strip_repo_context_files`` reaches both.
22
+
23
+ The two surfaces share the LIST but not the MATCHING, and that gap is real:
24
+ this filter compares BASENAMES, while ``worktree_paths._strip_files`` REFUSES
25
+ any entry containing ``/`` as a path-escape guard. So ``docs/CLAUDE.md``
26
+ strips the diff hunk here and leaves the file readable in the worktree — a
27
+ leak through the surface the other one covers. Bare basenames are the only
28
+ shape both handle identically. (An earlier version of this docstring claimed
29
+ the two "can never diverge"; that was false — corrected in PR-h-03.)
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import logging
35
+ import re
36
+ from collections.abc import Iterable
37
+
38
+ _log = logging.getLogger(__name__)
39
+
40
+ REVIEWER_STRIP_FILES: tuple[str, ...] = ("CLAUDE.md", "AGENTS.md")
41
+ """Files removed from reviewer worktrees AND from the reviewer-facing
42
+ diff. Single source of truth: the default of
43
+ :attr:`syncade.config.ReviewConfig.strip_repo_context_files` is
44
+ ``list(REVIEWER_STRIP_FILES)``, and the orchestrator passes that config
45
+ value to both the worktree strip and :func:`filter_diff_for_reviewer`."""
46
+
47
+ # The canonical per-file boundary in a unified diff: git emits a
48
+ # ``diff --git a/<path> b/<path>`` line for every file — including binary
49
+ # files and pure renames/deletions, which carry no ``@@`` chunk markers.
50
+ #
51
+ # Each side is matched INDEPENDENTLY as either a C-quoted or a bare path,
52
+ # because git quotes only the side that needs it — a rename of `plain.md` to
53
+ # `naïve.md` emits `diff --git a/plain.md "b/na\303\257ve.md"` (measured). A
54
+ # both-quoted-or-both-bare pair of patterns misses that header entirely, and a
55
+ # missed header is a repo-context file shown to a blind reviewer.
56
+ #
57
+ # The quoted alternative is tried FIRST so a genuine `"` in a bare path can
58
+ # never be mistaken for an opening quote. Greedy ``.+`` on a bare a-path
59
+ # backtracks to the last `` b/`` so `a/my notes.md b/my notes.md` (git does not
60
+ # quote spaces) still parses, exactly as before.
61
+ _QUOTED = r'"(?:[^"\\]|\\.)*"'
62
+ _DIFF_GIT_HEADER = re.compile(rf"^diff --git (?P<qa>{_QUOTED}|a/.+) (?P<qb>{_QUOTED}|b/.+)$")
63
+
64
+ # Git's C-style escapes. Anything else after a backslash is an octal byte.
65
+ _C_ESCAPES = {
66
+ "a": 0x07,
67
+ "b": 0x08,
68
+ "f": 0x0C,
69
+ "n": 0x0A,
70
+ "r": 0x0D,
71
+ "t": 0x09,
72
+ "v": 0x0B,
73
+ "\\": 0x5C,
74
+ '"': 0x22,
75
+ }
76
+
77
+
78
+ def _unquote_path(token: str) -> str:
79
+ """Decode one ``diff --git`` path token to its real filesystem name.
80
+
81
+ A bare token is returned as-is. A C-quoted token (git quotes a path
82
+ containing a non-ASCII byte, a quote, a backslash, or a control character —
83
+ NOT a plain space) is unescaped.
84
+
85
+ Octal escapes are BYTES, not code points: `naïve.md` is written
86
+ `na\\303\\257ve.md` because `ï` is UTF-8 `\\xc3\\xaf`. So escapes are
87
+ accumulated into a bytearray and decoded once at the end. Invalid UTF-8
88
+ raises ``UnicodeDecodeError`` (a ``ValueError`` subclass), which the caller
89
+ catches and maps to a fail-closed drop — an invalid-byte path is treated the
90
+ same as an unparseable header rather than silently mismatching any name.
91
+ """
92
+ if not token.startswith('"'):
93
+ return token
94
+ out = bytearray()
95
+ body = token[1:-1]
96
+ i = 0
97
+ while i < len(body):
98
+ ch = body[i]
99
+ if ch != "\\":
100
+ out.extend(ch.encode("utf-8"))
101
+ i += 1
102
+ continue
103
+ nxt = body[i + 1]
104
+ if nxt in _C_ESCAPES:
105
+ out.append(_C_ESCAPES[nxt])
106
+ i += 2
107
+ else:
108
+ # \nnn octal — git always emits exactly three [0-7] digits; fewer
109
+ # means a truncated/malformed header, which must fail closed.
110
+ digits = body[i + 1 : i + 4]
111
+ if len(digits) != 3 or not all(c in "01234567" for c in digits):
112
+ raise ValueError(f"malformed octal escape at position {i}")
113
+ out.append(int(digits, 8))
114
+ i += 4
115
+ return out.decode("utf-8")
116
+
117
+
118
+ def _split_sections(diff_text: str) -> list[list[str]]:
119
+ """Split a unified diff into per-file sections on ``diff --git`` boundaries.
120
+
121
+ Line endings are kept so retained sections reassemble byte-for-byte. Content before
122
+ the first ``diff --git`` (anomalous for git diff, but handled defensively) becomes a
123
+ leading section with no header, which callers always keep — it is the PREAMBLE, not a
124
+ file, and conflating the two would delete legitimate leading text.
125
+
126
+ Splits on ``\\n`` only, not Python's universal newlines. ``splitlines()`` treats ``\\r``
127
+ as a line separator, so a binary payload containing ``\\r`` immediately followed by
128
+ ``diff --git …`` would be split into a fake section boundary — either leaking binary
129
+ bytes that follow a syntactically valid fake header, or triggering a spurious
130
+ ``diff_malformed`` refusal when the fake header is malformed. Splitting on ``\\n`` alone
131
+ keeps the ``\\r`` inside its containing line.
132
+ """
133
+ sections: list[list[str]] = []
134
+ current: list[str] = []
135
+ raw_parts = diff_text.split("\n")
136
+ # Reconstruct lines with their \n endings; last fragment has none.
137
+ lines: list[str] = [p + "\n" for p in raw_parts[:-1]]
138
+ if raw_parts[-1]: # non-empty trailing fragment (diff not ending with \n)
139
+ lines.append(raw_parts[-1])
140
+ for line in lines:
141
+ if line.startswith("diff --git ") and current:
142
+ sections.append(current)
143
+ current = [line]
144
+ else:
145
+ current.append(line)
146
+ if current:
147
+ sections.append(current)
148
+ return sections
149
+
150
+
151
+ def elide_binary_hunks(diff_text: str) -> tuple[str, list[str]]:
152
+ """Replace binary file content with a one-line notice. Returns ``(diff, elided_paths)``.
153
+
154
+ ``snapshot`` renders the diff with ``--text`` — the only lever against attribute-driven
155
+ suppression, since a committed ``.gitattributes`` marked ``-diff`` otherwise collapses a
156
+ real source change to "Binary files ... differ". The cost is that a genuine binary is
157
+ emitted as raw text: measured on the reported repo, 12 committed PNG baselines turned
158
+ 65,961 B of real diff into 3,129,026 B, which no reviewer can read and which displaces
159
+ the diff it is meant to judge.
160
+
161
+ **Detection is by CONTENT, never by git's own report, and that is the whole design.**
162
+ ``git diff --numstat`` looks like the right oracle and is not: measured, it reports
163
+ ``-\t-`` for a plain text file an attacker marked ``*.py -diff``, exactly as it does for
164
+ a real PNG, and ``--text`` does not change its answer. Filtering on it would let a
165
+ committed ``.gitattributes`` erase source changes from the reviewer's diff — reopening
166
+ the hole ``--text`` exists to close.
167
+
168
+ So this applies git's OWN heuristic to the bytes instead: a NUL byte means binary.
169
+ That cannot be forged by an attributes file. U+FFFD replacement characters are
170
+ deliberately NOT a signal — a Latin-1-encoded source file is full of them, and dropping
171
+ real source from review is far worse than passing some noise through.
172
+
173
+ Both sides are covered by construction: a DELETED binary inflates identically (its
174
+ content is emitted in full as removed lines), so a PR that merely drops a vendored
175
+ asset is as exposed as one that adds it. Nothing here keys on added paths.
176
+
177
+ The header lines are preserved so the reviewer still sees that the path changed, and
178
+ the notice names the size withheld — the omission is disclosed, never silent.
179
+ """
180
+ if not diff_text or "\0" not in diff_text:
181
+ return diff_text, []
182
+
183
+ kept: list[str] = []
184
+ elided: list[str] = []
185
+ for section in _split_sections(diff_text):
186
+ body_start = _binary_body_start(section)
187
+ if body_start is None:
188
+ kept.extend(section)
189
+ continue
190
+ withheld = sum(len(line.encode("utf-8", errors="replace")) for line in section[body_start:])
191
+ paths, _ = _decode_header(section[0])
192
+ path = paths[1] if paths else "<unparseable path>"
193
+ elided.append(path)
194
+ kept.extend(section[:body_start])
195
+ kept.append(f"[syncade: binary file content omitted — {withheld} bytes not shown]\n")
196
+ # Total contract: no NUL leaves this function. Section-based elision covers every
197
+ # `diff --git` file, but a headerless PREAMBLE is kept by design (it is not a file),
198
+ # so a stray NUL there would otherwise reach a provider that has no reason to accept
199
+ # one. Sectioned binary content is already gone; this only sweeps the remainder.
200
+ return "".join(kept).replace("\x00", ""), elided
201
+
202
+
203
+ def _binary_body_start(section: list[str]) -> int | None:
204
+ """Index of the first content line of a binary section, or ``None`` if it is not binary.
205
+
206
+ Only a section with a ``diff --git`` header can be elided; a headerless leading section
207
+ is the preamble. The split point is the first hunk header (``@@``) when there is one, so
208
+ the file-mode/index metadata a reviewer may want survives; otherwise it is the line after
209
+ the header, which covers git's own ``Binary files ... differ`` form.
210
+ """
211
+ if not section or not section[0].startswith("diff --git "):
212
+ return None
213
+ if not any("\0" in line for line in section):
214
+ return None
215
+ for i, line in enumerate(section):
216
+ if line.startswith("@@"):
217
+ return i + 1
218
+ return 1
219
+
220
+
221
+ def filter_diff_for_reviewer(diff_text: str, strip_files: Iterable[str]) -> str:
222
+ """Remove the diff hunks of ``strip_files`` from a unified diff.
223
+
224
+ Used by the orchestrator before calling ``render_reviewer_prompt`` so
225
+ reviewers don't see edits to files that are stripped from their
226
+ worktrees. ``snapshot.diff_text`` itself is never modified — callers
227
+ rebind the filtered result locally.
228
+
229
+ Matching is by basename on EITHER side of the ``diff --git`` header,
230
+ so ``CLAUDE.md``, ``docs/CLAUDE.md``, and a tracked deletion of
231
+ ``CLAUDE.md`` (which git still writes as ``a/CLAUDE.md b/CLAUDE.md``)
232
+ are all recognized. A ``diff --git`` line whose identity cannot be
233
+ established (unparseable header or malformed path encoding) is
234
+ DROPPED — fail-closed, with a WARNING logged. Over-stripping costs a
235
+ reviewer some context and is visible to them; under-stripping is an
236
+ invisible leak of the property the review is sold on.
237
+
238
+ Edge cases (per the design):
239
+
240
+ - empty ``diff_text`` → ``""``
241
+ - no hunk matches → original text returned byte-for-byte
242
+ - every hunk matches → ``""``
243
+
244
+ Args:
245
+ diff_text: A unified diff string (``git diff <base>..HEAD``).
246
+ strip_files: File names whose hunks to remove. Entries are
247
+ compared by basename, so bare ``"CLAUDE.md"`` matches a
248
+ nested ``docs/CLAUDE.md`` hunk.
249
+
250
+ Returns:
251
+ The diff with the matching files' hunks removed; retained hunks
252
+ are preserved byte-for-byte and in their original order.
253
+ """
254
+ if not diff_text:
255
+ return diff_text
256
+
257
+ strip_basenames = {name.rsplit("/", 1)[-1] for name in strip_files}
258
+ if not strip_basenames:
259
+ return diff_text
260
+
261
+ kept: list[str] = []
262
+ for section in _split_sections(diff_text):
263
+ redacted = _redacted_boundary_rename(section[0], strip_basenames)
264
+ if redacted is not None:
265
+ kept.append(redacted)
266
+ continue
267
+ if _section_targets_stripped_file(section[0], strip_basenames):
268
+ continue
269
+ kept.extend(section)
270
+ return "".join(kept)
271
+
272
+
273
+ def _redacted_boundary_rename(header_line: str, strip_basenames: set[str]) -> str | None:
274
+ """A placeholder section for a rename OUT of a strip target, or ``None``.
275
+
276
+ ``git mv CLAUDE.md app.py`` moves content from a path reviewers must not see to one
277
+ they must. Dropping the section conceals that ``app.py`` appeared (PR-h-02c.5);
278
+ keeping it leaks project memory three ways -- the a-path in the header, the
279
+ ``rename from`` line, and, on a rename WITH edits, the stripped file's own lines as
280
+ hunk CONTEXT. Measured: a one-line edit yields ``@@ ... @@ memory line`` plus
281
+ ``memory line`` context rows.
282
+
283
+ So the body is not redacted, it is DISCARDED, and the destination is announced
284
+ instead. The file is present in the reviewer's worktree (the strip matches basenames,
285
+ and the destination is not one), so pointing at it costs the reviewer nothing and
286
+ leaks nothing. That reviewers cannot see repo-context files is already stated in
287
+ their prompt, so naming the fact of a rename discloses nothing new -- only the source
288
+ NAME and CONTENT must not survive, and neither does.
289
+ """
290
+ paths, _ = _decode_header(header_line.rstrip("\n"))
291
+ if paths is None:
292
+ return None
293
+ a_path, b_path = paths
294
+ a_base = a_path.rsplit("/", 1)[-1]
295
+ b_base = b_path.rsplit("/", 1)[-1]
296
+ if a_base not in strip_basenames or b_base in strip_basenames:
297
+ return None
298
+ return (
299
+ f"diff --git a/(stripped repo-context file) b/{b_path}\n"
300
+ f"rename to {b_path}\n"
301
+ f"(body withheld: renamed out of a stripped repo-context file. "
302
+ f"Review {b_path} directly in your worktree.)\n"
303
+ )
304
+
305
+
306
+ def _decode_header(header_line: str) -> tuple[tuple[str, str] | None, str]:
307
+ """``((a_path, b_path), "")``, or ``(None, reason)`` when the header's
308
+ identity cannot be established.
309
+
310
+ Returns FULL paths with the ``a/``/``b/`` prefix stripped, so a rename
311
+ ``CLAUDE.md -> src/app.py`` yields ``("CLAUDE.md", "src/app.py")``.
312
+ Callers that need basenames compute ``path.rsplit("/", 1)[-1]`` themselves.
313
+
314
+ This is the SINGLE definition of "identifiable", shared by the filter (which DROPS
315
+ such a section) and by :func:`unidentifiable_sections` (which reports it). Two
316
+ separate notions would eventually disagree, and the disagreement would be a leak on
317
+ one side and a spurious refusal on the other.
318
+
319
+ ``reason`` exists so the warning can say *why* rather than only *that* — the three
320
+ causes are diagnosed differently by an operator.
321
+ """
322
+ match = _DIFF_GIT_HEADER.match(header_line.rstrip("\n"))
323
+ if match is None:
324
+ return None, "unidentifiable header"
325
+ try:
326
+ a = _unquote_path(match.group("qa"))
327
+ b = _unquote_path(match.group("qb"))
328
+ except ValueError:
329
+ # Git's own quoting never emits a malformed escape or invalid UTF-8, but a
330
+ # TRUNCATED diff can (increment E caps diff size).
331
+ return None, "malformed path encoding"
332
+ # Git's trusted prefixes. A bare token gets these from the regex; a quoted token is
333
+ # any syntactically valid quoted string, so `"x/app.py"` must be caught here.
334
+ if not a.startswith("a/"):
335
+ return None, "untrusted a-side prefix"
336
+ if not b.startswith("b/"):
337
+ return None, "untrusted b-side prefix"
338
+ # A bare a-side containing " b/" means the greedy regex consumed part of the b-side
339
+ # path — the header has multiple valid " b/" split points and the a/b boundary is
340
+ # indeterminate. Example: `diff --git a/CLAUDE.md b/foo b/bar.py` is produced by
341
+ # `git mv CLAUDE.md 'foo b/bar.py'`; the regex yields a='CLAUDE.md b/foo' / b='bar.py'
342
+ # instead of a='CLAUDE.md' / b='foo b/bar.py', so the strip check uses the wrong
343
+ # basename and the source content leaks. Fail closed: treat as unidentifiable.
344
+ qa_raw = match.group("qa")
345
+ if not qa_raw.startswith('"') and " b/" in qa_raw:
346
+ return None, "ambiguous bare header (multiple ` b/` separators)"
347
+ return (a[2:], b[2:]), ""
348
+
349
+
350
+ def unidentifiable_sections(diff_text: str) -> list[str]:
351
+ """The ``diff --git`` header lines whose identity cannot be established.
352
+
353
+ These are exactly the sections :func:`filter_diff_for_reviewer` drops fail-closed, so
354
+ a caller can tell "the diff filtered to nothing because there was nothing to review"
355
+ from "…because we could not read it" — the difference between an honest no-change
356
+ result and shipping a change BECAUSE it was unreadable (PR-h-02d D2).
357
+
358
+ Returns the header lines themselves so the refusal can name what it could not read.
359
+ """
360
+ # split('\n') rather than splitlines() — see _split_sections for why.
361
+ return [
362
+ line
363
+ for line in diff_text.split("\n")
364
+ if line.startswith("diff --git ") and _decode_header(line)[0] is None
365
+ ]
366
+
367
+
368
+ def concealed_destinations(diff_text: str, strip_files: Iterable[str]) -> list[str]:
369
+ """Destination paths that filtering hid even though a reviewer should see them.
370
+
371
+ A section is dropped when EITHER side's basename is a strip target. That is right when
372
+ the DESTINATION is a strip target — the content lands somewhere a reviewer must not
373
+ look. It is wrong when only the SOURCE was: ``git mv CLAUDE.md app.py`` drops the
374
+ section, so the reviewer is never told ``app.py`` came into existence.
375
+
376
+ Alone that is a blind spot (the file is still in the reviewer's worktree, and a rename
377
+ with a large rewrite falls below git's similarity threshold and is emitted as
378
+ delete+add, which survives). It becomes a FALSE SHIP in composition with PR-h-02d: if
379
+ such a rename is the only change, the filtered diff is empty, the run terminates
380
+ ``no_changes_to_review`` at exit 0, and zero reviewers are dispatched — so nothing
381
+ reads that worktree, because nothing runs.
382
+
383
+ Callers use a non-empty result to refuse the known-empty conclusion. Note this is
384
+ deliberately NOT "the unfiltered diff was non-empty": that would revert PR-h-02d's D3,
385
+ where a legitimately all-repo-context change SHOULD be known-empty. The question is
386
+ whether a drop was justified by its destination, not whether anything was dropped.
387
+ """
388
+ strip_basenames = {name.rsplit("/", 1)[-1] for name in strip_files}
389
+ if not strip_basenames:
390
+ return []
391
+ concealed: list[str] = []
392
+ for line in diff_text.split("\n"): # split('\n') — see _split_sections for why
393
+ if not line.startswith("diff --git "):
394
+ continue
395
+ paths, _ = _decode_header(line)
396
+ if paths is None:
397
+ continue # unidentifiable — a separate refusal path
398
+ a_path, b_path = paths
399
+ a_base = a_path.rsplit("/", 1)[-1]
400
+ b_base = b_path.rsplit("/", 1)[-1]
401
+ if a_base in strip_basenames and b_base not in strip_basenames:
402
+ concealed.append(b_path)
403
+ return concealed
404
+
405
+
406
+ def _section_targets_stripped_file(header_line: str, strip_basenames: set[str]) -> bool:
407
+ """True iff this section must be dropped from the reviewer's diff.
408
+
409
+ Two distinct "we cannot read this" cases, which must NOT be conflated:
410
+
411
+ - The section does not begin with ``diff --git`` at all. That is the
412
+ defensive preamble section, not a file — KEEP it, or filtering would
413
+ delete legitimate leading text.
414
+ - The section IS a ``diff --git`` header but cannot be parsed or decoded.
415
+ That is a real file section whose identity is unknown, so it could be a
416
+ strip target — DROP it (D1(a), PR-h-02c).
417
+
418
+ The second rule REVERSES a previous deliberate choice to keep unparseable
419
+ headers, justified then by "under-stripping leaves the reviewer-template
420
+ 'Stripped files' note as a backstop, whereas over-stripping could drop real
421
+ code from review". That trade is wrong for this failure: over-stripping
422
+ costs a reviewer some context and is visible to them, while under-stripping
423
+ is an INVISIBLE leak of the blindness property the review is sold on. Note
424
+ the caller returns early when no strip targets are configured, so this can
425
+ never drop anything in a run that does no stripping.
426
+ """
427
+ line = header_line.rstrip("\n")
428
+ if not line.startswith("diff --git "):
429
+ return False
430
+ paths, reason = _decode_header(line)
431
+ if paths is None:
432
+ _log.warning("diff_filter: dropping section (%s, fail-closed): %r", reason, line)
433
+ return True
434
+ a_path, b_path = paths
435
+ return (
436
+ a_path.rsplit("/", 1)[-1] in strip_basenames or b_path.rsplit("/", 1)[-1] in strip_basenames
437
+ )