doubleblind-audit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """doubleblind - two layers that cannot see each other's mistakes."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ import sys
2
+ from .cli import main
3
+ sys.exit(main())
doubleblind/cli.py ADDED
@@ -0,0 +1,323 @@
1
+ """Command line: trace numbers, lint a review brief, build a packet."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import fnmatch
7
+ import os
8
+ import sys
9
+
10
+ from . import packet as _packet
11
+ from . import trace as _trace
12
+
13
+ ALLOW_DEFAULT = "doubleblind-allow.txt"
14
+
15
+ # Above this many evidence values, coincidental matches stop being negligible.
16
+ POOL_WARN = 2000
17
+
18
+
19
+ # --------------------------------------------------------------- allow files
20
+
21
+ def read_allow(path: str | None, document: str) -> dict[str, str]:
22
+ """Numbers exempt from tracing in ONE document, each with a stated reason.
23
+
24
+ Two rules, both of which exist because the obvious design fails:
25
+
26
+ * **A reason is mandatory.** An allow list without reasons becomes the place
27
+ inconvenient numbers go to stop being checked, and nobody reads it again.
28
+ * **Every entry names the document it applies to.** A repository-wide list
29
+ silently disarms the check everywhere: the entry that exempts an
30
+ illustrative figure in the README also exempted the fabricated figure in a
31
+ test fixture, and the test that was supposed to see the fixture fail
32
+ started passing. An exemption is about one claim in one document.
33
+
34
+ Format: ``<document glob> <number> <reason>``
35
+ """
36
+ path = path or ALLOW_DEFAULT
37
+ if not os.path.exists(path):
38
+ return {}
39
+ out: dict[str, str] = {}
40
+ doc = os.path.normpath(document)
41
+ base = os.path.basename(doc)
42
+ with open(path, encoding="utf-8") as fh:
43
+ lines = fh.readlines()
44
+ for i, line in enumerate(lines, 1):
45
+ line = line.split("#", 1)[0].rstrip()
46
+ if not line.strip():
47
+ continue
48
+ parts = line.split(None, 2)
49
+ if len(parts) < 3 or not parts[2].strip():
50
+ sys.exit(f"{path}:{i}: each line needs <document> <number> <reason> "
51
+ f"and the reason may not be empty -> {line.strip()!r}")
52
+ pattern, number, reason = parts[0], parts[1], parts[2].strip()
53
+ if fnmatch.fnmatch(doc, pattern) or fnmatch.fnmatch(base, pattern):
54
+ out[number] = reason
55
+ return out
56
+
57
+
58
+ # ----------------------------------------------------------------- reporting
59
+
60
+ def _c(s: str, code: str, on: bool) -> str:
61
+ return f"\033[{code}m{s}\033[0m" if on else s
62
+
63
+
64
+ def cmd_trace(args: argparse.Namespace) -> int:
65
+ colour = sys.stdout.isatty() and not args.no_color
66
+ with open(args.document, encoding="utf-8", errors="replace") as fh:
67
+ prose = fh.read()
68
+ evidence = _trace.load_evidence(args.data)
69
+ evidence += _trace.run_derivations(args.derive or [])
70
+ if not evidence:
71
+ print(f"no numbers found in {args.data} - nothing to trace against", file=sys.stderr)
72
+ return 2
73
+ allow = read_allow(args.allow, args.document)
74
+ findings = _trace.trace(prose, evidence, allow)
75
+
76
+ bad = [f for f in findings if f.verdict == "unsupported"]
77
+ ok = [f for f in findings if f.verdict == "traced"]
78
+ allowed = [f for f in findings if f.verdict == "allowed"]
79
+
80
+ for f in bad:
81
+ n = f.number
82
+ print(_c(f" UNSUPPORTED {n.raw}", "31;1", colour) +
83
+ f" {args.document}:{n.line_no}")
84
+ print(f" {n.line[:100]}")
85
+ if f.nearest:
86
+ near = ", ".join(f"{v.value:g} ({os.path.basename(v.source)} {v.where})"
87
+ for v in f.nearest)
88
+ print(f" nearest committed values: {near}")
89
+ if args.verbose:
90
+ for f in ok:
91
+ print(f" ok {f.number.raw:<12} <- {os.path.basename(f.matched.source)} "
92
+ f"{f.matched.where}{'' if f.scale == 'exact' else f.scale}")
93
+ for f in allowed:
94
+ print(f" allowed {f.number.raw:<12} {f.reason}")
95
+
96
+ vague = _trace.vague_quantities(prose)
97
+ if vague and not args.no_vague:
98
+ print()
99
+ for line_no, phrase, line in vague:
100
+ print(_c(f" UNCHECKABLE \"{phrase}\"", "33;1", colour) +
101
+ f" {args.document}:{line_no}")
102
+ print(f" {line[:100]}")
103
+ print(" A quantity written as words cannot be traced to a file. "
104
+ "Write the number.")
105
+
106
+ rescaled = sum(1 for f in ok if f.scale == " x100")
107
+ print()
108
+ print(f"{len(findings)} numbers in {args.document}; {len(ok)} traced"
109
+ + (f" ({rescaled} of them only after reading a stored fraction as a percentage)"
110
+ if rescaled else "")
111
+ + f", {len(allowed)} allowed, {len(bad)} unsupported.")
112
+ print(f"Evidence: {len(evidence)} values from "
113
+ f"{len(set(v.source for v in evidence))} sources.")
114
+ if len(evidence) > POOL_WARN:
115
+ print()
116
+ print(_c(f" NOTE {len(evidence)} values is a large pool.", "33;1", colour))
117
+ print(" Raw per-item dumps contain every id, index and token count, so a")
118
+ print(" round number in prose will match one of them by coincidence and be")
119
+ print(" reported as traced. Point --data at summary files and use --derive")
120
+ print(" for quantities a script recomputes, or this check gets weaker the")
121
+ print(" more data you give it.")
122
+
123
+ failed = bool(bad) or (bool(vague) and args.strict)
124
+ if not failed:
125
+ print()
126
+ print("Every number here exists in a committed file. That is all this says.")
127
+ print("It does not say the sentences around them are true: a correct number")
128
+ print("inside a claim that does not follow from it passes this check every")
129
+ print("time. Run the reviewer layer - `doubleblind review` - for that.")
130
+ return 1 if failed else 0
131
+
132
+
133
+ def cmd_lint(args: argparse.Namespace) -> int:
134
+ colour = sys.stdout.isatty() and not args.no_color
135
+ with open(args.brief, encoding="utf-8", errors="replace") as fh:
136
+ text = fh.read()
137
+ leaks = _packet.intent_leaks(text)
138
+ by_mech: dict[str, list] = {}
139
+ for l in leaks:
140
+ by_mech.setdefault(l.mechanism, []).append(l)
141
+ for mech in sorted(by_mech):
142
+ print(_c(f" {mech}", "33;1", colour))
143
+ for l in by_mech[mech]:
144
+ print(f" {args.brief}:{l.line_no} \"{l.phrase}\"")
145
+ print(f" {l.line[:100]}")
146
+ print(f" -> {by_mech[mech][0].fix}")
147
+ print()
148
+ if leaks:
149
+ print(f"{len(leaks)} phrase(s) in this brief tell the reviewer what to conclude.")
150
+ print("A reviewer that knows the wanted answer is not a second opinion.")
151
+ return 1
152
+ print("No intent leakage found. The brief asks a question that cannot be")
153
+ print("answered by agreeing.")
154
+ return 0
155
+
156
+
157
+ def cmd_packet(args: argparse.Namespace) -> int:
158
+ text = _packet.build_packet(args.document, args.data, )
159
+ if args.output == "-":
160
+ sys.stdout.write(text)
161
+ else:
162
+ with open(args.output, "w", encoding="utf-8") as fh:
163
+ fh.write(text)
164
+ # The full digest, not a prefix: this line is what a blindness record
165
+ # quotes, and a truncated hash is a guard against a slip rather than a
166
+ # commitment to what the reviewer was given.
167
+ print(f"packet: {args.output} ({len(text)} bytes)")
168
+ print(f"sha256 {_packet.packet_fingerprint(text)}")
169
+ return 0
170
+
171
+
172
+ ADAPTERS = {
173
+ "claude": (
174
+ "Claude Code",
175
+ "Agent(\n"
176
+ " subagent_type='general-purpose',\n"
177
+ " model='<a model that is NOT the one that wrote the artifact>',\n"
178
+ " run_in_background=False,\n"
179
+ " prompt=open('{packet}').read(),\n"
180
+ ")\n"
181
+ "# A sub-agent starts with no conversation history by construction, so\n"
182
+ "# zero context needs no extra work. Pick a different model explicitly:\n"
183
+ "# same model with a fresh context still carries the priors that wrote it.",
184
+ ),
185
+ "codex": (
186
+ "Codex CLI",
187
+ "codex exec --skip-git-repo-check < {packet}\n"
188
+ "# `exec` starts a fresh session. Do not use `codex resume`.",
189
+ ),
190
+ "deepseek": (
191
+ "DeepSeek",
192
+ "curl -s https://api.deepseek.com/chat/completions \\\n"
193
+ " -H \"Authorization: Bearer $DEEPSEEK_API_KEY\" -H 'Content-Type: application/json' \\\n"
194
+ " -d \"$(jq -Rs '{{model:\"deepseek-reasoner\",messages:[{{role:\"user\",content:.}}]}}' {packet})\"\n"
195
+ "# One request, one message. Sending history is what you are avoiding.",
196
+ ),
197
+ "kimi": (
198
+ "Kimi / Moonshot",
199
+ "curl -s https://api.moonshot.cn/v1/chat/completions \\\n"
200
+ " -H \"Authorization: Bearer $MOONSHOT_API_KEY\" -H 'Content-Type: application/json' \\\n"
201
+ " -d \"$(jq -Rs '{{model:\"kimi-k2-turbo-preview\",messages:[{{role:\"user\",content:.}}]}}' {packet})\"",
202
+ ),
203
+ "generic": (
204
+ "any OpenAI-compatible endpoint",
205
+ "curl -s \"$OPENAI_BASE_URL/chat/completions\" \\\n"
206
+ " -H \"Authorization: Bearer $OPENAI_API_KEY\" -H 'Content-Type: application/json' \\\n"
207
+ " -d \"$(jq -Rs '{{model:\"'$MODEL'\",messages:[{{role:\"user\",content:.}}]}}' {packet})\"",
208
+ ),
209
+ }
210
+
211
+
212
+ def cmd_review(args: argparse.Namespace) -> int:
213
+ text = _packet.build_packet(args.document, args.data)
214
+ out = args.output
215
+ with open(out, "w", encoding="utf-8") as fh:
216
+ fh.write(text)
217
+ fp = _packet.packet_fingerprint(text)
218
+ name, cmd = ADAPTERS[args.agent]
219
+ print(f"packet: {out} ({len(text)} bytes)")
220
+ print(f"sha256 {fp}")
221
+ print()
222
+ print(f"Send it with {name}:")
223
+ print()
224
+ for line in cmd.format(packet=out).splitlines():
225
+ print(f" {line}")
226
+ print()
227
+ print("Then, before you believe the verdict, record two things next to it:")
228
+ print(" 1. which model answered - it must not be the one that wrote the artifact;")
229
+ print(f" 2. this packet's sha256 - {fp}")
230
+ print("Without both, 'a different model checked it' is a claim about a")
231
+ print("conversation nobody can inspect.")
232
+ return 0
233
+
234
+
235
+ def cmd_render(args: argparse.Namespace) -> int:
236
+ """Run a figure script and audit what it leaves in `fig`."""
237
+ import runpy
238
+
239
+ from . import render as _render
240
+
241
+ ns = runpy.run_path(args.script)
242
+ fig = ns.get("fig")
243
+ if fig is None:
244
+ for v in ns.values():
245
+ if type(v).__name__ == "Figure":
246
+ fig = v
247
+ break
248
+ if fig is None:
249
+ print(f"{args.script} leaves no Figure at module level.\n"
250
+ " Assign the figure to `fig` so it can be audited after the script runs.",
251
+ file=sys.stderr)
252
+ return 2
253
+ problems = _render.audit(fig, scale=args.scale, min_ratio=args.min_ratio)
254
+ if not problems:
255
+ print()
256
+ print("Nothing in this figure is geometrically wrong at that scale. That is")
257
+ print("all this says: whether the shape a reader takes from it is the shape")
258
+ print("the data supports is not a thing any of these rules can see.")
259
+ return 1 if problems else 0
260
+
261
+
262
+ def main(argv: list[str] | None = None) -> int:
263
+ ap = argparse.ArgumentParser(
264
+ prog="doubleblind",
265
+ description="Two layers that cannot see each other's mistakes.")
266
+ ap.add_argument("--no-color", action="store_true")
267
+ sub = ap.add_subparsers(dest="cmd", required=True)
268
+
269
+ def add(name, help):
270
+ # --no-color is accepted on either side of the subcommand; people type
271
+ # it where it reads naturally, not where argparse prefers it.
272
+ sp = sub.add_parser(name, help=help)
273
+ sp.add_argument("--no-color", action="store_true")
274
+ return sp
275
+
276
+ t = add("trace", "every number in a document must exist in a committed file")
277
+ t.add_argument("document")
278
+ t.add_argument("--data", nargs="+", required=True, help="result files or directories")
279
+ t.add_argument("--allow", help=f"exemptions with reasons (default: {ALLOW_DEFAULT})")
280
+ t.add_argument("--derive", action="append", metavar="CMD",
281
+ help="a committed command whose printed numbers also count as "
282
+ "evidence; repeatable")
283
+ t.add_argument("--strict", action="store_true", help="fail on word-quantities too")
284
+ t.add_argument("--no-vague", action="store_true", help="do not report word-quantities")
285
+ t.add_argument("-v", "--verbose", action="store_true")
286
+ t.set_defaults(func=cmd_trace)
287
+
288
+ l = add("lint", "find phrases that tell the reviewer what to conclude")
289
+ l.add_argument("brief")
290
+ l.set_defaults(func=cmd_lint)
291
+
292
+ p = add("packet", "assemble artifact + evidence into one review packet")
293
+ p.add_argument("document", nargs="+")
294
+ p.add_argument("--data", nargs="*", default=[])
295
+ p.add_argument("-o", "--output", default="packet.md")
296
+ p.set_defaults(func=cmd_packet)
297
+
298
+ g = add("render", "audit a figure for what a reader sees and a number-checker cannot")
299
+ g.add_argument("script", help="a python script that leaves its figure in `fig`")
300
+ g.add_argument("--scale", type=float, default=0.30,
301
+ help="fraction of drawn size the figure will be seen at "
302
+ "(0.30 = a link unfurl, 1.0 = read at full size)")
303
+ g.add_argument("--min-ratio", type=float, default=4.5,
304
+ help="WCAG contrast required of body text")
305
+ g.set_defaults(func=cmd_render)
306
+
307
+ r = add("review", "build the packet and print how to send it")
308
+ r.add_argument("document", nargs="+")
309
+ r.add_argument("--data", nargs="*", default=[])
310
+ r.add_argument("--agent", choices=sorted(ADAPTERS), default="claude")
311
+ r.add_argument("-o", "--output", default="packet.md")
312
+ r.set_defaults(func=cmd_review)
313
+
314
+ args = ap.parse_args(argv)
315
+ for attr in ("no_color", "strict", "no_vague", "verbose", "allow", "derive",
316
+ "scale", "min_ratio"):
317
+ if not hasattr(args, attr):
318
+ setattr(args, attr, False if attr not in ("allow", "derive") else None)
319
+ return args.func(args)
320
+
321
+
322
+ if __name__ == "__main__":
323
+ sys.exit(main())
doubleblind/packet.py ADDED
@@ -0,0 +1,209 @@
1
+ """Assemble a review packet, and refuse to send one that gives the answer away.
2
+
3
+ A reviewer with no context is only useful while it stays without context. The
4
+ way that property is lost is almost never dramatic - nobody pastes the
5
+ conversation. It is lost in the request itself:
6
+
7
+ "Confirm that all four estimators fall below the card number."
8
+
9
+ There is exactly one way to answer that and still sound useful, and the model
10
+ takes it. The same request without its answer -
11
+
12
+ "For each numeric claim, state what these files actually support."
13
+
14
+ - cannot be answered by agreeing, because it does not contain anything to agree
15
+ with.
16
+
17
+ So this module does two things. It builds a packet out of the artifact and the
18
+ evidence it rests on, and it reads the instructions you were about to send and
19
+ tells you which phrases have the conclusion in them. The second is the part
20
+ that matters; the first is bookkeeping.
21
+
22
+ The patterns below are not a style guide. Each is a way a request stops being a
23
+ question, grouped by the mechanism rather than the wording, because the wording
24
+ changes and the mechanism does not.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import hashlib
30
+ import os
31
+ import re
32
+ from dataclasses import dataclass
33
+
34
+ __all__ = ["Leak", "intent_leaks", "build_packet", "packet_fingerprint", "REVIEWER_BRIEF"]
35
+
36
+
37
+ @dataclass
38
+ class Leak:
39
+ line_no: int
40
+ phrase: str
41
+ mechanism: str
42
+ line: str
43
+ fix: str
44
+
45
+
46
+ # (regex, mechanism, suggested rewrite)
47
+ _PATTERNS: list[tuple[str, str, str]] = [
48
+ # --- the request contains its own answer -------------------------------
49
+ (r"\b(?:please\s+)?(?:confirm|verify|validate|make sure|ensure)\s+(?:that|the|these|this|all|it)\b",
50
+ "answer-in-the-request",
51
+ "Ask what the evidence supports, not whether a stated thing is true: "
52
+ "'state what these files support' instead of 'confirm that X'."),
53
+ (r"(?:^|(?<=[.!?])\s+|\bplease\s+|\bcan you\s+|\bcould you\s+)check\s+that\b",
54
+ "answer-in-the-request",
55
+ "'Check that X' presumes X. 'Check whether X' does not - or better, ask "
56
+ "what the evidence gives."),
57
+ (r"\bdouble[- ]check\b",
58
+ "answer-in-the-request",
59
+ "'Double-check' presumes a first check that passed. Ask for a first check."),
60
+ (r"\b(?:is|are|does|do|was|were)\s+(?:this|that|it|the\s+\w+)\s+(?:correct|right|accurate|fine|ok|okay)\b\s*\?",
61
+ "answer-in-the-request",
62
+ "A yes/no question about correctness invites yes. Ask for the derivation instead."),
63
+
64
+ # --- the conclusion is stated before the review ------------------------
65
+ (r"\bwe\s+(?:show|find|demonstrate|prove|establish|conclude)\b",
66
+ "conclusion-stated",
67
+ "Remove the claim from the brief. The artifact can state it; the request must not."),
68
+ (r"\b(?:as|which is)\s+expected\b",
69
+ "conclusion-stated",
70
+ "Delete. It tells the reviewer which outcome is the acceptable one."),
71
+ (r"\bthis\s+(?:demonstrates|proves|confirms|establishes|shows)\b",
72
+ "conclusion-stated",
73
+ "Delete; let the reviewer decide what it demonstrates."),
74
+ (r"\bthe\s+(?:result|answer|number|conclusion)\s+should\s+be\b",
75
+ "conclusion-stated",
76
+ "Stating the expected value makes disagreement a contradiction of you."),
77
+
78
+ # --- polarity steering --------------------------------------------------
79
+ (r"\b(?:should|ought to|is supposed to|must)\s+(?:be|show|give|match|agree)\b",
80
+ "polarity-steering",
81
+ "Replace the expectation with the question: 'what do the files give?'"),
82
+ (r"\b(?:obviously|clearly|of course|naturally|needless to say)\b",
83
+ "polarity-steering",
84
+ "These words mark a claim as not-up-for-review. Remove them."),
85
+ (r"\b(?:minor|small|cosmetic|trivial)\s+(?:issue|problem|change|fix)\b",
86
+ "polarity-steering",
87
+ "Pre-sizing a problem tells the reviewer how seriously to take it."),
88
+
89
+ # --- social pressure ----------------------------------------------------
90
+ (r"\bI\s+(?:already\s+)?(?:checked|verified|confirmed|tested|reviewed)\b",
91
+ "social-pressure",
92
+ "Saying you checked makes a finding an accusation. Leave it out."),
93
+ (r"\bI'?m\s+(?:fairly\s+|pretty\s+|quite\s+)?(?:sure|confident|certain)\b",
94
+ "social-pressure",
95
+ "Stating your confidence sets the bar a finding has to clear."),
96
+ (r"\b(?:quick|brief|fast|light)\s+(?:sanity\s+)?(?:check|look|pass|review)\b",
97
+ "social-pressure",
98
+ "Asking for a quick look licenses a shallow one."),
99
+ (r"\b(?:just|only)\s+(?:need|want)\s+(?:you\s+)?to\b",
100
+ "social-pressure",
101
+ "Minimising the task lowers the effort it gets."),
102
+
103
+ # --- scope narrowing ----------------------------------------------------
104
+ (r"\b(?:ignore|skip|don'?t worry about|no need to (?:check|look at))\b",
105
+ "scope-narrowing",
106
+ "Exclusions hide exactly where the problem is. If something is out of "
107
+ "scope, remove it from the packet instead of naming it."),
108
+ (r"\b(?:only|just)\s+(?:look at|check|review|consider)\b",
109
+ "scope-narrowing",
110
+ "Narrowing the scope in the brief pre-decides where the error is not."),
111
+
112
+ # --- authorship leak ----------------------------------------------------
113
+ (r"\b(?:my|our)\s+(?:readme|paper|repo|repository|result|analysis|code|figure|draft)\b",
114
+ "authorship-leak",
115
+ "Ownership words make the reviewer a guest in your work. Refer to it as "
116
+ "'the attached document'."),
117
+ (r"\bI\s+(?:wrote|made|built|produced|ran)\b",
118
+ "authorship-leak",
119
+ "Same: the reviewer should not know who produced the artifact."),
120
+ ]
121
+
122
+ _COMPILED = [(re.compile(p, re.I), mech, fix) for p, mech, fix in _PATTERNS]
123
+
124
+
125
+ def intent_leaks(text: str) -> list[Leak]:
126
+ """Phrases in a review request that tell the reviewer what to conclude."""
127
+ out: list[Leak] = []
128
+ lines = text.splitlines()
129
+ for i, line in enumerate(lines, 1):
130
+ if line.lstrip().startswith(">"):
131
+ continue # quoted material is the artifact, not the request
132
+ for rx, mech, fix in _COMPILED:
133
+ for m in rx.finditer(line):
134
+ out.append(Leak(i, m.group(0).strip(), mech, line.strip(), fix))
135
+ return out
136
+
137
+
138
+ REVIEWER_BRIEF = """\
139
+ You are reviewing an attached document against the data files attached with it.
140
+
141
+ You have not seen this work before and you are not being told what it concluded.
142
+ Do not try to infer what answer is wanted; there is no wanted answer.
143
+
144
+ For every numeric claim and every causal or comparative claim in the document:
145
+
146
+ 1. Name the file and field that would settle it.
147
+ 2. State what that file actually contains.
148
+ 3. Say whether the sentence in the document follows from it - not whether the
149
+ number matches, but whether the sentence follows. A correct number inside a
150
+ sentence that does not follow from it is the most common defect here and the
151
+ one you are most needed for.
152
+
153
+ Report each finding as:
154
+
155
+ <file>:<line> CLAIM: "<quoted sentence>"
156
+ EVIDENCE: <what the data says>
157
+ COMMAND: <one shell command that settles it>
158
+
159
+ A finding with no command is an opinion. Drop it or turn it into one.
160
+
161
+ If a claim is supported, say so and move on. Do not pad the report, and do not
162
+ soften a finding because the rest of the document is careful.
163
+ """
164
+
165
+
166
+ def packet_fingerprint(text: str) -> str:
167
+ return hashlib.sha256(text.encode("utf-8")).hexdigest()
168
+
169
+
170
+ def build_packet(artifacts: list[str], evidence: list[str], brief: str | None = None,
171
+ max_bytes: int = 400_000) -> str:
172
+ """Return a single self-contained packet: brief, artifacts, evidence.
173
+
174
+ Nothing about how the artifact came to exist goes in - no commit messages,
175
+ no changelog, no summary of what was tried. Those are the parts that carry
176
+ intent, and they are the reason this function takes explicit paths rather
177
+ than a repository.
178
+ """
179
+ parts = [brief if brief is not None else REVIEWER_BRIEF, ""]
180
+
181
+ def _emit(path: str, kind: str) -> None:
182
+ try:
183
+ with open(path, encoding="utf-8", errors="replace") as fh:
184
+ raw = fh.read()
185
+ except OSError as exc:
186
+ parts.append(f"<<{kind} {path}: unreadable: {exc}>>")
187
+ return
188
+ if len(raw) > max_bytes:
189
+ raw = raw[:max_bytes] + f"\n<<truncated at {max_bytes} bytes>>\n"
190
+ parts.append(f"===== {kind}: {path} =====")
191
+ parts.append(raw.rstrip())
192
+ parts.append("")
193
+
194
+ files: list[str] = []
195
+ for p in evidence:
196
+ if os.path.isdir(p):
197
+ for root, _d, names in os.walk(p):
198
+ if any(part.startswith(".") and part not in (".", "..")
199
+ for part in root.split(os.sep)):
200
+ continue
201
+ files += [os.path.join(root, n) for n in sorted(names) if not n.startswith(".")]
202
+ else:
203
+ files.append(p)
204
+
205
+ for a in artifacts:
206
+ _emit(a, "DOCUMENT UNDER REVIEW")
207
+ for f in sorted(set(files)):
208
+ _emit(f, "DATA FILE")
209
+ return "\n".join(parts)