pi-crew 0.9.47 → 0.9.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +18 -0
- package/CHANGELOG.md +132 -4
- package/README.md +1 -1
- package/dist/build-meta.json +22 -12
- package/dist/index.mjs +430 -388
- package/dist/index.mjs.map +3 -3
- package/docs/decisions/2026-07-24-oidc-trusted-publishing.md +112 -0
- package/docs/publishing.md +5 -2
- package/package.json +2 -2
- package/scripts/postinstall.mjs +43 -18
- package/skills/.gitkeep +0 -0
- package/skills/distill-persona/BUILD-NOTES.md +55 -0
- package/skills/distill-persona/SKILL.md +612 -0
- package/skills/distill-persona/UPGRADE-LOG-RESEARCH-SKILLS.md +100 -0
- package/skills/distill-persona/references/coverage-manifest.md +65 -0
- package/skills/distill-persona/references/distillation-field-synthesis-pass2.md +59 -0
- package/skills/distill-persona/references/distillation-field-synthesis.md +108 -0
- package/skills/distill-persona/references/handoff.md +42 -0
- package/skills/distill-persona/references/research/lesson-memory-shortcut.md +33 -0
- package/skills/distill-persona/references/research/r1-a-examples.md +23 -0
- package/skills/distill-persona/references/research/r1-b-scripts.md +26 -0
- package/skills/distill-persona/references/research/r1-c-human-readme.md +31 -0
- package/skills/distill-persona/references/research/r1-d-tests.md +28 -0
- package/skills/distill-persona/references/research/r1-verification.md +36 -0
- package/skills/distill-persona/references/research/r2-low-yield.md +26 -0
- package/skills/distill-persona/scripts/fidelity_eval.py +244 -0
- package/skills/distill-persona/scripts/validate-skill-structure.mjs +177 -0
- package/skills/distill-software/BUILD-NOTES.md +56 -0
- package/skills/distill-software/SKILL.md +302 -0
- package/skills/distill-software/references/handoff.md +47 -0
- package/skills/distill-software/scripts/code_dna.py +290 -0
- package/skills/research/DISTILLATION-PROCESS-CHECKLIST.md +120 -0
- package/skills/research/EXCAVATION-CHECKLIST.md +142 -0
- package/skills/research/FIDELITY.md +180 -0
- package/skills/research/SKILL.md +432 -0
- package/skills/research/references/anti-patterns.md +184 -0
- package/skills/research/references/fidelity.md +241 -0
- package/skills/research/references/handoff.md +48 -0
- package/skills/research/references/research-protocol.md +162 -0
- package/skills/research/references/source-inventory.md +135 -0
- package/skills/research/references/verified-models.md +163 -0
- package/skills/research/scripts/__pycache__/safe_io.cpython-312.pyc +0 -0
- package/skills/research/scripts/code_dna.py +233 -0
- package/skills/research/scripts/emit_run_summary.py +142 -0
- package/skills/research/scripts/safe_io.py +314 -0
- package/skills/research/scripts/source_evaluator.py +234 -0
- package/skills/research/scripts/validate-skill-structure.mjs +177 -0
- package/skills/research/scripts/verify_citations.py +225 -0
- package/skills/security-priority.json +28 -0
- package/src/config/config.ts +1 -0
- package/src/config/role-tools.ts +6 -3
- package/src/config/types.ts +8 -0
- package/src/runtime/background-runner.ts +11 -16
- package/src/runtime/heartbeat-watcher.ts +28 -1
- package/src/runtime/task-runner.ts +165 -119
- package/src/schema/config-schema.ts +1 -0
- package/src/utils/gh-protocol.ts +9 -8
- package/workflows/distill.workflow.md +198 -0
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
safe_io.py — SSRF-safe fetch guard + secret/PII redaction helper for the
|
|
4
|
+
research/distill operational scripts (MEDIUM-3 + MEDIUM-4 hardening).
|
|
5
|
+
|
|
6
|
+
Two functions:
|
|
7
|
+
- is_safe_url(url): SSRF guard. Returns True ONLY for http(s) URLs whose host
|
|
8
|
+
is not a private/loopback/link-local/metadata/reserved address. Rejects
|
|
9
|
+
non-http(s) schemes (file://, gopher://, dict://, ...). Optionally resolves
|
|
10
|
+
DNS and rejects if any A/AAAA record lands in a blocked range (DNS-rebind).
|
|
11
|
+
- redact_secrets(text): regex-based secret/PII redaction. Masks the VALUE of
|
|
12
|
+
private-key blocks, Authorization/Bearer headers, AWS keys, named
|
|
13
|
+
VAR=secret assignments, and common service tokens (sk_/pk_/xox_/gh*_),
|
|
14
|
+
keeping the finding TYPE + location intact
|
|
15
|
+
(e.g. "API_KEY=sk-abc123" -> "API_KEY=***REDACTED***").
|
|
16
|
+
|
|
17
|
+
Stdlib only. Python 3.9+.
|
|
18
|
+
|
|
19
|
+
Usage:
|
|
20
|
+
from safe_io import is_safe_url, redact_secrets
|
|
21
|
+
python3 safe_io.py --self-test # exits 0 on success
|
|
22
|
+
|
|
23
|
+
Imported by scripts that persist source content (redact_secrets) or fetch URLs
|
|
24
|
+
(is_safe_url). Referenced from the 3 SKILL.md files so agents invoke it before
|
|
25
|
+
persisting fetched content or fetching live URLs.
|
|
26
|
+
"""
|
|
27
|
+
import ipaddress
|
|
28
|
+
import re
|
|
29
|
+
import socket
|
|
30
|
+
import sys
|
|
31
|
+
from urllib.parse import urlparse
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
REDACTED = "***REDACTED***"
|
|
35
|
+
|
|
36
|
+
# Hostnames that ALWAYS resolve to loopback / cloud-metadata and must be
|
|
37
|
+
# rejected even when resolve_dns=False (no network call). SSRF defense for the
|
|
38
|
+
# common bypass: is_safe_url("http://localhost") would otherwise pass because
|
|
39
|
+
# the hostname-only path skips DNS.
|
|
40
|
+
_BLOCKED_HOSTNAMES = frozenset({
|
|
41
|
+
"localhost",
|
|
42
|
+
"localhost.localdomain",
|
|
43
|
+
"ip6-localhost",
|
|
44
|
+
"ip6-loopback",
|
|
45
|
+
"metadata", # cloud metadata shorthand
|
|
46
|
+
"metadata.google.internal",
|
|
47
|
+
"metadata.aws.internal",
|
|
48
|
+
"metadata.azure.com",
|
|
49
|
+
})
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# --------------------------------------------------------------------------- #
|
|
53
|
+
# redact_secrets — regex rules. Each preserves the type/key and blanks only
|
|
54
|
+
# the secret VALUE. Order matters: blocks first, then headers, then inline.
|
|
55
|
+
# --------------------------------------------------------------------------- #
|
|
56
|
+
_SECRET_RULES = [
|
|
57
|
+
# 1. PEM/DER private-key blocks (keep the BEGIN/END markers, blank the body)
|
|
58
|
+
(
|
|
59
|
+
"private_key_block",
|
|
60
|
+
re.compile(
|
|
61
|
+
r"(-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP |ENCRYPTED )?PRIVATE KEY-----)"
|
|
62
|
+
r"[\s\S]*?"
|
|
63
|
+
r"(-----END (?:RSA |EC |DSA |OPENSSH |PGP |ENCRYPTED )?PRIVATE KEY-----)",
|
|
64
|
+
re.MULTILINE,
|
|
65
|
+
),
|
|
66
|
+
lambda m: m.group(1) + "\n" + REDACTED + "\n-----END PRIVATE KEY-----",
|
|
67
|
+
),
|
|
68
|
+
# 1b. Unterminated private-key BEGIN (leaked partial key — no END marker).
|
|
69
|
+
# Rule 1 above already handled any complete BEGIN/END block, so this only
|
|
70
|
+
# fires on partials. Matches BEGIN + base64-ish content to end of line.
|
|
71
|
+
(
|
|
72
|
+
"private_key_begin_only",
|
|
73
|
+
re.compile(
|
|
74
|
+
r"(-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP |ENCRYPTED )?PRIVATE KEY-----)"
|
|
75
|
+
r"([A-Za-z0-9+/=\s]+)$"
|
|
76
|
+
),
|
|
77
|
+
lambda m: m.group(1) + "\n" + REDACTED,
|
|
78
|
+
),
|
|
79
|
+
# 2. Authorization / Proxy-Authorization / X-Authorization header lines
|
|
80
|
+
(
|
|
81
|
+
"auth_header",
|
|
82
|
+
re.compile(
|
|
83
|
+
r"(?im)^(\s*(?:authorization|proxy-authorization|x-authorization)\s*:\s*"
|
|
84
|
+
r"(?:bearer|basic|token|jwt)?\s*)([^\r\n]+)"
|
|
85
|
+
),
|
|
86
|
+
lambda m: m.group(1) + REDACTED,
|
|
87
|
+
),
|
|
88
|
+
# 3. Inline "Bearer <token>" / "Basic <token>" anywhere
|
|
89
|
+
(
|
|
90
|
+
"bearer_inline",
|
|
91
|
+
re.compile(r"(?i)\b(bearer|basic)\s+([A-Za-z0-9\-._~+/]+=*)"),
|
|
92
|
+
lambda m: m.group(1) + " " + REDACTED,
|
|
93
|
+
),
|
|
94
|
+
# 4. AWS access-key id (AKIA...) — the id itself is a credential
|
|
95
|
+
(
|
|
96
|
+
"aws_key_id",
|
|
97
|
+
re.compile(r"\b(AKIA[0-9A-Z]{16})\b"),
|
|
98
|
+
lambda m: REDACTED,
|
|
99
|
+
),
|
|
100
|
+
# 4b. High-confidence credential names with SHORT values (password/passwd/pwd)
|
|
101
|
+
# — rule 5 below requires value length >=8, which misses weak short
|
|
102
|
+
# passwords. These names are almost always secrets, so redact any value.
|
|
103
|
+
(
|
|
104
|
+
"short_credential",
|
|
105
|
+
re.compile(r"(?i)\b(password|passwd|pwd)\s*[:=]\s*([^\s\"'#&|]+)"),
|
|
106
|
+
lambda m: m.group(1) + "=" + REDACTED,
|
|
107
|
+
),
|
|
108
|
+
# 5. .env-style / JSON named assignments with a secret-like name:
|
|
109
|
+
# keeps NAME= / "name": and blanks the value
|
|
110
|
+
(
|
|
111
|
+
"named_assignment",
|
|
112
|
+
re.compile(
|
|
113
|
+
r"(?i)([\"']?[A-Za-z0-9_\-\.]*(?:"
|
|
114
|
+
r"pass(?:word)?|secret|api[_-]?key|token|access[_-]?key|private[_-]?key|"
|
|
115
|
+
r"client[_-]?secret|auth(?:orization)?|credential|refresh[_-]?token|"
|
|
116
|
+
r"jwt|passwd"
|
|
117
|
+
r")[A-Za-z0-9_\-\.]*[\"']?\s*[:=]\s*)([\"']?)([^\"'\r\n#&|]{8,})"
|
|
118
|
+
),
|
|
119
|
+
lambda m: m.group(1) + m.group(2) + REDACTED + m.group(2),
|
|
120
|
+
),
|
|
121
|
+
# 6. Common service tokens standing alone (Stripe/OpenAI sk_/pk_, Slack xox, GitHub gh*)
|
|
122
|
+
(
|
|
123
|
+
"service_token",
|
|
124
|
+
re.compile(r"\b(sk|pk|xox[abp]|gh[opsu]|github_pat)_[A-Za-z0-9]{16,}\b"),
|
|
125
|
+
lambda m: REDACTED,
|
|
126
|
+
),
|
|
127
|
+
]
|
|
128
|
+
def redact_secrets(text: str) -> str:
|
|
129
|
+
"""Redact secret VALUES in *text*, keeping the type/location.
|
|
130
|
+
|
|
131
|
+
A second pass is safe: already-redacted spans do not match value patterns.
|
|
132
|
+
"""
|
|
133
|
+
if not text:
|
|
134
|
+
return text
|
|
135
|
+
out = text
|
|
136
|
+
for _name, rx, build in _SECRET_RULES:
|
|
137
|
+
out = rx.sub(build, out)
|
|
138
|
+
return out
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
# --------------------------------------------------------------------------- #
|
|
142
|
+
# is_safe_url — SSRF guard
|
|
143
|
+
# --------------------------------------------------------------------------- #
|
|
144
|
+
def _is_ip_literal(host: str) -> bool:
|
|
145
|
+
h = host.strip("[]")
|
|
146
|
+
try:
|
|
147
|
+
ipaddress.ip_address(h)
|
|
148
|
+
return True
|
|
149
|
+
except ValueError:
|
|
150
|
+
return False
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _is_blocked_ip(ip: str) -> bool:
|
|
154
|
+
"""True if the IP is private/loopback/link-local/reserved/multicast/etc.
|
|
155
|
+
|
|
156
|
+
Fail-closed: an unparseable address is treated as blocked.
|
|
157
|
+
"""
|
|
158
|
+
try:
|
|
159
|
+
addr = ipaddress.ip_address(ip.strip("[]"))
|
|
160
|
+
except ValueError:
|
|
161
|
+
return True
|
|
162
|
+
return (
|
|
163
|
+
addr.is_loopback
|
|
164
|
+
or addr.is_private
|
|
165
|
+
or addr.is_link_local
|
|
166
|
+
or addr.is_reserved
|
|
167
|
+
or addr.is_multicast
|
|
168
|
+
or addr.is_unspecified
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def is_safe_url(url: str, resolve_dns: bool = False) -> bool:
|
|
173
|
+
"""SSRF guard. Returns True ONLY for http(s) URLs whose host is not a
|
|
174
|
+
private/loopback/link-local/metadata/reserved address.
|
|
175
|
+
|
|
176
|
+
- Rejects non-http(s) schemes (file://, gopher://, dict://, ftp://, ...).
|
|
177
|
+
- Rejects IP-literal hosts in blocked ranges (127.0.0.1, 169.254.169.254,
|
|
178
|
+
10/8, 172.16/12, 192.168/16, ::1, fc00::/7 ...).
|
|
179
|
+
- For hostnames, when resolve_dns=True: resolves DNS and rejects if ANY
|
|
180
|
+
A/AAAA record is in a blocked range (catches DNS-rebinding to 127.0.0.1).
|
|
181
|
+
Default resolve_dns=False does a hostname-only check (no network call) —
|
|
182
|
+
the DNS-rebinding guard is opt-in for callers that will actually fetch.
|
|
183
|
+
"""
|
|
184
|
+
try:
|
|
185
|
+
parsed = urlparse(url)
|
|
186
|
+
except Exception:
|
|
187
|
+
return False
|
|
188
|
+
if parsed.scheme.lower() not in ("http", "https"):
|
|
189
|
+
return False
|
|
190
|
+
host = (parsed.hostname or "").lower()
|
|
191
|
+
if not host:
|
|
192
|
+
return False
|
|
193
|
+
# IP-literal host: check range directly (no DNS needed)
|
|
194
|
+
if _is_ip_literal(host):
|
|
195
|
+
return not _is_blocked_ip(host)
|
|
196
|
+
# Hostname: block well-known loopback / cloud-metadata aliases without DNS
|
|
197
|
+
if host in _BLOCKED_HOSTNAMES:
|
|
198
|
+
return False
|
|
199
|
+
# Hostname: optional DNS-rebinding check
|
|
200
|
+
if resolve_dns:
|
|
201
|
+
try:
|
|
202
|
+
infos = socket.getaddrinfo(host, None)
|
|
203
|
+
except socket.gaierror:
|
|
204
|
+
return False # fail closed — cannot verify, refuse to fetch
|
|
205
|
+
for info in infos:
|
|
206
|
+
if _is_blocked_ip(info[4][0]):
|
|
207
|
+
return False
|
|
208
|
+
return True
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
# --------------------------------------------------------------------------- #
|
|
212
|
+
# self-test
|
|
213
|
+
# --------------------------------------------------------------------------- #
|
|
214
|
+
def self_test():
|
|
215
|
+
"""Verify the SSRF guard rejects known-bad URLs and the redactor masks a
|
|
216
|
+
sample secret. Exits 0 on success, 1 on failure."""
|
|
217
|
+
ok = True
|
|
218
|
+
|
|
219
|
+
# --- SSRF: known-bad must REJECT (no DNS needed — IP literals) ---
|
|
220
|
+
bad_urls = [
|
|
221
|
+
"http://127.0.0.1/admin",
|
|
222
|
+
"http://169.254.169.254/latest/meta-data/", # cloud metadata
|
|
223
|
+
"http://10.0.0.1/",
|
|
224
|
+
"http://192.168.1.1/",
|
|
225
|
+
"http://172.16.0.1/",
|
|
226
|
+
"http://[::1]/",
|
|
227
|
+
"http://0.0.0.0/",
|
|
228
|
+
"file:///etc/passwd", # non-http scheme
|
|
229
|
+
"gopher://127.0.0.1:6379/", # non-http scheme
|
|
230
|
+
"dict://localhost:11211/", # non-http scheme
|
|
231
|
+
"ftp://internal/", # non-http scheme
|
|
232
|
+
"http://localhost/admin", # loopback HOSTNAME (not IP literal)
|
|
233
|
+
"http://metadata.google.internal/", # cloud metadata hostname
|
|
234
|
+
]
|
|
235
|
+
for u in bad_urls:
|
|
236
|
+
if is_safe_url(u):
|
|
237
|
+
print(f" ✗ FAIL: bad URL allowed: {u}")
|
|
238
|
+
ok = False
|
|
239
|
+
# --- SSRF: known-good must ALLOW (resolve_dns=False — no network needed) ---
|
|
240
|
+
good_urls = [
|
|
241
|
+
"https://example.com/",
|
|
242
|
+
"http://example.org/path?q=1",
|
|
243
|
+
"https://arxiv.org/abs/2605.23899",
|
|
244
|
+
"https://8.8.8.8/", # public IP literal
|
|
245
|
+
]
|
|
246
|
+
for u in good_urls:
|
|
247
|
+
if not is_safe_url(u, resolve_dns=False):
|
|
248
|
+
print(f" ✗ FAIL: good URL rejected: {u}")
|
|
249
|
+
ok = False
|
|
250
|
+
|
|
251
|
+
# --- redaction: sample secrets must be masked, type/location kept ---
|
|
252
|
+
samples = {
|
|
253
|
+
"API_KEY=sk-live-abcdef1234567890XYZ": "API_KEY=***REDACTED***",
|
|
254
|
+
'config = {"token": "ghp_abcdefghijklmnopqrstuvwxyz0123456789AB"}': "***REDACTED***",
|
|
255
|
+
"Authorization: Bearer eyJhbGciOiJIUzI1NiJ9.payload.sig": "Authorization: ***REDACTED***",
|
|
256
|
+
"aws_access_key_id = AKIAIOSFODNN7EXAMPLE": "***REDACTED***",
|
|
257
|
+
"DATABASE_PASSWORD=hunter2-super-secret": "DATABASE_PASSWORD=***REDACTED***",
|
|
258
|
+
}
|
|
259
|
+
for inp, _expected_fragment in samples.items():
|
|
260
|
+
masked = redact_secrets(inp)
|
|
261
|
+
# The secret VALUE must no longer appear; the key/type must remain.
|
|
262
|
+
if "sk-live-abcdef1234567890XYZ" in masked and inp != masked:
|
|
263
|
+
print(f" ✗ FAIL: secret value leaked: {masked}")
|
|
264
|
+
ok = False
|
|
265
|
+
if inp == masked:
|
|
266
|
+
print(f" ✗ FAIL: nothing redacted in: {inp}")
|
|
267
|
+
ok = False
|
|
268
|
+
if REDACTED not in masked:
|
|
269
|
+
print(f" ✗ FAIL: no redaction marker in: {masked}")
|
|
270
|
+
ok = False
|
|
271
|
+
|
|
272
|
+
# private-key block
|
|
273
|
+
pem = (
|
|
274
|
+
"-----BEGIN RSA PRIVATE KEY-----\n"
|
|
275
|
+
"MIIEpAIBAAKCAQEAabcd...supersecretkeymaterial\n"
|
|
276
|
+
"-----END RSA PRIVATE KEY-----"
|
|
277
|
+
)
|
|
278
|
+
masked_pem = redact_secrets(pem)
|
|
279
|
+
if "supersecretkeymaterial" in masked_pem or REDACTED not in masked_pem:
|
|
280
|
+
print(f" ✗ FAIL: private-key body leaked: {masked_pem}")
|
|
281
|
+
ok = False
|
|
282
|
+
|
|
283
|
+
# partial private key WITHOUT an END marker (rule 1b)
|
|
284
|
+
partial_pem = "-----BEGIN PRIVATE KEY-----MIIEvQIBADANBsupersecret"
|
|
285
|
+
masked_partial = redact_secrets(partial_pem)
|
|
286
|
+
if "supersecret" in masked_partial or REDACTED not in masked_partial:
|
|
287
|
+
print(f" ✗ FAIL: partial private-key leaked: {masked_partial}")
|
|
288
|
+
ok = False
|
|
289
|
+
|
|
290
|
+
# short password (rule 4b — rule 5 requires value >=8 chars)
|
|
291
|
+
masked_pw = redact_secrets("password=hunter2")
|
|
292
|
+
if "hunter2" in masked_pw or REDACTED not in masked_pw:
|
|
293
|
+
print(f" ✗ FAIL: short password leaked: {masked_pw}")
|
|
294
|
+
ok = False
|
|
295
|
+
|
|
296
|
+
# benign text must pass through unchanged (no false positives)
|
|
297
|
+
benign = "The quick brown fox jumps over the lazy dog. https://example.com/path info@example.com"
|
|
298
|
+
if redact_secrets(benign) != benign:
|
|
299
|
+
print(f" ✗ FAIL: false positive on benign text")
|
|
300
|
+
ok = False
|
|
301
|
+
|
|
302
|
+
print("safe_io self-test:", "PASS" if ok else "FAIL")
|
|
303
|
+
return 0 if ok else 1
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
if __name__ == "__main__":
|
|
307
|
+
argv = sys.argv[1:]
|
|
308
|
+
if not argv or argv[0] in ("-h", "--help"):
|
|
309
|
+
print(__doc__)
|
|
310
|
+
sys.exit(0)
|
|
311
|
+
if argv[0] == "--self-test":
|
|
312
|
+
sys.exit(self_test())
|
|
313
|
+
print("Usage: safe_io.py --self-test", file=sys.stderr)
|
|
314
|
+
sys.exit(2)
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
source_evaluator.py — borrowed from Geek-skills-deep-research, ported for the
|
|
4
|
+
research skill (F13: wired INTO the Agentic Protocol Step 2, never orphaned).
|
|
5
|
+
|
|
6
|
+
3D filter on a source list:
|
|
7
|
+
1. Authority (domain class, peer-review, primary-vs-secondary)
|
|
8
|
+
2. Freshness (publication date / last-modified)
|
|
9
|
+
3. Primary-vs-secondary (original research vs aggregation)
|
|
10
|
+
|
|
11
|
+
Accepts JSON input, emits a scored report. Default thresholds below.
|
|
12
|
+
|
|
13
|
+
Stdlib only. Python 3.9+.
|
|
14
|
+
|
|
15
|
+
Usage:
|
|
16
|
+
python3 source_evaluator.py <sources.json> [--min-authority 0.6] [--max-age-days 365]
|
|
17
|
+
python3 source_evaluator.py --self-test
|
|
18
|
+
|
|
19
|
+
Exit codes:
|
|
20
|
+
0 ≥80% of sources pass thresholds
|
|
21
|
+
1 <80% of sources pass; see output
|
|
22
|
+
2 usage error
|
|
23
|
+
"""
|
|
24
|
+
import json
|
|
25
|
+
import re
|
|
26
|
+
import sys
|
|
27
|
+
from datetime import datetime, timezone
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
from urllib.parse import urlparse
|
|
30
|
+
|
|
31
|
+
# MEDIUM-4: redact secret/PII values before persisting source-derived content.
|
|
32
|
+
# Same-dir import (script dir is on sys.path[0]); graceful fallback if absent.
|
|
33
|
+
try:
|
|
34
|
+
from safe_io import redact_secrets
|
|
35
|
+
except ImportError: # pragma: no cover - safe_io bundled alongside this script
|
|
36
|
+
redact_secrets = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
DEFAULT_MIN_AUTHORITY = 0.6
|
|
40
|
+
DEFAULT_MAX_AGE_DAYS = 365
|
|
41
|
+
DEFAULT_ACCEPT_RATIO = 0.8
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def host_match(host: str, trusted: str) -> bool:
|
|
45
|
+
"""Exact host or subdomain match — rejects lookalikes like evilgithub.com."""
|
|
46
|
+
return host == trusted or host.endswith("." + trusted)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
NEWS_DOMAINS = {"nytimes.com", "reuters.com", "bloomberg.com", "ft.com", "wsj.com"}
|
|
50
|
+
BLOG_DOMAINS = {"medium.com", "substack.com", "wordpress.com"}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def score_authority(url: str, source_meta: dict) -> float:
|
|
54
|
+
"""
|
|
55
|
+
Heuristic authority score 0-1.
|
|
56
|
+
Primary author / official docs / peer-reviewed = 0.9-1.0
|
|
57
|
+
Engineering blog / canonical framework = 0.7-0.9
|
|
58
|
+
Medium-tier publication / reputable blog = 0.5-0.7
|
|
59
|
+
Forum / aggregation / personal blog = 0.2-0.5
|
|
60
|
+
Unknown = 0.4
|
|
61
|
+
|
|
62
|
+
NOTE: `primary` and `official` flags are self-declared hints from input JSON.
|
|
63
|
+
They boost scores but should NOT be the sole basis for trust — verify
|
|
64
|
+
provenance independently before relying on them for high-stakes decisions.
|
|
65
|
+
Domain matching uses exact-host-or-subdomain (no substring), so
|
|
66
|
+
`evilgithub.com` will NOT match `github.com`.
|
|
67
|
+
"""
|
|
68
|
+
parsed = urlparse(url.lower())
|
|
69
|
+
host = (parsed.hostname or "").replace("www.", "")
|
|
70
|
+
# Authoritative classes
|
|
71
|
+
if source_meta.get("primary"):
|
|
72
|
+
return 1.0
|
|
73
|
+
if host.endswith((".edu", ".gov", ".ac.uk", ".ac.jp")):
|
|
74
|
+
return 0.95
|
|
75
|
+
if any(host_match(host, d) for d in ("arxiv.org", "nature.com", "science.org")):
|
|
76
|
+
return 0.95
|
|
77
|
+
if host_match(host, "github.io") or host_match(host, "github.com"):
|
|
78
|
+
# Official source code / docs — treat as authoritative if "official" flag
|
|
79
|
+
return 0.9 if source_meta.get("official") else 0.7
|
|
80
|
+
if host_match(host, "wikipedia.org"):
|
|
81
|
+
return 0.65 # secondary aggregator
|
|
82
|
+
# Class by exact domain allowlist (not substring — prevents spoofing)
|
|
83
|
+
if any(host_match(host, d) for d in NEWS_DOMAINS):
|
|
84
|
+
return 0.85
|
|
85
|
+
if any(host_match(host, d) for d in BLOG_DOMAINS):
|
|
86
|
+
return 0.5
|
|
87
|
+
return 0.4 # unknown
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def score_freshness(source_meta: dict, max_age_days: int, now: datetime) -> float:
|
|
91
|
+
"""
|
|
92
|
+
Heuristic freshness score 0-1.
|
|
93
|
+
- < 30 days: 1.0
|
|
94
|
+
- < 90 days: 0.9
|
|
95
|
+
- < 180 days: 0.8
|
|
96
|
+
- < 365 days: 0.7
|
|
97
|
+
- > 365 days: 0.5
|
|
98
|
+
- no date: 0.5
|
|
99
|
+
"""
|
|
100
|
+
date_str = source_meta.get("date") or source_meta.get("published") or source_meta.get("last_modified")
|
|
101
|
+
if not date_str:
|
|
102
|
+
return 0.5
|
|
103
|
+
try:
|
|
104
|
+
d = datetime.fromisoformat(date_str.replace("Z", "+00:00"))
|
|
105
|
+
except ValueError:
|
|
106
|
+
try:
|
|
107
|
+
d = datetime.strptime(date_str, "%Y-%m-%d")
|
|
108
|
+
except ValueError:
|
|
109
|
+
return 0.5
|
|
110
|
+
age = (now - d.replace(tzinfo=timezone.utc)).days if d.tzinfo is None else (now - d).days
|
|
111
|
+
if age < 30:
|
|
112
|
+
return 1.0
|
|
113
|
+
if age < 90:
|
|
114
|
+
return 0.9
|
|
115
|
+
if age < 180:
|
|
116
|
+
return 0.8
|
|
117
|
+
if age < max_age_days:
|
|
118
|
+
return 0.7
|
|
119
|
+
return 0.5
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def score_primary(source_meta: dict) -> float:
|
|
123
|
+
"""Heuristic primary-vs-secondary score 0-1."""
|
|
124
|
+
if source_meta.get("primary"):
|
|
125
|
+
return 1.0
|
|
126
|
+
if source_meta.get("secondary"):
|
|
127
|
+
return 0.3
|
|
128
|
+
return 0.6 # neutral
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def main(argv):
|
|
132
|
+
if not argv or argv[0] in ("-h", "--help"):
|
|
133
|
+
print(__doc__)
|
|
134
|
+
return 0
|
|
135
|
+
if argv[0] == "--self-test":
|
|
136
|
+
return self_test()
|
|
137
|
+
if len(argv) < 1:
|
|
138
|
+
print("Usage: source_evaluator.py <sources.json> [--min-authority 0.6] [--max-age-days 365]", file=sys.stderr)
|
|
139
|
+
return 2
|
|
140
|
+
|
|
141
|
+
sources_path = Path(argv[0])
|
|
142
|
+
min_authority = DEFAULT_MIN_AUTHORITY
|
|
143
|
+
max_age = DEFAULT_MAX_AGE_DAYS
|
|
144
|
+
accept_ratio = DEFAULT_ACCEPT_RATIO
|
|
145
|
+
|
|
146
|
+
i = 1
|
|
147
|
+
while i < len(argv):
|
|
148
|
+
if argv[i] == "--min-authority" and i + 1 < len(argv):
|
|
149
|
+
min_authority = float(argv[i + 1])
|
|
150
|
+
i += 2
|
|
151
|
+
elif argv[i] == "--max-age-days" and i + 1 < len(argv):
|
|
152
|
+
max_age = int(argv[i + 1])
|
|
153
|
+
i += 2
|
|
154
|
+
elif argv[i] == "--accept-ratio" and i + 1 < len(argv):
|
|
155
|
+
accept_ratio = float(argv[i + 1])
|
|
156
|
+
i += 2
|
|
157
|
+
else:
|
|
158
|
+
i += 1
|
|
159
|
+
|
|
160
|
+
data = json.loads(sources_path.read_text(encoding="utf-8"))
|
|
161
|
+
sources = data.get("sources", [])
|
|
162
|
+
now = datetime.now(timezone.utc)
|
|
163
|
+
|
|
164
|
+
scored = []
|
|
165
|
+
for s in sources:
|
|
166
|
+
url = s.get("url", "")
|
|
167
|
+
meta = {k: v for k, v in s.items() if k != "url"}
|
|
168
|
+
a = score_authority(url, meta)
|
|
169
|
+
f = score_freshness(meta, max_age, now)
|
|
170
|
+
p = score_primary(meta)
|
|
171
|
+
# Composite: 0.5 * authority + 0.3 * freshness + 0.2 * primary
|
|
172
|
+
composite = 0.5 * a + 0.3 * f + 0.2 * p
|
|
173
|
+
passes = a >= min_authority and f >= 0.5
|
|
174
|
+
scored.append({
|
|
175
|
+
"url": url,
|
|
176
|
+
"title": s.get("title", ""),
|
|
177
|
+
"authority": round(a, 3),
|
|
178
|
+
"freshness": round(f, 3),
|
|
179
|
+
"primary": round(p, 3),
|
|
180
|
+
"composite": round(composite, 3),
|
|
181
|
+
"passes": passes,
|
|
182
|
+
})
|
|
183
|
+
|
|
184
|
+
accepted = sum(1 for s in scored if s["passes"])
|
|
185
|
+
ratio = accepted / max(len(scored), 1)
|
|
186
|
+
ok = ratio >= accept_ratio
|
|
187
|
+
|
|
188
|
+
result = {
|
|
189
|
+
"ok": ok,
|
|
190
|
+
"thresholds": {
|
|
191
|
+
"min_authority": min_authority,
|
|
192
|
+
"max_age_days": max_age,
|
|
193
|
+
"accept_ratio": accept_ratio,
|
|
194
|
+
},
|
|
195
|
+
"stats": {
|
|
196
|
+
"total": len(scored),
|
|
197
|
+
"accepted": accepted,
|
|
198
|
+
"rejected": len(scored) - accepted,
|
|
199
|
+
"ratio": round(ratio, 3),
|
|
200
|
+
},
|
|
201
|
+
"sources": scored,
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
# MEDIUM-4: mask any secret/PII values that leaked into source metadata
|
|
205
|
+
# (titles, echoed content) before printing the report.
|
|
206
|
+
out = json.dumps(result, indent=2)
|
|
207
|
+
if redact_secrets:
|
|
208
|
+
out = redact_secrets(out)
|
|
209
|
+
print(out)
|
|
210
|
+
return 0 if ok else 1
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def self_test():
|
|
214
|
+
"""Run a quick self-test on a fabricated source list."""
|
|
215
|
+
sources = {
|
|
216
|
+
"sources": [
|
|
217
|
+
{"url": "https://arxiv.org/abs/2605.23899", "date": "2026-06-01", "primary": True, "title": "SkillLens"},
|
|
218
|
+
{"url": "https://github.com/foo/bar", "official": True, "date": "2026-05-01", "title": "Official repo"},
|
|
219
|
+
{"url": "https://en.wikipedia.org/wiki/X", "date": "2026-07-01", "title": "Wikipedia article"},
|
|
220
|
+
{"url": "https://nytimes.com/article", "date": "2026-07-15", "title": "NYTimes article"},
|
|
221
|
+
]
|
|
222
|
+
}
|
|
223
|
+
import tempfile
|
|
224
|
+
with tempfile.NamedTemporaryFile("w", suffix=".json", delete=False) as f:
|
|
225
|
+
json.dump(sources, f)
|
|
226
|
+
sp = f.name
|
|
227
|
+
code = main([sp])
|
|
228
|
+
Path(sp).unlink()
|
|
229
|
+
print(f"\nself-test: exit={code} (expected 0 — 4 sources, all above 0.6 authority)")
|
|
230
|
+
return code
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
if __name__ == "__main__":
|
|
234
|
+
sys.exit(main(sys.argv[1:]))
|