honestweek 0.0.0-stage → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +17 -0
- package/.claude-plugin/plugin.json +9 -0
- package/LICENSE +21 -0
- package/README.md +661 -2
- package/SKILL.md +129 -0
- package/bin/honestweek.mjs +240 -0
- package/honestweek.config.example.json +76 -0
- package/lib/archive.mjs +63 -0
- package/lib/atomic-json.mjs +42 -0
- package/lib/badges.mjs +43 -0
- package/lib/bounded-jsonl.mjs +39 -0
- package/lib/build.mjs +680 -0
- package/lib/carry-receipts.mjs +102 -0
- package/lib/carry-recovery.mjs +19 -0
- package/lib/claude-adapter.mjs +602 -0
- package/lib/client.mjs +397 -0
- package/lib/codex-records.mjs +167 -0
- package/lib/config.mjs +511 -0
- package/lib/curation-similarity.mjs +20 -0
- package/lib/demo/content.mjs +262 -0
- package/lib/demo/extra.mjs +1061 -0
- package/lib/demo/repo.mjs +385 -0
- package/lib/demo/week.mjs +2558 -0
- package/lib/digest-carry.mjs +538 -0
- package/lib/digest-curation.mjs +548 -0
- package/lib/digest-evidence.mjs +348 -0
- package/lib/digest-lifecycle.mjs +142 -0
- package/lib/digest-schema.mjs +58 -0
- package/lib/digest-source-bound.mjs +46 -0
- package/lib/digest-store.mjs +485 -0
- package/lib/digest.mjs +437 -0
- package/lib/discover.mjs +193 -0
- package/lib/emit/_shared.mjs +122 -0
- package/lib/emit/changelog.mjs +62 -0
- package/lib/emit/client.mjs +311 -0
- package/lib/emit/digest.mjs +53 -0
- package/lib/emit/goals-page.mjs +416 -0
- package/lib/emit/index.mjs +155 -0
- package/lib/emit/page.mjs +423 -0
- package/lib/emit/post.mjs +22 -0
- package/lib/emit/report.mjs +51 -0
- package/lib/git.mjs +568 -0
- package/lib/goals.mjs +539 -0
- package/lib/handoffs.mjs +151 -0
- package/lib/harvest.mjs +162 -0
- package/lib/history.mjs +110 -0
- package/lib/init.mjs +576 -0
- package/lib/invocation.mjs +102 -0
- package/lib/loopback-host.mjs +16 -0
- package/lib/mine/corpus.mjs +641 -0
- package/lib/mine/detect.mjs +681 -0
- package/lib/mine/draft.mjs +262 -0
- package/lib/mine/ledger.mjs +189 -0
- package/lib/mine/rank.mjs +236 -0
- package/lib/mine.mjs +381 -0
- package/lib/preview.mjs +504 -0
- package/lib/private-words.mjs +31 -0
- package/lib/problems/catalog.json +5884 -0
- package/lib/problems/checks.mjs +1454 -0
- package/lib/problems/classify.mjs +910 -0
- package/lib/problems/context.mjs +448 -0
- package/lib/problems/drafts.mjs +57 -0
- package/lib/problems/fix-tests.mjs +270 -0
- package/lib/problems/index.mjs +332 -0
- package/lib/problems/scope.mjs +401 -0
- package/lib/prompt-adapters.mjs +229 -0
- package/lib/prompt-curation.mjs +71 -0
- package/lib/prompt-identity.mjs +18 -0
- package/lib/prompt-lane.mjs +84 -0
- package/lib/prompt-lock.mjs +23 -0
- package/lib/prompt-privacy.mjs +451 -0
- package/lib/prompt-store.mjs +111 -0
- package/lib/prompts.mjs +100 -0
- package/lib/reader.mjs +189 -0
- package/lib/readers/client.json +9 -0
- package/lib/readers/default.json +5 -0
- package/lib/redact.mjs +577 -0
- package/lib/redaction-patterns.mjs +993 -0
- package/lib/replay/assemble.mjs +523 -0
- package/lib/replay/classify.mjs +464 -0
- package/lib/replay/claude.mjs +937 -0
- package/lib/replay/codex-program.mjs +459 -0
- package/lib/replay/codex.mjs +908 -0
- package/lib/replay/evidence.mjs +78 -0
- package/lib/replay/goals.mjs +287 -0
- package/lib/replay/ids.mjs +54 -0
- package/lib/replay/index.mjs +690 -0
- package/lib/replay/jsonl.mjs +102 -0
- package/lib/replay/launch.mjs +164 -0
- package/lib/replay/lookup.mjs +374 -0
- package/lib/replay/metrics.mjs +51 -0
- package/lib/replay/outcomes.mjs +233 -0
- package/lib/replay/parse-common.mjs +281 -0
- package/lib/replay/sources.mjs +205 -0
- package/lib/replay/timeline.mjs +331 -0
- package/lib/replay/views.mjs +340 -0
- package/lib/repo-identity.mjs +101 -0
- package/lib/resolve-week.mjs +178 -0
- package/lib/site/adapter.mjs +168 -0
- package/lib/site/archive.mjs +48 -0
- package/lib/site/derive.mjs +443 -0
- package/lib/site/detect.mjs +138 -0
- package/lib/site/emit-site.mjs +94 -0
- package/lib/site/fact-fence.mjs +148 -0
- package/lib/site/inspect.mjs +102 -0
- package/lib/site/load-adapter.mjs +30 -0
- package/lib/site/sessions.mjs +274 -0
- package/lib/site/transform.mjs +114 -0
- package/lib/site/values.mjs +153 -0
- package/lib/site/week-grid.mjs +49 -0
- package/lib/validate.mjs +258 -0
- package/lib/view/assets/common.css +789 -0
- package/lib/view/assets/common.js +1007 -0
- package/lib/view/assets/evidence.js +64 -0
- package/lib/view/assets/facts.js +100 -0
- package/lib/view/assets/form.js +211 -0
- package/lib/view/assets/goal.html +92 -0
- package/lib/view/assets/goal.js +638 -0
- package/lib/view/assets/insights.js +188 -0
- package/lib/view/assets/key.js +355 -0
- package/lib/view/assets/prefs.js +250 -0
- package/lib/view/assets/private-text.js +41 -0
- package/lib/view/assets/problems.css +191 -0
- package/lib/view/assets/problems.html +75 -0
- package/lib/view/assets/problems.js +1069 -0
- package/lib/view/assets/replay-model.js +704 -0
- package/lib/view/assets/replay.css +306 -0
- package/lib/view/assets/replay.html +122 -0
- package/lib/view/assets/replay.js +2079 -0
- package/lib/view/assets/search.html +48 -0
- package/lib/view/assets/search.js +615 -0
- package/lib/view/assets/sessions.js +144 -0
- package/lib/view/assets/settings.html +99 -0
- package/lib/view/assets/settings.js +183 -0
- package/lib/view/assets/setup.html +96 -0
- package/lib/view/assets/setup.js +131 -0
- package/lib/view/assets/strip.js +465 -0
- package/lib/view/codex-judge.mjs +411 -0
- package/lib/view/data.mjs +1339 -0
- package/lib/view/facts.mjs +306 -0
- package/lib/view/insights.mjs +309 -0
- package/lib/view/leaks.mjs +252 -0
- package/lib/view/problems-route.mjs +425 -0
- package/lib/view/progressive.mjs +134 -0
- package/lib/view/replay-export.mjs +252 -0
- package/lib/view/selftest/clickthrough.html +32 -0
- package/lib/view/selftest/clickthrough.js +2678 -0
- package/lib/view/server.mjs +336 -0
- package/lib/view/settings.mjs +369 -0
- package/lib/view/setup.mjs +286 -0
- package/lib/view/window.mjs +204 -0
- package/lib/view/word-index.mjs +192 -0
- package/lib/view.mjs +572 -0
- package/lib/voice-fence.mjs +213 -0
- package/lib/windows-root.mjs +17 -0
- package/lib/worktrees.mjs +260 -0
- package/package.json +44 -4
|
@@ -0,0 +1,993 @@
|
|
|
1
|
+
// Shared pattern authority for the canonical scrubber, the secrets-only scrubber, and the
|
|
2
|
+
// replayable prompt audit.
|
|
3
|
+
export const REDACTION_SOURCES=Object.freeze({
|
|
4
|
+
uuid:String.raw`\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b`,
|
|
5
|
+
// Searched with emailSpans() below, which finds what a global search finds in linear time.
|
|
6
|
+
email:String.raw`\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b`,
|
|
7
|
+
api:[String.raw`\bsk-[A-Za-z0-9_-]{16,}\b`,String.raw`\bgh[pousr]_[A-Za-z0-9]{20,}\b`,String.raw`\bAKIA[0-9A-Z]{12,}\b`,String.raw`\bxox[abprs]-[A-Za-z0-9-]{10,}`,String.raw`\beyJ[A-Za-z0-9_-]+(?:\.[A-Za-z0-9_-]+){2,}`,String.raw`\bglpat-[A-Za-z0-9_-]{20,}`],
|
|
8
|
+
keyValue:String.raw`\b([A-Za-z_][A-Za-z0-9_]*)\s*=(?!>)\s*("[^"]*"|'[^']*'|\S+)`,
|
|
9
|
+
sensitiveKey:String.raw`(?<![A-Za-z])(?:API_KEY|APIKEY|ACCESS_KEY|PRIVATE_KEY|PASSWORD|PASSWD|AUTHORIZATION|SECRET|TOKEN|AUTH)(?![A-Za-z])`,
|
|
10
|
+
currency:[String.raw`\$\s?\d[\d,]*(?:\.\d+)?`,String.raw`\b(?:USD|EUR|GBP|CAD|AUD|JPY)\s?\$?\s?\d[\d,]*(?:\.\d+)?`,String.raw`(?<![\d,])\b\d[\d,]*(?:\.\d+)?\s?(?:dollars?|euros?|pounds?|cents?|USD|EUR|GBP)\b`],
|
|
11
|
+
// Home / user paths. The trailing `~/…` form is as revealing as an absolute
|
|
12
|
+
// one: session logs are full of it, and the segment after it routinely names
|
|
13
|
+
// a client or a private project. Keep it LAST so the `i===1` git-bash flag
|
|
14
|
+
// mapping in redact.mjs stays put.
|
|
15
|
+
//
|
|
16
|
+
// The absolute forms tolerate a space because the USERNAME segment can hold
|
|
17
|
+
// one (Windows "Alex Jordan"). `~` is already the home dir, so there is no
|
|
18
|
+
// username to span and the tilde form must NOT consume spaces: paths are
|
|
19
|
+
// redacted at step 5, before SHAs are protected at step 7, so a space-greedy
|
|
20
|
+
// `~/` would swallow the rest of the line and take the git SHAs and counts
|
|
21
|
+
// the module promises to spare with it.
|
|
22
|
+
//
|
|
23
|
+
// The Windows form accepts runs of separators: a path inside JSON-encoded text
|
|
24
|
+
// (a Codex tool call's arguments, for one) arrives as `C:\\Users\\name\\…`, and
|
|
25
|
+
// a single-separator pattern let the username through. Three forward slashes
|
|
26
|
+
// after the colon are a URL scheme (`file:///Users/…`), left to the POSIX form.
|
|
27
|
+
// The username never crosses a double quote (so a match can't run into the next
|
|
28
|
+
// JSON field), and a bare home folder whose username holds a space is taken
|
|
29
|
+
// whole when a double quote closes it. Every quantified piece is disjoint from
|
|
30
|
+
// its neighbour, so matching stays linear on long runs of separators; redact.mjs
|
|
31
|
+
// hands back backslashes that end a match right before a quote (`\"`).
|
|
32
|
+
paths:[String.raw`[A-Za-z]:(?!///)[\\/]+(?:Users|users|USERS)[\\/]+(?:[^/\\\n"]+[\\/][^\s"'\n]*|[^\s/\\"'\n]+(?: [^\s/\\"'\n]+){1,2}(?=")|[^\s/\\"'\n]+)`,String.raw`/[a-z]/Users/(?:[^/\\\n]+[\\/][^\s"'\n]*(?:[\\/][^\s"'\n]*)*|[^\s/\\"'\n]+)`,String.raw`/home/(?:[^/\\\n]+[\\/][^\s"'\n]*(?:[\\/][^\s"'\n]*)*|[^\s/\\"'\n]+)`,String.raw`/Users/(?:[^/\\\n]+[\\/][^\s"'\n]*(?:[\\/][^\s"'\n]*)*|[^\s/\\"'\n]+)`,String.raw`(?<![\w-])(?:[A-Za-z]-)?-(?:Users|users|home)-[^\s"'/\\-][^\s"'/\\]*`,String.raw`(?:[A-Za-z]%3[Aa])?(?<!%(?:5[Cc]|2[Ff]))(?:%(?:5[Cc]|2[Ff]))+(?:Users|users|USERS|home)(?:%(?:5[Cc]|2[Ff]))+[^\s"'&?#<>%][^\s"'&?#<>]*`,String.raw`~[\\/][^\s"'\n]+`],
|
|
33
|
+
sha:String.raw`\b[0-9a-f]{7,40}\b`,account:String.raw`(?<!\.)\b\d{9,}\b(?![%.])`,opaque:String.raw`\b[A-Za-z0-9_+/=-]{32,}\b`,
|
|
34
|
+
});
|
|
35
|
+
export const regex=(source,flags='g')=>new RegExp(source,flags);
|
|
36
|
+
/** `/root/…`, the root account's home folder (in a container, say). Like `~/…` it has no
|
|
37
|
+
* username to span. It's read only where it starts a path, so `/var/root/x`, `./root/x` and
|
|
38
|
+
* `example.com/root/x` stay, and a bare `/root` stays: Codex names its root agent that. It's
|
|
39
|
+
* hidden over a scrubber's finished text (finishedSpans), so it only ever hides more. */
|
|
40
|
+
export const ROOT_PATH_SOURCE=String.raw`(?<![\w.~-])/root/[^\s"'\n]+`;
|
|
41
|
+
/** A Codex agent address, the whole of a value: the root agent `/root`, and a spawned one
|
|
42
|
+
* under it (`/root/wide_fixtures`). Shown as written where it comes from Codex's own agent
|
|
43
|
+
* fields (lib/replay/codex.mjs), never found by its shape in other text. */
|
|
44
|
+
export const AGENT_ADDRESS_RE=/^\/root(?:\/[A-Za-z0-9_-]+)*$/;
|
|
45
|
+
/** Where a home-path match should end: backslashes that end it right before a double quote
|
|
46
|
+
* belong to an escaped quote in JSON text (`…\\repo\"`), so they stay. A backward loop, so
|
|
47
|
+
* a long run of backslashes costs linear time. `end` is exclusive, in UTF-16 units. */
|
|
48
|
+
export function pathMatchEnd(text,start,end){let n=0;while(end-n>start&&text[end-1-n]==='\\')n+=1;return n&&text[end]==='"'?end-n:end;}
|
|
49
|
+
export const escapeRegex=(s)=>s.replace(/[.*+?^${}()|[\]\\]/g,'\\$&');
|
|
50
|
+
const EMAIL_CHAR=/[A-Za-z0-9._%+-]/;
|
|
51
|
+
const WORD_CHAR=/[A-Za-z0-9_]/;
|
|
52
|
+
/** Where the email pattern matches in `text`, as [start, end) pairs in UTF-16 units: exactly
|
|
53
|
+
* what a global search with it finds, in linear time. The pattern is tried only where a
|
|
54
|
+
* word boundary meets an address character, and when it fails there the rest of that run
|
|
55
|
+
* of address characters is skipped: "@" isn't one of them, so every later start in the run
|
|
56
|
+
* reaches the same "@" (or none) and the same domain, and fails too. A global search tries
|
|
57
|
+
* each of those starts, which made a long run with no address (`x-x-x-…`) quadratic. */
|
|
58
|
+
export function emailSpans(text){
|
|
59
|
+
const re=new RegExp(REDACTION_SOURCES.email,'y');
|
|
60
|
+
const spans=[];
|
|
61
|
+
for(let p=0;p<text.length;){
|
|
62
|
+
if(!EMAIL_CHAR.test(text[p])||WORD_CHAR.test(text[p-1]??'')===WORD_CHAR.test(text[p])){p+=1;continue;}
|
|
63
|
+
re.lastIndex=p;
|
|
64
|
+
const m=re.exec(text);
|
|
65
|
+
if(m){spans.push([p,p+m[0].length]);p+=m[0].length;continue;}
|
|
66
|
+
while(p<text.length&&EMAIL_CHAR.test(text[p]))p+=1;
|
|
67
|
+
}
|
|
68
|
+
return spans;
|
|
69
|
+
}
|
|
70
|
+
/** Where an email address with an encoded "@" sits in `text` (`name%40example.com`,
|
|
71
|
+
* `name@example.com`, `name@example.com`), as [start, end) pairs in UTF-16 units.
|
|
72
|
+
* Found from each encoded "@" outward, the name capped at 64 characters and the domain at
|
|
73
|
+
* 255, so a long run costs linear time. */
|
|
74
|
+
const ENCODED_AT_RE = /%40|\\u0040|\\x40|�*64;|�*40;/gi;
|
|
75
|
+
const DOMAIN_RE = /[A-Za-z0-9](?:[A-Za-z0-9-]*[A-Za-z0-9])?(?:\.[A-Za-z0-9](?:[A-Za-z0-9-]*[A-Za-z0-9])?)*\.[A-Za-z]{2,}(?![A-Za-z0-9])/y;
|
|
76
|
+
export function encodedEmailSpans(text) {
|
|
77
|
+
const spans = [];
|
|
78
|
+
let last = 0;
|
|
79
|
+
for (const m of text.matchAll(ENCODED_AT_RE)) {
|
|
80
|
+
if (m.index < last) continue;
|
|
81
|
+
let from = m.index;
|
|
82
|
+
while (from > Math.max(last, m.index - 64) && /[A-Za-z0-9._+-]/.test(text[from - 1])) from -= 1;
|
|
83
|
+
if (from === m.index) continue;
|
|
84
|
+
const at = m.index + m[0].length;
|
|
85
|
+
DOMAIN_RE.lastIndex = at;
|
|
86
|
+
const d = DOMAIN_RE.exec(text.slice(0, Math.min(text.length, at + 255)));
|
|
87
|
+
if (!d) continue;
|
|
88
|
+
spans.push([from, at + d[0].length]);
|
|
89
|
+
last = at + d[0].length;
|
|
90
|
+
}
|
|
91
|
+
return spans;
|
|
92
|
+
}
|
|
93
|
+
/** Each configured term as patterns, used by the canonical scrubber and the prompt audit.
|
|
94
|
+
* 1. A web-address or file-name part that starts with the term, or has it at a sub-word
|
|
95
|
+
* start (after "-", "_", a digit, or a camel-case boundary), is replaced whole:
|
|
96
|
+
* `www.acmehq.com`, `acmereport.pdf`, `api.my-acmecloud.io`, `http://acmehq:3000`. Only a
|
|
97
|
+
* term of four or more letters with no "." of its own is matched this way, so a short
|
|
98
|
+
* name like "Ion" never takes `session.ts` or `window.location.href` with it. A person's
|
|
99
|
+
* name (`names`) is matched this way only in a web address (after `://`, `@` or `www.`,
|
|
100
|
+
* or before a common ending like `.com`), so "Bill" or "Mark" don't take `billing.ts`
|
|
101
|
+
* or `markdown.ts` with them.
|
|
102
|
+
* 2. The term as a word. Only a letter next to it hides it, so `name_report`, `name2` and
|
|
103
|
+
* `report-name` match, and so does a camel-case part (`NameReport`, `myName`,
|
|
104
|
+
* `NAMEReport`, `XMLNameThing`); `names` and `Namesake` don't. Letters are spelled in
|
|
105
|
+
* both cases instead of using the i flag, so the edges can tell a capital from a small
|
|
106
|
+
* letter. The scrubber relies on that too: it skips a plain English term when the
|
|
107
|
+
* lowercased text lacks one of its words, which is safe only while each of the term's
|
|
108
|
+
* letters matches just its own two cases.
|
|
109
|
+
* A multi-word term matches its words split by spaces, by "-", "_" or ".", or run together
|
|
110
|
+
* (`Jane Doe`, `jane_doe`, `JaneDoe`); in a web-address part, by "-", "_" or nothing.
|
|
111
|
+
* The web-address patterns are marked `acrossPlaceholders`: the scrubber runs them over its
|
|
112
|
+
* whole text, placeholders included, so a later dotted piece may be one
|
|
113
|
+
* (`acmelogo.<hash>.png`); they never match inside a placeholder. The scrubber and the
|
|
114
|
+
* audit both take every pattern's matches together and replace the widest, so they agree
|
|
115
|
+
* when terms overlap. Each start is tried once from the front of a word or part, so
|
|
116
|
+
* matching stays linear. */
|
|
117
|
+
const anyCase=(s)=>[...s].map((c)=>{const lo=c.toLowerCase();const up=c.toUpperCase();return lo!==up&&[...lo].length===1&&[...up].length===1?`[${c===lo||c===up?'':c}${lo}${up}]`:escapeRegex(c);}).join('');
|
|
118
|
+
const START_EDGE='(?:(?<!\\p{L})|(?<=\\p{Ll})(?=\\p{Lu})|(?<=\\p{Lu})(?=\\p{Lu}\\p{Ll}))';
|
|
119
|
+
const edge=(c,side)=>/\p{L}/u.test(c)?(side==='start'?START_EDGE:'(?:(?!\\p{L})|(?<=\\p{Ll})(?=\\p{Lu})|(?<=\\p{Lu})(?=\\p{Lu}\\p{Ll}))'):/\p{N}/u.test(c)?(side==='start'?'(?<![\\p{L}\\p{N}])':'(?![\\p{L}\\p{N}])'):(side==='start'?'(?<![\\p{L}\\p{N}_])':'(?![\\p{L}\\p{N}_])');
|
|
120
|
+
const PART='[\\p{L}\\p{N}_-]';
|
|
121
|
+
// A later dotted piece: an ordinary one, or a placeholder the scrubber set aside (a token
|
|
122
|
+
// made of a Private-Use-Area delimiter, see freshDelimiter).
|
|
123
|
+
const PIECE=`(?:${PART}{1,63}|[\\uE000-\\uF8FF]+\\d+[\\uE000-\\uF8FF]+)`;
|
|
124
|
+
const ENDING='\\.\\p{L}[\\p{L}\\p{N}]{1,23}(?![\\p{L}\\p{N}])';
|
|
125
|
+
const WEB_ENDING='\\.(?:com|org|net|io|dev|app|co|ai|me|us|uk|de|fr|ca|au|nz|edu|gov|info|biz|xyz|tech|cloud|site|online|page|blog|gg|tv|fm|nl|ch|eu|es|it|se|no|dk|fi|jp|br)(?![\\p{L}\\p{N}])';
|
|
126
|
+
const AFTER_SCHEME='(?<=:\\/\\/(?:[^\\s/@]{1,256}@)?)';
|
|
127
|
+
const across=(re)=>Object.assign(re,{acrossPlaceholders:true});
|
|
128
|
+
// --- secret fields, shared by both scrubbers and the prompt audit -------------------
|
|
129
|
+
//
|
|
130
|
+
// The value of a sensitive field (KEY=VALUE, key: value to the end of the line, "key":
|
|
131
|
+
// "value" with escaped quotes, :=, ?=, ==, a typed `password: string = …`), a sensitive
|
|
132
|
+
// --flag or -Parameter value, Authorization and Cookie headers, Bearer/Basic credentials,
|
|
133
|
+
// the password in a web address or after curl's -u, and a PowerShell SecureString literal.
|
|
134
|
+
|
|
135
|
+
// A sensitive key, tested after camelCase and "-" become "_" (dbPassword, x-api-key,
|
|
136
|
+
// _authToken, authtoken, PGPASSWORD, MYSQL_PWD, ENCRYPTION_KEY, clientsecret, secretkey,
|
|
137
|
+
// X-Amz-Signature, ?sig=).
|
|
138
|
+
// A key merely ending in "Key" isn't one: the engine's own fileKey and sessionKey hold ids.
|
|
139
|
+
const SECRET_KEY_RE = /(?:(?<![A-Za-z])(?:API_?KEYS?|ACCESS_?KEY|PRIVATE_?KEY|PASSPHRASE|PASS|AUTHORIZATION|AUTH|COOKIE|CREDENTIALS?|SIGNATURE|SIG)|PASSWORD|PASSWD|TOKEN|SECRET(?:_?KEY)?|(?<=_)PWD|(?:ENCRYPTION|SIGNING|MASTER|CLIENT|LICENSE|SERVICE|DEPLOY|SESSION_SECRET|SSH|GPG|PGP|HMAC|JWT|AES|APP)_?KEY)(?![A-Za-z])/i;
|
|
140
|
+
export const normalKey = (key) => key.replace(/([a-z0-9])([A-Z])/g, '$1_$2').replace(/-/g, '_');
|
|
141
|
+
// ODBC's `PWD=…;` (any case) counts only as the whole key, so `pwd-…` in a token isn't one.
|
|
142
|
+
export const keyIsSecret = (key) => SECRET_KEY_RE.test(normalKey(key)) || /(?:^|\.)pwd$/i.test(key);
|
|
143
|
+
// Keys whose value may hold spaces, so a "key: value" runs to the end of the line.
|
|
144
|
+
const PASSWORD_KEY_RE = /(?:PASSWORD|PASSWD|PASSPHRASE|(?<=_)PASS|(?<=_)PWD)(?![A-Za-z])/i;
|
|
145
|
+
const TYPE_AHEAD_RE = /^(?:string|str|String|number|boolean|bytes|SecretStr|SecureString|Option<|Optional\[|any)\b/;
|
|
146
|
+
// A value read after ":" that is a type and nothing else (`string`, `str[]`, `Option<String>`).
|
|
147
|
+
const TYPE_ONLY_RE = /^(?:(?:string|str|String|number|boolean|bytes|SecretStr|SecureString|any)(?:\[\])*|Option<[\w<>]*>|Optional\[[\w[\]]*\])$/;
|
|
148
|
+
const WHOLE_LINE_KEY_RE = /(?:^|_)(?:AUTHORIZATION|COOKIE)$/i;
|
|
149
|
+
// Each identifier is read once from its front; its value is read only when the key is
|
|
150
|
+
// sensitive, so an ordinary key never swallows the next one and matching stays linear.
|
|
151
|
+
const IDENTIFIER_RE = /(?<![A-Za-z0-9_])[A-Za-z0-9_][A-Za-z0-9_-]*/g;
|
|
152
|
+
// The rest of a dotted flag name after its first part (`.token` in `--docs.token`). Each part
|
|
153
|
+
// starts with a letter, digit or "_", so a sentence's closing dot isn't one.
|
|
154
|
+
const DOTTED_NAME_RE = /(?:\.[A-Za-z0-9_][A-Za-z0-9_-]*)+/y;
|
|
155
|
+
const HSPACE = String.raw`[^\S\r\n]*`;
|
|
156
|
+
// After the key: its closing quote (escaped or not), a Go type (`var password string =`),
|
|
157
|
+
// an optional `?` (`password?: string`), the separator, and a type name between ":" and
|
|
158
|
+
// "=" when there is one (`password: string = "…"`).
|
|
159
|
+
const TYPE_NAME = String.raw`(?:string|str|String|bytes|\[\]byte|SecretStr|SecureString|any|&str|char\s?\*)`;
|
|
160
|
+
const FIELD_SEP_RE = new RegExp(String.raw`\\*["']?(?:[^\S\r\n]+${TYPE_NAME}(?=${HSPACE}=))?${HSPACE}\??(===|=>|:=|\?=|==|=|:)${HSPACE}(?:${TYPE_NAME}${HSPACE}=${HSPACE})?`, 'y');
|
|
161
|
+
// The next `key: ` on a joined line: a colon followed by a space or a quote, so a credential
|
|
162
|
+
// like `AWS id:signature`, a web address or a time doesn't count.
|
|
163
|
+
const NEXT_KEY_RE = /[A-Za-z_][A-Za-z0-9_-]*["']?[ \t]*:(?=[ \t"'])/y;
|
|
164
|
+
// A quoted value whose would-be closing quote opens a sensitive key's own value instead
|
|
165
|
+
// (`token: "Ab3d, api_key: "Xk9…"`) was left unclosed: it ends before that key.
|
|
166
|
+
const KEY_BEFORE_QUOTE_RE = /([A-Za-z_][A-Za-z0-9_-]*)\\*["']?[ \t]*(?::=|\?=|===|=>|==|=|:)[ \t]*\\*$/;
|
|
167
|
+
// After a --flag or -Parameter: spaces, then a value that isn't another flag.
|
|
168
|
+
const FLAG_SEP_RE = new RegExp(String.raw`[^\S\r\n]+(?=[^\s-])`, 'y');
|
|
169
|
+
const SCHEME_RE = /(?:bearer|basic|token|digest)[^\S\r\n]+/iy;
|
|
170
|
+
/** Bearer and Basic credentials: groups are the scheme, the gap, then the credential. The
|
|
171
|
+
* scheme and one credential character are exported for a scrubber that reads the
|
|
172
|
+
* credential across its own placeholders. */
|
|
173
|
+
export const BEARER_SCHEME = 'bearer|basic';
|
|
174
|
+
export const BEARER_CHAR = '[A-Za-z0-9._~+/=-]';
|
|
175
|
+
export const BEARER_RE = new RegExp(String.raw`\b(${BEARER_SCHEME})([ \t]+)${BEARER_CHAR}{8,}`, 'gi');
|
|
176
|
+
/** True when what BEARER_RE took for a credential is a plain word (`basic validation`,
|
|
177
|
+
* `Bearer authentication`, `basic end-to-end`): lowercase, Capitalized or all-caps parts
|
|
178
|
+
* joined by "-", maybe ending a sentence. The published scrubber and the prompt audit
|
|
179
|
+
* leave those; a real credential mixes cases inside a part, or holds a digit or a symbol. */
|
|
180
|
+
export function plainWords(credential) {
|
|
181
|
+
let end = credential.length;
|
|
182
|
+
while (end > 0 && credential[end - 1] === '.') end -= 1;
|
|
183
|
+
return credential.slice(0, end).split('-').every((p) => /^(?:[a-z]+|[A-Z][a-z]*|[A-Z]+)$/.test(p));
|
|
184
|
+
}
|
|
185
|
+
/** scheme://user:password@host. Group 1 runs through the ":" before the password. The user
|
|
186
|
+
* may be empty (redis://:password@host); the password may hold an "@" itself and ends at
|
|
187
|
+
* the last one. */
|
|
188
|
+
export const URL_PASSWORD_RE = /(?<![A-Za-z0-9+.-])([A-Za-z][A-Za-z0-9+.-]*:\/\/[^\s:/@"']*:)[^\s/"']+(?=@)/g;
|
|
189
|
+
/** curl and wget basic auth: -u user:password, --user user:password, --user=user:password.
|
|
190
|
+
* Groups are the flag, the user through its ":", then the password. The user part is
|
|
191
|
+
* capped, so a long run with no ":" is scanned once per flag (linear). */
|
|
192
|
+
export const USER_PASSWORD_RE = /(?<![\w-])(-u|--user)([ \t=]+["']?[^\s:"']{1,256}:)([^\s"']+)/g;
|
|
193
|
+
/** A PowerShell SecureString literal: groups are the command (with its -String,
|
|
194
|
+
* -AsPlainText and -Force switches), the quote, then the value. */
|
|
195
|
+
export const SECURE_STRING_RE = /(ConvertTo-SecureString(?:[ \t]+-(?:String|AsPlainText|Force)(?![\w-]))*[ \t]+)(["'])([^"'\n]*)\2/gi;
|
|
196
|
+
/** A literal piped into ConvertTo-SecureString (`"…" | ConvertTo-SecureString`): groups are
|
|
197
|
+
* the quote, the value, then the pipe and the command. */
|
|
198
|
+
export const SECURE_PIPE_RE = /(["'])([^"'\n]*)\1([ \t]*\|[ \t]*ConvertTo-SecureString\b)/gi;
|
|
199
|
+
/** An XML element whose tag names a secret (`<password>…</password>`): groups are the opening
|
|
200
|
+
* tag, its name, the value, then the closing tag. Names and values are capped, so a long run
|
|
201
|
+
* is scanned once per "<". */
|
|
202
|
+
export const XML_FIELD_RE = /(<((?:[A-Za-z_][\w.-]{0,63}:)?[A-Za-z_][\w.-]{0,63})(?:[ \t][^<>\n]{0,256})?>)([^<\n]{1,4096})(<\/\2>)/g;
|
|
203
|
+
/** True when an XML tag's name (without a namespace prefix) names a secret. */
|
|
204
|
+
export const xmlTagIsSecret = (name) => keyIsSecret(name.replace(/^.*:/, ''));
|
|
205
|
+
|
|
206
|
+
/** Where a value starting at `p` ends, and the part of it to hide. A double-quoted string
|
|
207
|
+
* opened by a quote escaped with `n` backslashes (n = 0 in plain JSON, 1 in JSON inside a
|
|
208
|
+
* JSON string, 3 a level deeper) closes at the next quote whose backslash count is n, or n
|
|
209
|
+
* plus a multiple of 2(n + 1): an escaped backslash right before the closing quote doesn't
|
|
210
|
+
* hide it, and an escaped quote inside the value doesn't end it. A quoted value cut off
|
|
211
|
+
* before its close runs to the end of the line. `mode` sets where an unquoted value ends:
|
|
212
|
+
* 'line' at the end of the line (a password line, a header), 'word' at the next space
|
|
213
|
+
* (after "="), 'tight' also at , ; and at ) } ] that close the value (after ":", where JSON
|
|
214
|
+
* and code go on). `lineEnd(p)` gives the end of the line holding p. */
|
|
215
|
+
function readValue(s, p, mode, lineEnd) {
|
|
216
|
+
const open = /\\*"/y;
|
|
217
|
+
open.lastIndex = p;
|
|
218
|
+
const o = open.exec(s);
|
|
219
|
+
if (o) {
|
|
220
|
+
const nl = lineEnd(p);
|
|
221
|
+
const n = o[0].length - 1;
|
|
222
|
+
const step = ((n + 1) & n) === 0 ? 2 * (n + 1) : 0;
|
|
223
|
+
for (let i = p + o[0].length; i < nl; ) {
|
|
224
|
+
const q = s.indexOf('"', i);
|
|
225
|
+
if (q < 0 || q > nl) break;
|
|
226
|
+
let k = 0;
|
|
227
|
+
while (k < q - i && s[q - 1 - k] === '\\') k += 1;
|
|
228
|
+
if (k === n || (step && k > n && (k - n) % step === 0)) {
|
|
229
|
+
const lead = s.slice(Math.max(p + o[0].length, q - 96), q);
|
|
230
|
+
const km = KEY_BEFORE_QUOTE_RE.exec(lead);
|
|
231
|
+
if (km && keyIsSecret(km[1]) && !/[A-Za-z0-9_-]/.test(lead[km.index - 1] ?? s[q - lead.length - 1] ?? '')) {
|
|
232
|
+
let end = q - lead.length + km.index;
|
|
233
|
+
while (end > p + o[0].length && /\s/.test(s[end - 1])) end -= 1;
|
|
234
|
+
if (end > p + o[0].length) return { start: p + o[0].length, end, after: q - lead.length + km.index };
|
|
235
|
+
// Nothing before the key: this quote opens that key's value, so read on to the next close.
|
|
236
|
+
i = q + 1;
|
|
237
|
+
continue;
|
|
238
|
+
}
|
|
239
|
+
return { start: p + o[0].length, end: q - n, after: q + 1 };
|
|
240
|
+
}
|
|
241
|
+
i = q + 1;
|
|
242
|
+
}
|
|
243
|
+
let end = nl;
|
|
244
|
+
while (end > p && /\s/.test(s[end - 1])) end -= 1;
|
|
245
|
+
return end > p + o[0].length ? { start: p + o[0].length, end, after: end } : null;
|
|
246
|
+
}
|
|
247
|
+
if (s[p] === "'") {
|
|
248
|
+
const nl1 = lineEnd(p);
|
|
249
|
+
const first = s.indexOf("'", p + 1);
|
|
250
|
+
for (let q = first; q > p && q < nl1; q = s.indexOf("'", q + 1)) {
|
|
251
|
+
const lead = s.slice(Math.max(p + 1, q - 96), q);
|
|
252
|
+
const km = KEY_BEFORE_QUOTE_RE.exec(lead);
|
|
253
|
+
if (km && keyIsSecret(km[1]) && !/[A-Za-z0-9_-]/.test(lead[km.index - 1] ?? s[q - lead.length - 1] ?? '')) {
|
|
254
|
+
let end = q - lead.length + km.index;
|
|
255
|
+
while (end > p + 1 && /\s/.test(s[end - 1])) end -= 1;
|
|
256
|
+
if (end > p + 1) return { start: p + 1, end, after: q - lead.length + km.index };
|
|
257
|
+
continue;
|
|
258
|
+
}
|
|
259
|
+
return { start: p + 1, end: q, after: q + 1 };
|
|
260
|
+
}
|
|
261
|
+
if (first > p && first < nl1) return { start: p + 1, end: first, after: first + 1 };
|
|
262
|
+
}
|
|
263
|
+
let end = p;
|
|
264
|
+
// A quote with a letter, digit or "_" on both sides is part of the value (`xk9'mp2qrz`), not
|
|
265
|
+
// the end of the text around it.
|
|
266
|
+
const quoteEnds = (i) => !(/\w/.test(s[i - 1] ?? '') && /\w/.test(s[i + 1] ?? ''));
|
|
267
|
+
if (mode === 'line') {
|
|
268
|
+
// The line ends at a newline, a quote, or the next `key:`, since the engine joins an
|
|
269
|
+
// excerpt's lines into one and a later key must be read on its own.
|
|
270
|
+
const nl = lineEnd(p);
|
|
271
|
+
while (end < nl && s[end] !== '\r' && !('"\''.includes(s[end]) && quoteEnds(end))) {
|
|
272
|
+
if (end > p && /\s/.test(s[end - 1]) && /[A-Za-z_]/.test(s[end])) {
|
|
273
|
+
NEXT_KEY_RE.lastIndex = end;
|
|
274
|
+
if (NEXT_KEY_RE.test(s)) break;
|
|
275
|
+
}
|
|
276
|
+
end += 1;
|
|
277
|
+
}
|
|
278
|
+
while (end > p && /\s/.test(s[end - 1])) end -= 1;
|
|
279
|
+
} else {
|
|
280
|
+
SCHEME_RE.lastIndex = p;
|
|
281
|
+
if (SCHEME_RE.exec(s)) end = SCHEME_RE.lastIndex;
|
|
282
|
+
const closes = (i) => mode === 'tight' && ')}]'.includes(s[i]) && (i + 1 >= s.length || /[\s"',;:)}\].?!]/.test(s[i + 1]));
|
|
283
|
+
const stops = mode === 'tight' ? (i) => /[\s,;]/.test(s[i]) || ('"\''.includes(s[i]) && quoteEnds(i)) : (i) => /\s/.test(s[i]);
|
|
284
|
+
while (end < s.length && !stops(end) && !closes(end)) end += 1;
|
|
285
|
+
}
|
|
286
|
+
return end > p ? { start: p, end, after: end } : null;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/** A sensitive key's value that isn't a secret: a flag value (`requiresAuth: true`) or a
|
|
290
|
+
* test count (`pass: 793`). */
|
|
291
|
+
export const notASecret = (key, value) => /^(?:true|false|null|undefined|none|nil)$/i.test(value) || (/^pass$/i.test(key) && /^\d+$/.test(value));
|
|
292
|
+
|
|
293
|
+
/** Hide the value of every sensitive field and flag in `s`. `hide(text, start, end)` gets
|
|
294
|
+
* one value and where it sits in `s` (UTF-16 indexes, `end` exclusive), and returns its
|
|
295
|
+
* replacement, or null to leave it. */
|
|
296
|
+
export function secretFields(s, hide) {
|
|
297
|
+
let out = '';
|
|
298
|
+
let at = 0;
|
|
299
|
+
const ids = new RegExp(IDENTIFIER_RE.source, 'g');
|
|
300
|
+
let nlAt = -1;
|
|
301
|
+
const lineEnd = (p) => {
|
|
302
|
+
if (nlAt < p) {
|
|
303
|
+
nlAt = s.indexOf('\n', p);
|
|
304
|
+
if (nlAt < 0) nlAt = s.length;
|
|
305
|
+
}
|
|
306
|
+
return nlAt;
|
|
307
|
+
};
|
|
308
|
+
for (let m = ids.exec(s); m; m = ids.exec(s)) {
|
|
309
|
+
// A flag's name may hold dots (`--docs.token`, `-Docs.Token`): it's read whole, each dot
|
|
310
|
+
// standing for "_", as `--docs-token` is. A name that isn't sensitive whole is left, and
|
|
311
|
+
// its later parts are read on their own as before.
|
|
312
|
+
let name = m[0];
|
|
313
|
+
let end = m.index + m[0].length;
|
|
314
|
+
if (s[m.index - 1] === '-') {
|
|
315
|
+
DOTTED_NAME_RE.lastIndex = end;
|
|
316
|
+
const dotted = DOTTED_NAME_RE.exec(s);
|
|
317
|
+
if (dotted) {
|
|
318
|
+
name = (m[0] + dotted[0]).replace(/\./g, '_');
|
|
319
|
+
end += dotted[0].length;
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
if (!keyIsSecret(name)) continue;
|
|
323
|
+
let v = null;
|
|
324
|
+
FIELD_SEP_RE.lastIndex = end;
|
|
325
|
+
const sep = FIELD_SEP_RE.exec(s);
|
|
326
|
+
const key = normalKey(name);
|
|
327
|
+
// A password line runs to its end unless what follows is a type (`password: string)`).
|
|
328
|
+
const typed = TYPE_AHEAD_RE.test(s.slice(FIELD_SEP_RE.lastIndex, FIELD_SEP_RE.lastIndex + 24));
|
|
329
|
+
// A type annotation has no value (`login(password: string, remember: boolean)`); one with
|
|
330
|
+
// a value after it (`password: string = "…"`) was read past by the separator. Hidden as a
|
|
331
|
+
// value, it made a second pass read the field another way (a placeholder isn't a type).
|
|
332
|
+
// Anything glued to the type word is read as a value, the way it would be without it. A
|
|
333
|
+
// header (Authorization, Cookie) always runs to the end of its line.
|
|
334
|
+
if (sep && sep[1] === ':' && typed && !WHOLE_LINE_KEY_RE.test(key)) {
|
|
335
|
+
const t = readValue(s, FIELD_SEP_RE.lastIndex, 'tight', lineEnd);
|
|
336
|
+
if (t && TYPE_ONLY_RE.test(s.slice(t.start, t.end))) continue;
|
|
337
|
+
}
|
|
338
|
+
if (sep) v = readValue(s, FIELD_SEP_RE.lastIndex, WHOLE_LINE_KEY_RE.test(key) || (sep[1] === ':' && PASSWORD_KEY_RE.test(key)) ? 'line' : sep[1] === ':' ? 'tight' : 'word', lineEnd);
|
|
339
|
+
else if (s[m.index - 1] === '-') {
|
|
340
|
+
FLAG_SEP_RE.lastIndex = end;
|
|
341
|
+
if (FLAG_SEP_RE.exec(s)) v = readValue(s, FLAG_SEP_RE.lastIndex, 'word', lineEnd);
|
|
342
|
+
}
|
|
343
|
+
if (!v) continue;
|
|
344
|
+
const text = s.slice(v.start, v.end);
|
|
345
|
+
if (notASecret(name, text)) continue;
|
|
346
|
+
const hidden = hide(text, v.start, v.end);
|
|
347
|
+
if (hidden === null) continue;
|
|
348
|
+
out += s.slice(at, v.start) + hidden;
|
|
349
|
+
at = v.end;
|
|
350
|
+
ids.lastIndex = Math.max(v.after, v.end);
|
|
351
|
+
}
|
|
352
|
+
return out + s.slice(at);
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
/** True when a field's value is only `*` and `_`: the close of a Markdown bold label
|
|
356
|
+
* (`**Auth:** …`) or a value already masked (`password: ****`), never a secret. */
|
|
357
|
+
export const markupOnly = (value) => /^[*_]+$/.test(value);
|
|
358
|
+
|
|
359
|
+
// Set-aside parts are written as tokens: a delimiter of Private-Use-Area characters that the
|
|
360
|
+
// text doesn't hold, a number, the delimiter again.
|
|
361
|
+
const SENTINEL = String.fromCharCode(0xe000);
|
|
362
|
+
/** A delimiter as a pattern: a run of one character as a counted repeat, so a long run (the
|
|
363
|
+
* text held a long run of U+E000) doesn't make a pattern too large to build. */
|
|
364
|
+
const delimiterSource = (d) => (d.length > 1 && [...d].every((c) => c === d[0]) ? `${d[0]}{${d.length}}` : d);
|
|
365
|
+
/** U+E000, repeated once more than its longest run in `text` (found in one pass). */
|
|
366
|
+
function sentinelFor(text) {
|
|
367
|
+
let longest = 0;
|
|
368
|
+
for (let i = 0, run = 0; i < text.length; i += 1) {
|
|
369
|
+
run = text.charCodeAt(i) === 0xe000 ? run + 1 : 0;
|
|
370
|
+
if (run > longest) longest = run;
|
|
371
|
+
}
|
|
372
|
+
return SENTINEL.repeat(longest + 1);
|
|
373
|
+
}
|
|
374
|
+
/** The delimiter for the tokens the scrubbers and the audit set parts aside as: one
|
|
375
|
+
* Private-Use-Area character `text` doesn't hold (U+E000 unless the text has it), found in
|
|
376
|
+
* one pass, so authored text can never impersonate a token. When the text holds every one of
|
|
377
|
+
* them, a run of U+E000 longer than any in the text. */
|
|
378
|
+
export function freshDelimiter(text) {
|
|
379
|
+
const seen = new Uint8Array(0xf900 - 0xe000);
|
|
380
|
+
for (let i = 0; i < text.length; i += 1) {
|
|
381
|
+
const c = text.charCodeAt(i);
|
|
382
|
+
if (c >= 0xe000 && c < 0xf900) seen[c - 0xe000] = 1;
|
|
383
|
+
}
|
|
384
|
+
const free = seen.indexOf(0);
|
|
385
|
+
return free >= 0 ? String.fromCharCode(0xe000 + free) : sentinelFor(text);
|
|
386
|
+
}
|
|
387
|
+
/** The pattern for one token, `delimiter digits delimiter`, with the digits as group 1. */
|
|
388
|
+
export const tokenSource = (delimiter) => `${delimiterSource(delimiter)}(\\d+)${delimiterSource(delimiter)}`;
|
|
389
|
+
|
|
390
|
+
/** The published scrubber's secret-field step, shared with the prompt audit: the password in
|
|
391
|
+
* a web address and after curl's -u, a SecureString literal, the value of every sensitive
|
|
392
|
+
* field and flag, and a Bearer or Basic credential that isn't a plain word. It reads `s`, a
|
|
393
|
+
* text in which the caller set its placeholders aside as tokens (`delimiter`, a number,
|
|
394
|
+
* `delimiter`); a value made only of those is left as it is. `partText(n)` gives token n's
|
|
395
|
+
* text. `hide(value)` returns the token to put in place of one value, which later rules here
|
|
396
|
+
* read as a placeholder. Returns the new text. */
|
|
397
|
+
export function hideSecretFields(s, { delimiter, partText, hide }) {
|
|
398
|
+
const d = delimiterSource(delimiter);
|
|
399
|
+
const tokens = new RegExp(tokenSource(delimiter), 'g');
|
|
400
|
+
const held = (v) => v.replace(tokens, '').trim() === '';
|
|
401
|
+
const restore = (v) => v.replace(tokens, (m, n) => partText(Number(n)) ?? m);
|
|
402
|
+
let out = s.replace(URL_PASSWORD_RE, (m, head) => (held(m.slice(head.length)) ? m : head + hide(m.slice(head.length))));
|
|
403
|
+
out = out.replace(USER_PASSWORD_RE, (m, flag, user, value) => (held(value) ? m : flag + user + hide(value)));
|
|
404
|
+
out = out.replace(SECURE_STRING_RE, (m, head, q, value) => (value && !held(value) ? `${head}${q}${hide(value)}${q}` : m));
|
|
405
|
+
out = out.replace(SECURE_PIPE_RE, (m, q, value, tail) => (value && !held(value) ? `${q}${hide(value)}${q}${tail}` : m));
|
|
406
|
+
out = out.replace(XML_FIELD_RE, (m, open, name, value, close) => (xmlTagIsSecret(name) && value.trim() && !held(value) && !markupOnly(value.trim()) && !notASecret(name.replace(/^.*:/, ''), value.trim()) ? `${open}${hide(value)}${close}` : m));
|
|
407
|
+
out = secretFields(out, (value) => (held(value) || markupOnly(value) ? null : hide(value)));
|
|
408
|
+
// A credential may run into a placeholder (`Bearer abc[redacted:term]def`); it's hidden whole.
|
|
409
|
+
const bearer = new RegExp(String.raw`\b(${BEARER_SCHEME})([ \t]+)((?:${BEARER_CHAR}|${d}\d+${d})+)`, 'gi');
|
|
410
|
+
out = out.replace(bearer, (m, scheme, gap, credential) => {
|
|
411
|
+
const text = restore(credential);
|
|
412
|
+
return held(credential) || text.length < 8 || plainWords(text) ? m : `${scheme}${gap}${hide(credential)}`;
|
|
413
|
+
});
|
|
414
|
+
// KEY=VALUE once more, on the text as it now reads. A sensitive key the scrubber's step 2 read
|
|
415
|
+
// as part of another key's value (`x = --token=**`, until `x` was hidden here) is read for
|
|
416
|
+
// itself now, as a second pass would read it. Only the value (or a comparison's operand)
|
|
417
|
+
// goes; one that is already a placeholder, maybe quoted, stays.
|
|
418
|
+
const leadToken = new RegExp(`^${tokenSource(delimiter)}`);
|
|
419
|
+
const settled = (v) => {
|
|
420
|
+
const w = v.replace(/^["']/, '');
|
|
421
|
+
if (w.trim() === '') return w !== '';
|
|
422
|
+
const t = leadToken.exec(w);
|
|
423
|
+
return t !== null && (w.length === t[0].length || w[t[0].length] === '"' || w[t[0].length] === "'");
|
|
424
|
+
};
|
|
425
|
+
const re = new RegExp(KV_AGAIN.source, 'g');
|
|
426
|
+
const rest = /\S*/y;
|
|
427
|
+
let result = '';
|
|
428
|
+
let at = 0;
|
|
429
|
+
for (let m = re.exec(out); m !== null; m = re.exec(out)) {
|
|
430
|
+
const [whole, key, value] = m;
|
|
431
|
+
if (!SENSITIVE_KV_KEY.test(key)) continue;
|
|
432
|
+
const operand = value.replace(/^=+/, '');
|
|
433
|
+
if (operand === '' || settled(operand)) continue;
|
|
434
|
+
// What is glued after the value, read only for a value that is hidden: read at every
|
|
435
|
+
// KEY=VALUE, it made a long run with no space in it (`x='a='x='a='…`) quadratic.
|
|
436
|
+
rest.lastIndex = m.index + whole.length;
|
|
437
|
+
const tail = rest.exec(out)[0];
|
|
438
|
+
result += out.slice(at, m.index) + whole.slice(0, whole.length - operand.length) + hide(operand + tail);
|
|
439
|
+
at = m.index + whole.length + tail.length;
|
|
440
|
+
re.lastIndex = at;
|
|
441
|
+
}
|
|
442
|
+
return result + out.slice(at);
|
|
443
|
+
}
|
|
444
|
+
const KV_AGAIN = new RegExp(REDACTION_SOURCES.keyValue, 'g');
|
|
445
|
+
const SENSITIVE_KV_KEY = new RegExp(REDACTION_SOURCES.sensitiveKey, 'i');
|
|
446
|
+
|
|
447
|
+
// A record read back as JSON loses the text around its values, so a value is hidden when
|
|
448
|
+
// its own key is sensitive ({"password": "…"}, {"x-api-key": "…"}). Only a key shaped like
|
|
449
|
+
// an identifier counts: an answer keyed by a question's text keeps its value.
|
|
450
|
+
export const IDENTIFIER_KEY = /^[A-Za-z_$][\w$.-]{0,63}$/;
|
|
451
|
+
// A long number under a password-like key ({"password": 12345678}) is hidden too.
|
|
452
|
+
export const NUMERIC_SECRET_KEY_RE = /(?:PASSWORD|PASSWD|PASSPHRASE|SECRET|(?<![A-Za-z])PIN)(?![A-Za-z])/i;
|
|
453
|
+
const ANY_PLACEHOLDER_RE = /\[redacted:\w+\]/g;
|
|
454
|
+
/** True when `value` holds a placeholder and nothing else but whitespace. Checked by removing
|
|
455
|
+
* them, not by one pattern with nested repeats, which backtracks exponentially when
|
|
456
|
+
* whitespace between many placeholders is followed by other text. */
|
|
457
|
+
export function onlyPlaceholders(value) {
|
|
458
|
+
const rest = value.replace(ANY_PLACEHOLDER_RE, '');
|
|
459
|
+
return rest.length < value.length && rest.trim() === '';
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/** A term's words, split the way termMatchers splits them: each word is matched whole and in
|
|
463
|
+
* order, with only separators or nothing between them, so a text that matches a term holds
|
|
464
|
+
* every one of its words. The scrubber reads the same words to skip a term whose words
|
|
465
|
+
* aren't all in a text. */
|
|
466
|
+
export const termWords=(t)=>t.trim().split(/\s+/);
|
|
467
|
+
/** The characters that lowercase onto a plain English letter, or match one when a pattern
|
|
468
|
+
* ignores case (/iu): dotted capital I, long s and the Kelvin sign. A text holding one runs
|
|
469
|
+
* every term's patterns, whatever its words. */
|
|
470
|
+
export const FOLDS_ONTO_ASCII=/[\u0130\u017F\u212A]/;
|
|
471
|
+
/** `text` lowercased for the term check, with those three characters folded onto the letter
|
|
472
|
+
* they could stand for (i, s, k), so the check still finds every word a pattern could match
|
|
473
|
+
* and the fast path stays on. */
|
|
474
|
+
export const lowerForTerms=(text)=>(FOLDS_ONTO_ASCII.test(text)?text.replace(/\u0130/g,'i').replace(/\u017F/g,'s').replace(/\u212A/g,'k'):text).toLowerCase();
|
|
475
|
+
/** A term in each Unicode spelling it may be written in (as configured, composed NFC and
|
|
476
|
+
* decomposed NFD), so `café` listed one way still matches text written the other way. */
|
|
477
|
+
export const termSpellings=(t)=>[...new Set([t,t.normalize('NFC'),t.normalize('NFD')])];
|
|
478
|
+
/** The lowercased words the scrubber looks for before it runs a term's patterns, or null for
|
|
479
|
+
* a term with any character outside plain ASCII, which always runs. It's safe because
|
|
480
|
+
* anyCase spells each plain English letter as a class of its two cases, so no other
|
|
481
|
+
* character can match one. */
|
|
482
|
+
export const skipWords=(t)=>/^[\x00-\x7f]*$/.test(t)?termWords(t).map((w)=>w.toLowerCase()):null;
|
|
483
|
+
/** termMatchers(terms, names = []): `names` are the person names among `terms`. */
|
|
484
|
+
export function termMatchers(terms,names=[]){const people=new Set(names.filter((t)=>typeof t==='string').flatMap(termSpellings).map((t)=>t.trim().toLowerCase()));return terms.filter((t)=>typeof t==='string'&&t.trim()).flatMap(termSpellings).sort((x,y)=>y.trim().length-x.trim().length).flatMap((t)=>{const words=termWords(t);const cs=[...t.trim()];const word=new RegExp(`${edge(cs[0],'start')}${words.map(anyCase).join('(?:\\s+|[-_.]?)')}${edge(cs.at(-1),'end')}`,'gu');if(t.includes('.')||(t.match(/\p{L}/gu)??[]).length<4)return [word];const glued=words.map(anyCase).join('[-_]?');const part=`(?<!${PART})(?=${PART}*?${START_EDGE}${glued})(?=(${PART}+))\\1`;const dotless=across(new RegExp(`${AFTER_SCHEME}${part}(?!\\.[\\p{L}\\p{N}])`,'gu'));if(people.has(t.trim().toLowerCase()))return [across(new RegExp(`(?:${AFTER_SCHEME}|(?<=@)|(?<=(?<![\\p{L}\\p{N}_-])www\\.))${part}(?=(?:\\.${PIECE}){0,8}${ENDING})`,'gu')),across(new RegExp(`${part}(?=(?:\\.${PIECE}){0,8}${WEB_ENDING})`,'gu')),dotless,word];return [across(new RegExp(`${part}(?=(?:\\.${PIECE}){0,8}${ENDING})`,'gu')),dotless,word];});}
|
|
485
|
+
|
|
486
|
+
// --- fields read after the scrubbers' own rules ---------------------------------------
|
|
487
|
+
//
|
|
488
|
+
// Three more spellings of a secret field: a Java or Gradle property (`-Dapi.key=…`,
|
|
489
|
+
// `-Psigning.password=…`), a dotted key whose secret name runs across its last dot
|
|
490
|
+
// (`api.key=…`, `secret.key: …`), and a value after ":" whose opening single quote never
|
|
491
|
+
// closes on its line (`password: '…`).
|
|
492
|
+
//
|
|
493
|
+
// They are read in a pass of their own, over the text a scrubber has finished with, and never
|
|
494
|
+
// inside its rules. The pass can only put more placeholders down: a placeholder already there
|
|
495
|
+
// is opaque to it (never read as a name, never removed, split or moved), and no character
|
|
496
|
+
// outside the spans it hides changes. So whatever the rules above hide stays hidden, whatever
|
|
497
|
+
// this pass reads, and nothing it does can change what those rules saw.
|
|
498
|
+
|
|
499
|
+
const LATER_MASK = String.fromCharCode(0xe000);
|
|
500
|
+
const DOT_KEY_ALL_RE = new RegExp(SECRET_KEY_RE.source, 'gi');
|
|
501
|
+
/** True when a secret's name runs across the last dot of a dotted word (`api.key`,
|
|
502
|
+
* `secret.key`, `aws.secret.access.key`, `x.pwd`) and its last part alone doesn't name one
|
|
503
|
+
* (`db.password`, which the rules above read by its last part). `password.length` and
|
|
504
|
+
* `auth.ts` are a property and a file name. Only the last two parts are read. */
|
|
505
|
+
function secretAcrossDot(name) {
|
|
506
|
+
const cut = name.lastIndexOf('.');
|
|
507
|
+
const last = name.slice(cut + 1);
|
|
508
|
+
if (cut < 0 || keyIsSecret(last)) return false;
|
|
509
|
+
const head = normalKey(name.slice(name.lastIndexOf('.', cut - 1) + 1, cut));
|
|
510
|
+
const joined = `${head}_${normalKey(last)}`;
|
|
511
|
+
DOT_KEY_ALL_RE.lastIndex = 0;
|
|
512
|
+
for (let m = DOT_KEY_ALL_RE.exec(joined); m; m = DOT_KEY_ALL_RE.exec(joined)) {
|
|
513
|
+
if (m.index + m[0].length > head.length + 1) return true;
|
|
514
|
+
}
|
|
515
|
+
return false;
|
|
516
|
+
}
|
|
517
|
+
/** Whether a record's own key names a secret, for deepRedact and the view's leak count: a key
|
|
518
|
+
* the rules above call sensitive, or a dotted one whose secret name runs across its last dot
|
|
519
|
+
* (`api.key`). */
|
|
520
|
+
export const recordKeyIsSecret = (key) => keyIsSecret(key) || secretAcrossDot(key);
|
|
521
|
+
|
|
522
|
+
/** The spans of `s` the later pass hides, as [start, end) pairs in UTF-16 units, in order and
|
|
523
|
+
* apart. `s` is a scrubber's finished text. Each identifier is read once from its front and a
|
|
524
|
+
* value read is skipped past, as in secretFields, so the pass is linear. A value is read the
|
|
525
|
+
* way secretFields reads one (readValue), over a copy of the text with every placeholder
|
|
526
|
+
* masked, and the placeholders inside it are left out of what is hidden. */
|
|
527
|
+
export function laterFieldSpans(s) {
|
|
528
|
+
const marks = [];
|
|
529
|
+
const masked = s.replace(ANY_PLACEHOLDER_RE, (m, at) => {
|
|
530
|
+
marks.push([at, at + m.length]);
|
|
531
|
+
return LATER_MASK.repeat(m.length);
|
|
532
|
+
});
|
|
533
|
+
const spans = [];
|
|
534
|
+
let mark = 0;
|
|
535
|
+
// The parts of a value from `start` to `end` that aren't placeholders, trimmed of spaces.
|
|
536
|
+
const keep = (start, end) => {
|
|
537
|
+
while (mark < marks.length && marks[mark][1] <= start) mark += 1;
|
|
538
|
+
let from = start;
|
|
539
|
+
for (let k = mark; from < end; k += 1) {
|
|
540
|
+
const to = k < marks.length && marks[k][0] < end ? marks[k][0] : end;
|
|
541
|
+
let a = from;
|
|
542
|
+
let b = to;
|
|
543
|
+
while (a < b && /\s/.test(s[a])) a += 1;
|
|
544
|
+
while (b > a && /\s/.test(s[b - 1])) b -= 1;
|
|
545
|
+
if (b > a) spans.push([a, b]);
|
|
546
|
+
if (to === end) break;
|
|
547
|
+
from = Math.min(end, marks[k][1]);
|
|
548
|
+
}
|
|
549
|
+
};
|
|
550
|
+
const ids = new RegExp(IDENTIFIER_RE.source, 'g');
|
|
551
|
+
let nlAt = -1;
|
|
552
|
+
const lineEnd = (p) => {
|
|
553
|
+
if (nlAt < p) {
|
|
554
|
+
nlAt = masked.indexOf('\n', p);
|
|
555
|
+
if (nlAt < 0) nlAt = masked.length;
|
|
556
|
+
}
|
|
557
|
+
return nlAt;
|
|
558
|
+
};
|
|
559
|
+
// Where the last dotted word read ends: an identifier before it is one of its later parts.
|
|
560
|
+
let dottedEnd = 0;
|
|
561
|
+
for (let m = ids.exec(masked); m; m = ids.exec(masked)) {
|
|
562
|
+
const flag = masked[m.index - 1] === '-';
|
|
563
|
+
let rest = '';
|
|
564
|
+
if (flag || m.index >= dottedEnd) {
|
|
565
|
+
DOTTED_NAME_RE.lastIndex = m.index + m[0].length;
|
|
566
|
+
rest = DOTTED_NAME_RE.exec(masked)?.[0] ?? '';
|
|
567
|
+
}
|
|
568
|
+
const whole = m[0] + rest;
|
|
569
|
+
const wholeEnd = m.index + whole.length;
|
|
570
|
+
if (wholeEnd > dottedEnd) dottedEnd = wholeEnd;
|
|
571
|
+
// A name only this pass reads (`added`), or one the rules above read too: for that one,
|
|
572
|
+
// only a value they left because its single quote never closes is read here.
|
|
573
|
+
let name = m[0];
|
|
574
|
+
let end = m.index + m[0].length;
|
|
575
|
+
let added = false;
|
|
576
|
+
if (flag) {
|
|
577
|
+
const dotted = whole.replace(/\./g, '_');
|
|
578
|
+
const property = masked[m.index - 2] !== '-' && /^[DP][A-Za-z0-9_]/.test(whole) && masked[wholeEnd] === '=';
|
|
579
|
+
if (keyIsSecret(dotted)) [name, end] = [dotted, wholeEnd];
|
|
580
|
+
else if (property && keyIsSecret(dotted.slice(1))) [name, end, added] = [dotted.slice(1), wholeEnd, true];
|
|
581
|
+
else continue;
|
|
582
|
+
} else if (rest && secretAcrossDot(whole)) [name, end, added] = [whole.replace(/\./g, '_'), wholeEnd, true];
|
|
583
|
+
else if (!keyIsSecret(name)) continue;
|
|
584
|
+
FIELD_SEP_RE.lastIndex = end;
|
|
585
|
+
const sep = FIELD_SEP_RE.exec(masked);
|
|
586
|
+
if (!sep) continue;
|
|
587
|
+
const p = FIELD_SEP_RE.lastIndex;
|
|
588
|
+
const key = normalKey(name);
|
|
589
|
+
if (sep[1] === ':' && TYPE_AHEAD_RE.test(masked.slice(p, p + 24)) && !WHOLE_LINE_KEY_RE.test(key)) {
|
|
590
|
+
const t = readValue(masked, p, 'tight', lineEnd);
|
|
591
|
+
if (t && TYPE_ONLY_RE.test(masked.slice(t.start, t.end))) continue;
|
|
592
|
+
}
|
|
593
|
+
const mode = WHOLE_LINE_KEY_RE.test(key) || (sep[1] === ':' && PASSWORD_KEY_RE.test(key)) ? 'line' : sep[1] === ':' ? 'tight' : 'word';
|
|
594
|
+
let v = readValue(masked, p, mode, lineEnd);
|
|
595
|
+
// The rules above read this value themselves, and hid it or left it for a reason.
|
|
596
|
+
if (v && !added) continue;
|
|
597
|
+
// After ":", a single quote that opens no closed value: read on from after it, the way the
|
|
598
|
+
// value would be read without it. A quote followed by a space, the end, or , ; ) ] } . +
|
|
599
|
+
// closes the text the key sits in (`'Password:',`), and opens no value.
|
|
600
|
+
if (!v && sep[1] === ':' && masked[p] === "'" && /[^\s,;)\]}.+]/.test(masked[p + 1] ?? ' ')) v = readValue(masked, p + 1, mode, lineEnd);
|
|
601
|
+
if (!v) continue;
|
|
602
|
+
const text = s.slice(v.start, v.end);
|
|
603
|
+
if (!notASecret(name, text) && !markupOnly(text)) keep(v.start, v.end);
|
|
604
|
+
ids.lastIndex = Math.max(v.after, v.end, ids.lastIndex);
|
|
605
|
+
}
|
|
606
|
+
return spans;
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
/** `s` with the later pass's spans replaced by `placeholder`; `counted()` is called once per span. */
|
|
610
|
+
export function hideLaterFields(s, placeholder, counted = () => {}) {
|
|
611
|
+
const spans = laterFieldSpans(s);
|
|
612
|
+
if (!spans.length) return s;
|
|
613
|
+
let out = '';
|
|
614
|
+
let at = 0;
|
|
615
|
+
for (const [start, end] of spans) {
|
|
616
|
+
counted();
|
|
617
|
+
out += s.slice(at, start) + placeholder;
|
|
618
|
+
at = end;
|
|
619
|
+
}
|
|
620
|
+
return out + s.slice(at);
|
|
621
|
+
}
|
|
622
|
+
|
|
623
|
+
// --- fields read from the text as it was written ---------------------------------------
|
|
624
|
+
//
|
|
625
|
+
// The rules above read a field's value and move on past it. When that value is itself a name
|
|
626
|
+
// (`api_key=token: …`, `password="token": "…"`, `token: "Ab3d, api.key: "…"`), they hide the
|
|
627
|
+
// name and the real value after it is left. And the KEY=VALUE rule reads a key only where an
|
|
628
|
+
// earlier `name=` hasn't taken it for a value (`x=` on one line, `apiKey` on the next, `=…`
|
|
629
|
+
// on the third).
|
|
630
|
+
//
|
|
631
|
+
// Those values are found here, in the text as it was written: by then the scrubbers have put a
|
|
632
|
+
// placeholder where the inner name was, so their finished text no longer says there was one.
|
|
633
|
+
// What is found is hidden in addition to everything the scrubbers hide. hideSourceFields takes
|
|
634
|
+
// a scrubber's finished text and only swaps more of its shown characters for a placeholder,
|
|
635
|
+
// and the audit only adds spans between the ones it has. So nothing the rules above hide can
|
|
636
|
+
// show again, whatever is read here.
|
|
637
|
+
|
|
638
|
+
// What the KEY=VALUE rule reads after its key, from a fixed start.
|
|
639
|
+
const KV_REST_RE = /\s*=\s*("[^"]*"|'[^']*'|\S+)/y;
|
|
640
|
+
// A word every sensitive key holds: each name SECRET_KEY_RE matches has one of these in it, and
|
|
641
|
+
// normalKey only adds "_" to a name, so a text with none of them has no sensitive key at all.
|
|
642
|
+
const SECRET_NAME_HINT_RE = /KEY|PASS|AUTH|COOKIE|CREDENTIAL|SIG|TOKEN|SECRET|PWD/i;
|
|
643
|
+
const TAIL_SEP_CHAR = /[\\"' \t:=?]/;
|
|
644
|
+
const NAME_CHAR = /[A-Za-z0-9_.-]/;
|
|
645
|
+
// What is trimmed from each end of a piece hidden here: spaces, and quotes with their escapes.
|
|
646
|
+
const EDGE_CHAR = /[\s\\"']/;
|
|
647
|
+
|
|
648
|
+
/** The spans of `s`, a text as it was written, that hold a value the rules above leave shown
|
|
649
|
+
* because of what sits before it, as [start, end) pairs in UTF-16 units, in order and apart:
|
|
650
|
+
* - the value after a name that ends a secret field's own value, or is all of it in quotes
|
|
651
|
+
* (`api_key=token: …`, `password="token": "…"`, `token: "see api.key: "…"`), any number of
|
|
652
|
+
* names deep. The inner name must name a secret itself, or be one quoted word;
|
|
653
|
+
* - the value of a KEY=VALUE key split from its "=" or its value by a line break, which the
|
|
654
|
+
* rule misses when an earlier `name=` took the key for its value.
|
|
655
|
+
* Placeholders already in `s` are masked and left out, as in laterFieldSpans. A field is read
|
|
656
|
+
* the way secretFields reads it and is skipped past, and each inner value starts at or after
|
|
657
|
+
* the end of the one before it, so the pass is linear. */
|
|
658
|
+
export function sourceFieldSpans(s) {
|
|
659
|
+
// Every span here follows a sensitive key. A text that names none is done in one search,
|
|
660
|
+
// so the audit, which reads its source and then its rendition, pays for neither.
|
|
661
|
+
if (!SECRET_NAME_HINT_RE.test(s)) return [];
|
|
662
|
+
const marks = [];
|
|
663
|
+
const masked = s.replace(ANY_PLACEHOLDER_RE, (m, at) => {
|
|
664
|
+
marks.push([at, at + m.length]);
|
|
665
|
+
return LATER_MASK.repeat(m.length);
|
|
666
|
+
});
|
|
667
|
+
const spans = [];
|
|
668
|
+
let mark = 0;
|
|
669
|
+
// The parts of a value from `start` to `end` that aren't placeholders, trimmed of spaces and of
|
|
670
|
+
// quotes: a quote is where the rules stop reading a line, and one left in place stops a second
|
|
671
|
+
// pass where it stopped the first. What is left of a value next to a placeholder inside it is
|
|
672
|
+
// kept only when it holds a letter or a digit (`":` between two placeholders isn't a value).
|
|
673
|
+
const keep = (start, end) => {
|
|
674
|
+
while (mark < marks.length && marks[mark][1] <= start) mark += 1;
|
|
675
|
+
let from = start;
|
|
676
|
+
for (let k = mark; from < end; k += 1) {
|
|
677
|
+
const to = k < marks.length && marks[k][0] < end ? marks[k][0] : end;
|
|
678
|
+
let a = from;
|
|
679
|
+
let b = to;
|
|
680
|
+
while (a < b && EDGE_CHAR.test(s[a])) a += 1;
|
|
681
|
+
while (b > a && EDGE_CHAR.test(s[b - 1])) b -= 1;
|
|
682
|
+
if (b > a && ((from === start && to === end) || /[\p{L}\p{N}]/u.test(s.slice(a, b)))) spans.push([a, b]);
|
|
683
|
+
if (to === end) break;
|
|
684
|
+
from = Math.min(end, marks[k][1]);
|
|
685
|
+
}
|
|
686
|
+
};
|
|
687
|
+
let nlAt = -1;
|
|
688
|
+
const lineEnd = (p) => {
|
|
689
|
+
if (nlAt < p) {
|
|
690
|
+
nlAt = masked.indexOf('\n', p);
|
|
691
|
+
if (nlAt < 0) nlAt = masked.length;
|
|
692
|
+
}
|
|
693
|
+
return nlAt;
|
|
694
|
+
};
|
|
695
|
+
// The value of the field whose name ends at `end`, read as secretFields reads it.
|
|
696
|
+
const valueOf = (name, end, flag) => {
|
|
697
|
+
const key = normalKey(name);
|
|
698
|
+
FIELD_SEP_RE.lastIndex = end;
|
|
699
|
+
const sep = FIELD_SEP_RE.exec(masked);
|
|
700
|
+
if (!sep) {
|
|
701
|
+
if (!flag) return null;
|
|
702
|
+
FLAG_SEP_RE.lastIndex = end;
|
|
703
|
+
return FLAG_SEP_RE.exec(masked) ? readValue(masked, FLAG_SEP_RE.lastIndex, 'word', lineEnd) : null;
|
|
704
|
+
}
|
|
705
|
+
const p = FIELD_SEP_RE.lastIndex;
|
|
706
|
+
const wholeLine = WHOLE_LINE_KEY_RE.test(key);
|
|
707
|
+
if (sep[1] === ':' && !wholeLine && TYPE_AHEAD_RE.test(masked.slice(p, p + 24))) {
|
|
708
|
+
const t = readValue(masked, p, 'tight', lineEnd);
|
|
709
|
+
if (t && TYPE_ONLY_RE.test(masked.slice(t.start, t.end))) return null;
|
|
710
|
+
}
|
|
711
|
+
const mode = wholeLine || (sep[1] === ':' && PASSWORD_KEY_RE.test(key)) ? 'line' : sep[1] === ':' ? 'tight' : 'word';
|
|
712
|
+
let v = readValue(masked, p, mode, lineEnd);
|
|
713
|
+
// A single quote that never closes, read the way laterFieldSpans reads it.
|
|
714
|
+
if (!v && sep[1] === ':' && masked[p] === "'" && /[^\s,;)\]}.+]/.test(masked[p + 1] ?? ' ')) v = readValue(masked, p + 1, mode, lineEnd);
|
|
715
|
+
return v && { ...v, line: mode === 'line' };
|
|
716
|
+
};
|
|
717
|
+
// The name a value ends with, when its separator runs to the value's end or past it: the
|
|
718
|
+
// value was a name, and the real one comes next. Null when there is none.
|
|
719
|
+
const innerName = (v) => {
|
|
720
|
+
let ne = v.end;
|
|
721
|
+
while (ne > v.start && TAIL_SEP_CHAR.test(masked[ne - 1])) ne -= 1;
|
|
722
|
+
let ns = ne;
|
|
723
|
+
while (ns > v.start && ne - ns <= 96 && NAME_CHAR.test(masked[ns - 1])) ns -= 1;
|
|
724
|
+
if (ne - ns > 96) return null;
|
|
725
|
+
while (ns < ne && /[.-]/.test(masked[ns])) ns += 1;
|
|
726
|
+
if (ns === ne) return null;
|
|
727
|
+
FIELD_SEP_RE.lastIndex = ne;
|
|
728
|
+
if (!FIELD_SEP_RE.exec(masked) || FIELD_SEP_RE.lastIndex < v.end) return null;
|
|
729
|
+
const name = masked.slice(ns, ne);
|
|
730
|
+
const quotedWord = ns === v.start && ne === v.end && /["']/.test(masked[ns - 1] ?? '') && /["'\\]/.test(masked[ne] ?? '');
|
|
731
|
+
const secret = keyIsSecret(name.slice(name.lastIndexOf('.') + 1)) || secretAcrossDot(name) || (masked[ns - 1] === '-' && keyIsSecret(name.replace(/\./g, '_')));
|
|
732
|
+
return secret || quotedWord ? { name: name.replace(/\./g, '_'), end: ne } : null;
|
|
733
|
+
};
|
|
734
|
+
const ids = new RegExp(IDENTIFIER_RE.source, 'g');
|
|
735
|
+
// Where the last dotted word read ends: an identifier before it is one of its later parts.
|
|
736
|
+
let dottedEnd = 0;
|
|
737
|
+
// Where the last field's own value ends.
|
|
738
|
+
let ownEnd = 0;
|
|
739
|
+
for (let m = ids.exec(masked); m; m = ids.exec(masked)) {
|
|
740
|
+
const flag = masked[m.index - 1] === '-';
|
|
741
|
+
let rest = '';
|
|
742
|
+
if (flag || m.index >= dottedEnd) {
|
|
743
|
+
DOTTED_NAME_RE.lastIndex = m.index + m[0].length;
|
|
744
|
+
rest = DOTTED_NAME_RE.exec(masked)?.[0] ?? '';
|
|
745
|
+
}
|
|
746
|
+
const whole = m[0] + rest;
|
|
747
|
+
const wholeEnd = m.index + whole.length;
|
|
748
|
+
if (wholeEnd > dottedEnd) dottedEnd = wholeEnd;
|
|
749
|
+
let name = m[0];
|
|
750
|
+
let end = m.index + m[0].length;
|
|
751
|
+
if (flag) {
|
|
752
|
+
const dotted = whole.replace(/\./g, '_');
|
|
753
|
+
const property = masked[m.index - 2] !== '-' && /^[DP][A-Za-z0-9_]/.test(whole) && masked[wholeEnd] === '=';
|
|
754
|
+
if (keyIsSecret(dotted)) [name, end] = [dotted, wholeEnd];
|
|
755
|
+
else if (property && keyIsSecret(dotted.slice(1))) [name, end] = [dotted.slice(1), wholeEnd];
|
|
756
|
+
else continue;
|
|
757
|
+
} else if (rest && secretAcrossDot(whole)) [name, end] = [whole.replace(/\./g, '_'), wholeEnd];
|
|
758
|
+
else if (!keyIsSecret(name)) continue;
|
|
759
|
+
// The field's own value is the rules' to hide; only what follows a name in it is read here.
|
|
760
|
+
// A name inside the value of the field before it isn't read as a field again (that would
|
|
761
|
+
// read a long value once per name in it); only a key split from its "=" is read there.
|
|
762
|
+
let v = m.index < ownEnd ? null : valueOf(name, end, flag);
|
|
763
|
+
const own = v;
|
|
764
|
+
if (own) ownEnd = Math.max(ownEnd, own.end);
|
|
765
|
+
if (!v) {
|
|
766
|
+
// No value on the key's own line: the KEY=VALUE rule's reading, across the line break.
|
|
767
|
+
const kvKey = m[0].slice(m[0].lastIndexOf('-') + 1);
|
|
768
|
+
if (!/^[A-Za-z_]/.test(kvKey) || !SENSITIVE_KV_KEY.test(kvKey)) continue;
|
|
769
|
+
KV_REST_RE.lastIndex = m.index + m[0].length;
|
|
770
|
+
const kv = KV_REST_RE.exec(masked);
|
|
771
|
+
if (!kv) continue;
|
|
772
|
+
const after = KV_REST_RE.lastIndex;
|
|
773
|
+
let from = after - kv[1].length;
|
|
774
|
+
let to = after;
|
|
775
|
+
// The rest of an == or === comparison: its operand.
|
|
776
|
+
while (from < to && masked[from] === '=') from += 1;
|
|
777
|
+
if (to - from >= 2 && '"\''.includes(masked[from]) && masked[to - 1] === masked[from]) [from, to] = [from + 1, to - 1];
|
|
778
|
+
if (to <= from) continue;
|
|
779
|
+
keep(from, to);
|
|
780
|
+
v = { start: from, end: to, after };
|
|
781
|
+
}
|
|
782
|
+
let hidden = 0;
|
|
783
|
+
for (let inner = innerName(v); ; inner = innerName(v)) {
|
|
784
|
+
// A value hidden here is skipped past; the field's own is read on through.
|
|
785
|
+
if (v !== own) ids.lastIndex = Math.max(v.after, v.end, ids.lastIndex);
|
|
786
|
+
if (!inner) break;
|
|
787
|
+
const next = valueOf(inner.name, inner.end, false);
|
|
788
|
+
if (!next) break;
|
|
789
|
+
v = next;
|
|
790
|
+
const text = s.slice(v.start, v.end);
|
|
791
|
+
if (notASecret(inner.name, text) || markupOnly(text)) continue;
|
|
792
|
+
keep(v.start, v.end);
|
|
793
|
+
hidden += 1;
|
|
794
|
+
}
|
|
795
|
+
// A password line or a header hides to the end of its line, or to the next `name:` on it.
|
|
796
|
+
// With that name's value now hidden, the rest of the line is the field's own, as a second
|
|
797
|
+
// pass would read it.
|
|
798
|
+
if (own?.line && hidden && !/["'\\]/.test(masked[v.end] ?? '"')) {
|
|
799
|
+
const more = readValue(masked, v.end, 'line', lineEnd);
|
|
800
|
+
if (more) {
|
|
801
|
+
keep(more.start, more.end);
|
|
802
|
+
ids.lastIndex = Math.max(more.after, more.end, ids.lastIndex);
|
|
803
|
+
}
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
return spans;
|
|
807
|
+
}
|
|
808
|
+
|
|
809
|
+
/** Where each run of text between the placeholders of `redacted` sits in `original`, the text
|
|
810
|
+
* it was made from: { at, length } in `redacted`, and the earliest and latest place it can
|
|
811
|
+
* start in `original` (`lo`, `hi`), found from the front and from the back. A scrubber's
|
|
812
|
+
* finished text is its input with spans swapped for placeholders, so the runs appear in the
|
|
813
|
+
* input in order, the first at its start and the last at its end. A run whose `lo` and `hi`
|
|
814
|
+
* agree sits exactly there. The KEY=VALUE rule writes its own "=" before its placeholder
|
|
815
|
+
* whatever spacing the input had, so a run's "=" right before a placeholder isn't matched
|
|
816
|
+
* (`equals` marks a run that has one).
|
|
817
|
+
* Null when the runs don't fit `original` that way. */
|
|
818
|
+
export function alignRedacted(original, redacted) {
|
|
819
|
+
const chunks = [];
|
|
820
|
+
let at = 0;
|
|
821
|
+
for (const m of redacted.matchAll(ANY_PLACEHOLDER_RE)) {
|
|
822
|
+
const equals = m.index > at && redacted[m.index - 1] === '=';
|
|
823
|
+
chunks.push({ at, length: m.index - at - (equals ? 1 : 0), equals, lo: 0, hi: 0 });
|
|
824
|
+
at = m.index + m[0].length;
|
|
825
|
+
}
|
|
826
|
+
chunks.push({ at, length: redacted.length - at, equals: false, lo: 0, hi: 0 });
|
|
827
|
+
const text = (c) => redacted.slice(c.at, c.at + c.length);
|
|
828
|
+
const first = chunks[0];
|
|
829
|
+
const last = chunks[chunks.length - 1];
|
|
830
|
+
if (!original.startsWith(text(first))) return null;
|
|
831
|
+
if (chunks.length === 1) return original.length === first.length ? chunks : null;
|
|
832
|
+
last.lo = original.length - last.length;
|
|
833
|
+
last.hi = last.lo;
|
|
834
|
+
if (last.lo < first.length || !original.endsWith(text(last))) return null;
|
|
835
|
+
let pos = first.length;
|
|
836
|
+
for (let i = 1; i < chunks.length - 1; i += 1) {
|
|
837
|
+
const c = chunks[i];
|
|
838
|
+
c.lo = original.indexOf(text(c), pos);
|
|
839
|
+
if (c.lo < 0 || c.lo + c.length > last.lo) return null;
|
|
840
|
+
pos = c.lo + c.length;
|
|
841
|
+
}
|
|
842
|
+
let limit = last.lo;
|
|
843
|
+
for (let i = chunks.length - 2; i > 0; i -= 1) {
|
|
844
|
+
const c = chunks[i];
|
|
845
|
+
c.hi = original.lastIndexOf(text(c), limit - c.length);
|
|
846
|
+
limit = c.hi;
|
|
847
|
+
}
|
|
848
|
+
return chunks;
|
|
849
|
+
}
|
|
850
|
+
|
|
851
|
+
/** `redacted`, a scrubber's finished text for `original`, with what sourceFieldSpans finds in
|
|
852
|
+
* `original` hidden too. Only characters `redacted` still shows are replaced, each run by
|
|
853
|
+
* `placeholder`; no placeholder in it is touched. A run of text that could sit in more than one
|
|
854
|
+
* place in `original` is hidden whole when a span reaches any of those places, and so is every
|
|
855
|
+
* run when the two texts don't line up, so a value found is never left for want of its place.
|
|
856
|
+
* `counted()` is called once per run hidden. */
|
|
857
|
+
export function hideSourceFields(original, redacted, placeholder, counted = () => {}) {
|
|
858
|
+
const spans = sourceFieldSpans(original);
|
|
859
|
+
if (!spans.length) return redacted;
|
|
860
|
+
const chunks = alignRedacted(original, redacted);
|
|
861
|
+
const cuts = [];
|
|
862
|
+
if (!chunks) {
|
|
863
|
+
let at = 0;
|
|
864
|
+
for (const m of redacted.matchAll(ANY_PLACEHOLDER_RE)) {
|
|
865
|
+
cuts.push([at, m.index]);
|
|
866
|
+
at = m.index + m[0].length;
|
|
867
|
+
}
|
|
868
|
+
cuts.push([at, redacted.length]);
|
|
869
|
+
} else {
|
|
870
|
+
let k = 0;
|
|
871
|
+
for (const [a, b] of spans) {
|
|
872
|
+
while (k < chunks.length && chunks[k].hi + chunks[k].length <= a) k += 1;
|
|
873
|
+
for (let j = k; j < chunks.length && chunks[j].lo < b; j += 1) {
|
|
874
|
+
const c = chunks[j];
|
|
875
|
+
if (c.lo === c.hi) {
|
|
876
|
+
const from = Math.max(a, c.lo);
|
|
877
|
+
const to = Math.min(b, c.lo + c.length);
|
|
878
|
+
if (to <= from) continue;
|
|
879
|
+
// The run's own "=" goes too when the span runs on over the "=" in `original`.
|
|
880
|
+
let q = to;
|
|
881
|
+
if (c.equals && to === c.lo + c.length) while (q < b && /\s/.test(original[q])) q += 1;
|
|
882
|
+
const equals = c.equals && to === c.lo + c.length && q < b && original[q] === '=';
|
|
883
|
+
cuts.push([c.at + from - c.lo, c.at + to - c.lo + (equals ? 1 : 0)]);
|
|
884
|
+
} else if (c.length && a < c.hi + c.length) cuts.push([c.at, c.at + c.length]);
|
|
885
|
+
}
|
|
886
|
+
}
|
|
887
|
+
}
|
|
888
|
+
let out = '';
|
|
889
|
+
let at = 0;
|
|
890
|
+
for (const cut of cuts) {
|
|
891
|
+
let start = Math.max(cut[0], at);
|
|
892
|
+
let end = cut[1];
|
|
893
|
+
while (start < end && /\s/.test(redacted[start])) start += 1;
|
|
894
|
+
while (end > start && /\s/.test(redacted[end - 1])) end -= 1;
|
|
895
|
+
if (end <= start) continue;
|
|
896
|
+
counted();
|
|
897
|
+
out += redacted.slice(at, start) + placeholder;
|
|
898
|
+
at = end;
|
|
899
|
+
}
|
|
900
|
+
return out + redacted.slice(at);
|
|
901
|
+
}
|
|
902
|
+
|
|
903
|
+
// --- last, over the finished text ------------------------------------------------------
|
|
904
|
+
//
|
|
905
|
+
// Two things the rules above leave, read once a scrubber is otherwise done, so they can only
|
|
906
|
+
// ever hide more: no character outside the spans they hide changes, and no placeholder moves.
|
|
907
|
+
// - A header (Authorization, Cookie) whose value is only placeholders and stops at a quote
|
|
908
|
+
// with text glued after it (`Authorization=[redacted:secret]'xk9…`, issue 80): the rest of
|
|
909
|
+
// the line, from the quote, is the header's too.
|
|
910
|
+
// - A `/root/…` home path (ROOT_PATH_SOURCE, issue 85), each part between placeholders.
|
|
911
|
+
|
|
912
|
+
const ROOT_PATH_RE = new RegExp(ROOT_PATH_SOURCE, 'g');
|
|
913
|
+
const onlyMasks = (text) => text.includes(LATER_MASK) && text.replace(/\s/g, '').split(LATER_MASK).join('') === '';
|
|
914
|
+
|
|
915
|
+
/** The spans of `s` hidden last, as [start, end, kind) triples in UTF-16 units, in order and
|
|
916
|
+
* apart: kind is 'secret' for a header's glued text and 'path' for a `/root/…` path, left out
|
|
917
|
+
* when `paths` is false. `s` is a scrubber's finished text; spans hold no placeholder. */
|
|
918
|
+
export function finishedSpans(s, { paths = true } = {}) {
|
|
919
|
+
if (!s.includes('/root/') && !SECRET_NAME_HINT_RE.test(s)) return [];
|
|
920
|
+
const marks = [];
|
|
921
|
+
const masked = s.replace(ANY_PLACEHOLDER_RE, (m, at) => {
|
|
922
|
+
marks.push([at, at + m.length]);
|
|
923
|
+
return LATER_MASK.repeat(m.length);
|
|
924
|
+
});
|
|
925
|
+
// The parts of [start, end) between placeholders, trimmed of spaces, as spans of `kind`.
|
|
926
|
+
const pieces = (start, end, kind, out) => {
|
|
927
|
+
let k = 0;
|
|
928
|
+
while (k < marks.length && marks[k][1] <= start) k += 1;
|
|
929
|
+
for (let from = start; from < end; k += 1) {
|
|
930
|
+
const to = k < marks.length && marks[k][0] < end ? marks[k][0] : end;
|
|
931
|
+
let a = from;
|
|
932
|
+
let b = to;
|
|
933
|
+
while (a < b && /\s/.test(s[a])) a += 1;
|
|
934
|
+
while (b > a && /\s/.test(s[b - 1])) b -= 1;
|
|
935
|
+
if (b > a) out.push([a, b, kind]);
|
|
936
|
+
if (to === end) break;
|
|
937
|
+
from = Math.min(end, marks[k][1]);
|
|
938
|
+
}
|
|
939
|
+
};
|
|
940
|
+
const headers = [];
|
|
941
|
+
let nlAt = -1;
|
|
942
|
+
const lineEnd = (p) => {
|
|
943
|
+
if (nlAt < p) {
|
|
944
|
+
nlAt = masked.indexOf('\n', p);
|
|
945
|
+
if (nlAt < 0) nlAt = masked.length;
|
|
946
|
+
}
|
|
947
|
+
return nlAt;
|
|
948
|
+
};
|
|
949
|
+
const ids = new RegExp(IDENTIFIER_RE.source, 'g');
|
|
950
|
+
for (let m = ids.exec(masked); m; m = ids.exec(masked)) {
|
|
951
|
+
if (!WHOLE_LINE_KEY_RE.test(normalKey(m[0])) || !keyIsSecret(m[0])) continue;
|
|
952
|
+
FIELD_SEP_RE.lastIndex = m.index + m[0].length;
|
|
953
|
+
if (!FIELD_SEP_RE.exec(masked)) continue;
|
|
954
|
+
const v = readValue(masked, FIELD_SEP_RE.lastIndex, 'line', lineEnd);
|
|
955
|
+
if (!v || !onlyMasks(masked.slice(v.start, v.end))) continue;
|
|
956
|
+
// What it hides can end at another quote with text glued after it: that's the header's too.
|
|
957
|
+
let end = v.end;
|
|
958
|
+
for (let more; `"'`.includes(masked[end] ?? '') && /\w/.test(masked[end + 1] ?? '') && (more = readValue(masked, end + 1, 'line', lineEnd)); end = more.end) {
|
|
959
|
+
pieces(end, more.end, 'secret', headers);
|
|
960
|
+
ids.lastIndex = Math.max(more.after, more.end);
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
const rootPaths = [];
|
|
964
|
+
if (paths) {
|
|
965
|
+
for (const r of masked.matchAll(ROOT_PATH_RE)) pieces(r.index, pathMatchEnd(masked, r.index, r.index + r[0].length), 'path', rootPaths);
|
|
966
|
+
}
|
|
967
|
+
// Spans that overlap become one, a path only when both were.
|
|
968
|
+
const out = [];
|
|
969
|
+
for (const span of [...headers, ...rootPaths].sort((x, y) => x[0] - y[0])) {
|
|
970
|
+
const last = out[out.length - 1];
|
|
971
|
+
if (last && span[0] < last[1]) {
|
|
972
|
+
last[1] = Math.max(last[1], span[1]);
|
|
973
|
+
if (span[2] !== 'path') last[2] = 'secret';
|
|
974
|
+
} else out.push([...span]);
|
|
975
|
+
}
|
|
976
|
+
return out;
|
|
977
|
+
}
|
|
978
|
+
|
|
979
|
+
/** `s` with its finishedSpans replaced: a header's glued text by `placeholder`, a path by
|
|
980
|
+
* `pathPlaceholder`, or left shown when that's null. `counted()` is called once per span. */
|
|
981
|
+
export function hideFinishedSpans(s, placeholder, counted = () => {}, pathPlaceholder = '[redacted:path]') {
|
|
982
|
+
if (typeof s !== 'string' || s.length === 0) return s;
|
|
983
|
+
const spans = finishedSpans(s, { paths: pathPlaceholder != null });
|
|
984
|
+
if (!spans.length) return s;
|
|
985
|
+
let out = '';
|
|
986
|
+
let at = 0;
|
|
987
|
+
for (const [start, end, kind] of spans) {
|
|
988
|
+
counted();
|
|
989
|
+
out += s.slice(at, start) + (kind === 'path' ? pathPlaceholder : placeholder);
|
|
990
|
+
at = end;
|
|
991
|
+
}
|
|
992
|
+
return out + s.slice(at);
|
|
993
|
+
}
|