honestweek 0.0.0-stage → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. package/.claude-plugin/marketplace.json +17 -0
  2. package/.claude-plugin/plugin.json +9 -0
  3. package/LICENSE +21 -0
  4. package/README.md +661 -2
  5. package/SKILL.md +129 -0
  6. package/bin/honestweek.mjs +240 -0
  7. package/honestweek.config.example.json +76 -0
  8. package/lib/archive.mjs +63 -0
  9. package/lib/atomic-json.mjs +42 -0
  10. package/lib/badges.mjs +43 -0
  11. package/lib/bounded-jsonl.mjs +39 -0
  12. package/lib/build.mjs +680 -0
  13. package/lib/carry-receipts.mjs +102 -0
  14. package/lib/carry-recovery.mjs +19 -0
  15. package/lib/claude-adapter.mjs +602 -0
  16. package/lib/client.mjs +397 -0
  17. package/lib/codex-records.mjs +167 -0
  18. package/lib/config.mjs +511 -0
  19. package/lib/curation-similarity.mjs +20 -0
  20. package/lib/demo/content.mjs +262 -0
  21. package/lib/demo/extra.mjs +1061 -0
  22. package/lib/demo/repo.mjs +385 -0
  23. package/lib/demo/week.mjs +2558 -0
  24. package/lib/digest-carry.mjs +538 -0
  25. package/lib/digest-curation.mjs +548 -0
  26. package/lib/digest-evidence.mjs +348 -0
  27. package/lib/digest-lifecycle.mjs +142 -0
  28. package/lib/digest-schema.mjs +58 -0
  29. package/lib/digest-source-bound.mjs +46 -0
  30. package/lib/digest-store.mjs +485 -0
  31. package/lib/digest.mjs +437 -0
  32. package/lib/discover.mjs +193 -0
  33. package/lib/emit/_shared.mjs +122 -0
  34. package/lib/emit/changelog.mjs +62 -0
  35. package/lib/emit/client.mjs +311 -0
  36. package/lib/emit/digest.mjs +53 -0
  37. package/lib/emit/goals-page.mjs +416 -0
  38. package/lib/emit/index.mjs +155 -0
  39. package/lib/emit/page.mjs +423 -0
  40. package/lib/emit/post.mjs +22 -0
  41. package/lib/emit/report.mjs +51 -0
  42. package/lib/git.mjs +568 -0
  43. package/lib/goals.mjs +539 -0
  44. package/lib/handoffs.mjs +151 -0
  45. package/lib/harvest.mjs +162 -0
  46. package/lib/history.mjs +110 -0
  47. package/lib/init.mjs +576 -0
  48. package/lib/invocation.mjs +102 -0
  49. package/lib/loopback-host.mjs +16 -0
  50. package/lib/mine/corpus.mjs +641 -0
  51. package/lib/mine/detect.mjs +681 -0
  52. package/lib/mine/draft.mjs +262 -0
  53. package/lib/mine/ledger.mjs +189 -0
  54. package/lib/mine/rank.mjs +236 -0
  55. package/lib/mine.mjs +381 -0
  56. package/lib/preview.mjs +504 -0
  57. package/lib/private-words.mjs +31 -0
  58. package/lib/problems/catalog.json +5884 -0
  59. package/lib/problems/checks.mjs +1454 -0
  60. package/lib/problems/classify.mjs +910 -0
  61. package/lib/problems/context.mjs +448 -0
  62. package/lib/problems/drafts.mjs +57 -0
  63. package/lib/problems/fix-tests.mjs +270 -0
  64. package/lib/problems/index.mjs +332 -0
  65. package/lib/problems/scope.mjs +401 -0
  66. package/lib/prompt-adapters.mjs +229 -0
  67. package/lib/prompt-curation.mjs +71 -0
  68. package/lib/prompt-identity.mjs +18 -0
  69. package/lib/prompt-lane.mjs +84 -0
  70. package/lib/prompt-lock.mjs +23 -0
  71. package/lib/prompt-privacy.mjs +451 -0
  72. package/lib/prompt-store.mjs +111 -0
  73. package/lib/prompts.mjs +100 -0
  74. package/lib/reader.mjs +189 -0
  75. package/lib/readers/client.json +9 -0
  76. package/lib/readers/default.json +5 -0
  77. package/lib/redact.mjs +577 -0
  78. package/lib/redaction-patterns.mjs +993 -0
  79. package/lib/replay/assemble.mjs +523 -0
  80. package/lib/replay/classify.mjs +464 -0
  81. package/lib/replay/claude.mjs +937 -0
  82. package/lib/replay/codex-program.mjs +459 -0
  83. package/lib/replay/codex.mjs +908 -0
  84. package/lib/replay/evidence.mjs +78 -0
  85. package/lib/replay/goals.mjs +287 -0
  86. package/lib/replay/ids.mjs +54 -0
  87. package/lib/replay/index.mjs +690 -0
  88. package/lib/replay/jsonl.mjs +102 -0
  89. package/lib/replay/launch.mjs +164 -0
  90. package/lib/replay/lookup.mjs +374 -0
  91. package/lib/replay/metrics.mjs +51 -0
  92. package/lib/replay/outcomes.mjs +233 -0
  93. package/lib/replay/parse-common.mjs +281 -0
  94. package/lib/replay/sources.mjs +205 -0
  95. package/lib/replay/timeline.mjs +331 -0
  96. package/lib/replay/views.mjs +340 -0
  97. package/lib/repo-identity.mjs +101 -0
  98. package/lib/resolve-week.mjs +178 -0
  99. package/lib/site/adapter.mjs +168 -0
  100. package/lib/site/archive.mjs +48 -0
  101. package/lib/site/derive.mjs +443 -0
  102. package/lib/site/detect.mjs +138 -0
  103. package/lib/site/emit-site.mjs +94 -0
  104. package/lib/site/fact-fence.mjs +148 -0
  105. package/lib/site/inspect.mjs +102 -0
  106. package/lib/site/load-adapter.mjs +30 -0
  107. package/lib/site/sessions.mjs +274 -0
  108. package/lib/site/transform.mjs +114 -0
  109. package/lib/site/values.mjs +153 -0
  110. package/lib/site/week-grid.mjs +49 -0
  111. package/lib/validate.mjs +258 -0
  112. package/lib/view/assets/common.css +789 -0
  113. package/lib/view/assets/common.js +1007 -0
  114. package/lib/view/assets/evidence.js +64 -0
  115. package/lib/view/assets/facts.js +100 -0
  116. package/lib/view/assets/form.js +211 -0
  117. package/lib/view/assets/goal.html +92 -0
  118. package/lib/view/assets/goal.js +638 -0
  119. package/lib/view/assets/insights.js +188 -0
  120. package/lib/view/assets/key.js +355 -0
  121. package/lib/view/assets/prefs.js +250 -0
  122. package/lib/view/assets/private-text.js +41 -0
  123. package/lib/view/assets/problems.css +191 -0
  124. package/lib/view/assets/problems.html +75 -0
  125. package/lib/view/assets/problems.js +1069 -0
  126. package/lib/view/assets/replay-model.js +704 -0
  127. package/lib/view/assets/replay.css +306 -0
  128. package/lib/view/assets/replay.html +122 -0
  129. package/lib/view/assets/replay.js +2079 -0
  130. package/lib/view/assets/search.html +48 -0
  131. package/lib/view/assets/search.js +615 -0
  132. package/lib/view/assets/sessions.js +144 -0
  133. package/lib/view/assets/settings.html +99 -0
  134. package/lib/view/assets/settings.js +183 -0
  135. package/lib/view/assets/setup.html +96 -0
  136. package/lib/view/assets/setup.js +131 -0
  137. package/lib/view/assets/strip.js +465 -0
  138. package/lib/view/codex-judge.mjs +411 -0
  139. package/lib/view/data.mjs +1339 -0
  140. package/lib/view/facts.mjs +306 -0
  141. package/lib/view/insights.mjs +309 -0
  142. package/lib/view/leaks.mjs +252 -0
  143. package/lib/view/problems-route.mjs +425 -0
  144. package/lib/view/progressive.mjs +134 -0
  145. package/lib/view/replay-export.mjs +252 -0
  146. package/lib/view/selftest/clickthrough.html +32 -0
  147. package/lib/view/selftest/clickthrough.js +2678 -0
  148. package/lib/view/server.mjs +336 -0
  149. package/lib/view/settings.mjs +369 -0
  150. package/lib/view/setup.mjs +286 -0
  151. package/lib/view/window.mjs +204 -0
  152. package/lib/view/word-index.mjs +192 -0
  153. package/lib/view.mjs +572 -0
  154. package/lib/voice-fence.mjs +213 -0
  155. package/lib/windows-root.mjs +17 -0
  156. package/lib/worktrees.mjs +260 -0
  157. package/package.json +44 -4
@@ -0,0 +1,993 @@
1
+ // Shared pattern authority for the canonical scrubber, the secrets-only scrubber, and the
2
+ // replayable prompt audit.
3
+ export const REDACTION_SOURCES=Object.freeze({
4
+ uuid:String.raw`\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b`,
5
+ // Searched with emailSpans() below, which finds what a global search finds in linear time.
6
+ email:String.raw`\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b`,
7
+ api:[String.raw`\bsk-[A-Za-z0-9_-]{16,}\b`,String.raw`\bgh[pousr]_[A-Za-z0-9]{20,}\b`,String.raw`\bAKIA[0-9A-Z]{12,}\b`,String.raw`\bxox[abprs]-[A-Za-z0-9-]{10,}`,String.raw`\beyJ[A-Za-z0-9_-]+(?:\.[A-Za-z0-9_-]+){2,}`,String.raw`\bglpat-[A-Za-z0-9_-]{20,}`],
8
+ keyValue:String.raw`\b([A-Za-z_][A-Za-z0-9_]*)\s*=(?!>)\s*("[^"]*"|'[^']*'|\S+)`,
9
+ sensitiveKey:String.raw`(?<![A-Za-z])(?:API_KEY|APIKEY|ACCESS_KEY|PRIVATE_KEY|PASSWORD|PASSWD|AUTHORIZATION|SECRET|TOKEN|AUTH)(?![A-Za-z])`,
10
+ currency:[String.raw`\$\s?\d[\d,]*(?:\.\d+)?`,String.raw`\b(?:USD|EUR|GBP|CAD|AUD|JPY)\s?\$?\s?\d[\d,]*(?:\.\d+)?`,String.raw`(?<![\d,])\b\d[\d,]*(?:\.\d+)?\s?(?:dollars?|euros?|pounds?|cents?|USD|EUR|GBP)\b`],
11
+ // Home / user paths. The trailing `~/…` form is as revealing as an absolute
12
+ // one: session logs are full of it, and the segment after it routinely names
13
+ // a client or a private project. Keep it LAST so the `i===1` git-bash flag
14
+ // mapping in redact.mjs stays put.
15
+ //
16
+ // The absolute forms tolerate a space because the USERNAME segment can hold
17
+ // one (Windows "Alex Jordan"). `~` is already the home dir, so there is no
18
+ // username to span and the tilde form must NOT consume spaces: paths are
19
+ // redacted at step 5, before SHAs are protected at step 7, so a space-greedy
20
+ // `~/` would swallow the rest of the line and take the git SHAs and counts
21
+ // the module promises to spare with it.
22
+ //
23
+ // The Windows form accepts runs of separators: a path inside JSON-encoded text
24
+ // (a Codex tool call's arguments, for one) arrives as `C:\\Users\\name\\…`, and
25
+ // a single-separator pattern let the username through. Three forward slashes
26
+ // after the colon are a URL scheme (`file:///Users/…`), left to the POSIX form.
27
+ // The username never crosses a double quote (so a match can't run into the next
28
+ // JSON field), and a bare home folder whose username holds a space is taken
29
+ // whole when a double quote closes it. Every quantified piece is disjoint from
30
+ // its neighbour, so matching stays linear on long runs of separators; redact.mjs
31
+ // hands back backslashes that end a match right before a quote (`\"`).
32
+ paths:[String.raw`[A-Za-z]:(?!///)[\\/]+(?:Users|users|USERS)[\\/]+(?:[^/\\\n"]+[\\/][^\s"'\n]*|[^\s/\\"'\n]+(?: [^\s/\\"'\n]+){1,2}(?=")|[^\s/\\"'\n]+)`,String.raw`/[a-z]/Users/(?:[^/\\\n]+[\\/][^\s"'\n]*(?:[\\/][^\s"'\n]*)*|[^\s/\\"'\n]+)`,String.raw`/home/(?:[^/\\\n]+[\\/][^\s"'\n]*(?:[\\/][^\s"'\n]*)*|[^\s/\\"'\n]+)`,String.raw`/Users/(?:[^/\\\n]+[\\/][^\s"'\n]*(?:[\\/][^\s"'\n]*)*|[^\s/\\"'\n]+)`,String.raw`(?<![\w-])(?:[A-Za-z]-)?-(?:Users|users|home)-[^\s"'/\\-][^\s"'/\\]*`,String.raw`(?:[A-Za-z]%3[Aa])?(?<!%(?:5[Cc]|2[Ff]))(?:%(?:5[Cc]|2[Ff]))+(?:Users|users|USERS|home)(?:%(?:5[Cc]|2[Ff]))+[^\s"'&?#<>%][^\s"'&?#<>]*`,String.raw`~[\\/][^\s"'\n]+`],
33
+ sha:String.raw`\b[0-9a-f]{7,40}\b`,account:String.raw`(?<!\.)\b\d{9,}\b(?![%.])`,opaque:String.raw`\b[A-Za-z0-9_+/=-]{32,}\b`,
34
+ });
35
+ export const regex=(source,flags='g')=>new RegExp(source,flags);
36
+ /** `/root/…`, the root account's home folder (in a container, say). Like `~/…` it has no
37
+ * username to span. It's read only where it starts a path, so `/var/root/x`, `./root/x` and
38
+ * `example.com/root/x` stay, and a bare `/root` stays: Codex names its root agent that. It's
39
+ * hidden over a scrubber's finished text (finishedSpans), so it only ever hides more. */
40
+ export const ROOT_PATH_SOURCE=String.raw`(?<![\w.~-])/root/[^\s"'\n]+`;
41
+ /** A Codex agent address, the whole of a value: the root agent `/root`, and a spawned one
42
+ * under it (`/root/wide_fixtures`). Shown as written where it comes from Codex's own agent
43
+ * fields (lib/replay/codex.mjs), never found by its shape in other text. */
44
+ export const AGENT_ADDRESS_RE=/^\/root(?:\/[A-Za-z0-9_-]+)*$/;
45
+ /** Where a home-path match should end: backslashes that end it right before a double quote
46
+ * belong to an escaped quote in JSON text (`…\\repo\"`), so they stay. A backward loop, so
47
+ * a long run of backslashes costs linear time. `end` is exclusive, in UTF-16 units. */
48
+ export function pathMatchEnd(text,start,end){let n=0;while(end-n>start&&text[end-1-n]==='\\')n+=1;return n&&text[end]==='"'?end-n:end;}
49
+ export const escapeRegex=(s)=>s.replace(/[.*+?^${}()|[\]\\]/g,'\\$&');
50
+ const EMAIL_CHAR=/[A-Za-z0-9._%+-]/;
51
+ const WORD_CHAR=/[A-Za-z0-9_]/;
52
+ /** Where the email pattern matches in `text`, as [start, end) pairs in UTF-16 units: exactly
53
+ * what a global search with it finds, in linear time. The pattern is tried only where a
54
+ * word boundary meets an address character, and when it fails there the rest of that run
55
+ * of address characters is skipped: "@" isn't one of them, so every later start in the run
56
+ * reaches the same "@" (or none) and the same domain, and fails too. A global search tries
57
+ * each of those starts, which made a long run with no address (`x-x-x-…`) quadratic. */
58
+ export function emailSpans(text){
59
+ const re=new RegExp(REDACTION_SOURCES.email,'y');
60
+ const spans=[];
61
+ for(let p=0;p<text.length;){
62
+ if(!EMAIL_CHAR.test(text[p])||WORD_CHAR.test(text[p-1]??'')===WORD_CHAR.test(text[p])){p+=1;continue;}
63
+ re.lastIndex=p;
64
+ const m=re.exec(text);
65
+ if(m){spans.push([p,p+m[0].length]);p+=m[0].length;continue;}
66
+ while(p<text.length&&EMAIL_CHAR.test(text[p]))p+=1;
67
+ }
68
+ return spans;
69
+ }
70
+ /** Where an email address with an encoded "@" sits in `text` (`name%40example.com`,
71
+ * `name@example.com`, `name&#64;example.com`), as [start, end) pairs in UTF-16 units.
72
+ * Found from each encoded "@" outward, the name capped at 64 characters and the domain at
73
+ * 255, so a long run costs linear time. */
74
+ const ENCODED_AT_RE = /%40|\\u0040|\\x40|&#0*64;|&#x0*40;/gi;
75
+ const DOMAIN_RE = /[A-Za-z0-9](?:[A-Za-z0-9-]*[A-Za-z0-9])?(?:\.[A-Za-z0-9](?:[A-Za-z0-9-]*[A-Za-z0-9])?)*\.[A-Za-z]{2,}(?![A-Za-z0-9])/y;
76
+ export function encodedEmailSpans(text) {
77
+ const spans = [];
78
+ let last = 0;
79
+ for (const m of text.matchAll(ENCODED_AT_RE)) {
80
+ if (m.index < last) continue;
81
+ let from = m.index;
82
+ while (from > Math.max(last, m.index - 64) && /[A-Za-z0-9._+-]/.test(text[from - 1])) from -= 1;
83
+ if (from === m.index) continue;
84
+ const at = m.index + m[0].length;
85
+ DOMAIN_RE.lastIndex = at;
86
+ const d = DOMAIN_RE.exec(text.slice(0, Math.min(text.length, at + 255)));
87
+ if (!d) continue;
88
+ spans.push([from, at + d[0].length]);
89
+ last = at + d[0].length;
90
+ }
91
+ return spans;
92
+ }
93
+ /** Each configured term as patterns, used by the canonical scrubber and the prompt audit.
94
+ * 1. A web-address or file-name part that starts with the term, or has it at a sub-word
95
+ * start (after "-", "_", a digit, or a camel-case boundary), is replaced whole:
96
+ * `www.acmehq.com`, `acmereport.pdf`, `api.my-acmecloud.io`, `http://acmehq:3000`. Only a
97
+ * term of four or more letters with no "." of its own is matched this way, so a short
98
+ * name like "Ion" never takes `session.ts` or `window.location.href` with it. A person's
99
+ * name (`names`) is matched this way only in a web address (after `://`, `@` or `www.`,
100
+ * or before a common ending like `.com`), so "Bill" or "Mark" don't take `billing.ts`
101
+ * or `markdown.ts` with them.
102
+ * 2. The term as a word. Only a letter next to it hides it, so `name_report`, `name2` and
103
+ * `report-name` match, and so does a camel-case part (`NameReport`, `myName`,
104
+ * `NAMEReport`, `XMLNameThing`); `names` and `Namesake` don't. Letters are spelled in
105
+ * both cases instead of using the i flag, so the edges can tell a capital from a small
106
+ * letter. The scrubber relies on that too: it skips a plain English term when the
107
+ * lowercased text lacks one of its words, which is safe only while each of the term's
108
+ * letters matches just its own two cases.
109
+ * A multi-word term matches its words split by spaces, by "-", "_" or ".", or run together
110
+ * (`Jane Doe`, `jane_doe`, `JaneDoe`); in a web-address part, by "-", "_" or nothing.
111
+ * The web-address patterns are marked `acrossPlaceholders`: the scrubber runs them over its
112
+ * whole text, placeholders included, so a later dotted piece may be one
113
+ * (`acmelogo.<hash>.png`); they never match inside a placeholder. The scrubber and the
114
+ * audit both take every pattern's matches together and replace the widest, so they agree
115
+ * when terms overlap. Each start is tried once from the front of a word or part, so
116
+ * matching stays linear. */
117
+ const anyCase=(s)=>[...s].map((c)=>{const lo=c.toLowerCase();const up=c.toUpperCase();return lo!==up&&[...lo].length===1&&[...up].length===1?`[${c===lo||c===up?'':c}${lo}${up}]`:escapeRegex(c);}).join('');
118
+ const START_EDGE='(?:(?<!\\p{L})|(?<=\\p{Ll})(?=\\p{Lu})|(?<=\\p{Lu})(?=\\p{Lu}\\p{Ll}))';
119
+ const edge=(c,side)=>/\p{L}/u.test(c)?(side==='start'?START_EDGE:'(?:(?!\\p{L})|(?<=\\p{Ll})(?=\\p{Lu})|(?<=\\p{Lu})(?=\\p{Lu}\\p{Ll}))'):/\p{N}/u.test(c)?(side==='start'?'(?<![\\p{L}\\p{N}])':'(?![\\p{L}\\p{N}])'):(side==='start'?'(?<![\\p{L}\\p{N}_])':'(?![\\p{L}\\p{N}_])');
120
+ const PART='[\\p{L}\\p{N}_-]';
121
+ // A later dotted piece: an ordinary one, or a placeholder the scrubber set aside (a token
122
+ // made of a Private-Use-Area delimiter, see freshDelimiter).
123
+ const PIECE=`(?:${PART}{1,63}|[\\uE000-\\uF8FF]+\\d+[\\uE000-\\uF8FF]+)`;
124
+ const ENDING='\\.\\p{L}[\\p{L}\\p{N}]{1,23}(?![\\p{L}\\p{N}])';
125
+ const WEB_ENDING='\\.(?:com|org|net|io|dev|app|co|ai|me|us|uk|de|fr|ca|au|nz|edu|gov|info|biz|xyz|tech|cloud|site|online|page|blog|gg|tv|fm|nl|ch|eu|es|it|se|no|dk|fi|jp|br)(?![\\p{L}\\p{N}])';
126
+ const AFTER_SCHEME='(?<=:\\/\\/(?:[^\\s/@]{1,256}@)?)';
127
+ const across=(re)=>Object.assign(re,{acrossPlaceholders:true});
128
+ // --- secret fields, shared by both scrubbers and the prompt audit -------------------
129
+ //
130
+ // The value of a sensitive field (KEY=VALUE, key: value to the end of the line, "key":
131
+ // "value" with escaped quotes, :=, ?=, ==, a typed `password: string = …`), a sensitive
132
+ // --flag or -Parameter value, Authorization and Cookie headers, Bearer/Basic credentials,
133
+ // the password in a web address or after curl's -u, and a PowerShell SecureString literal.
134
+
135
+ // A sensitive key, tested after camelCase and "-" become "_" (dbPassword, x-api-key,
136
+ // _authToken, authtoken, PGPASSWORD, MYSQL_PWD, ENCRYPTION_KEY, clientsecret, secretkey,
137
+ // X-Amz-Signature, ?sig=).
138
+ // A key merely ending in "Key" isn't one: the engine's own fileKey and sessionKey hold ids.
139
+ const SECRET_KEY_RE = /(?:(?<![A-Za-z])(?:API_?KEYS?|ACCESS_?KEY|PRIVATE_?KEY|PASSPHRASE|PASS|AUTHORIZATION|AUTH|COOKIE|CREDENTIALS?|SIGNATURE|SIG)|PASSWORD|PASSWD|TOKEN|SECRET(?:_?KEY)?|(?<=_)PWD|(?:ENCRYPTION|SIGNING|MASTER|CLIENT|LICENSE|SERVICE|DEPLOY|SESSION_SECRET|SSH|GPG|PGP|HMAC|JWT|AES|APP)_?KEY)(?![A-Za-z])/i;
140
+ export const normalKey = (key) => key.replace(/([a-z0-9])([A-Z])/g, '$1_$2').replace(/-/g, '_');
141
+ // ODBC's `PWD=…;` (any case) counts only as the whole key, so `pwd-…` in a token isn't one.
142
+ export const keyIsSecret = (key) => SECRET_KEY_RE.test(normalKey(key)) || /(?:^|\.)pwd$/i.test(key);
143
+ // Keys whose value may hold spaces, so a "key: value" runs to the end of the line.
144
+ const PASSWORD_KEY_RE = /(?:PASSWORD|PASSWD|PASSPHRASE|(?<=_)PASS|(?<=_)PWD)(?![A-Za-z])/i;
145
+ const TYPE_AHEAD_RE = /^(?:string|str|String|number|boolean|bytes|SecretStr|SecureString|Option<|Optional\[|any)\b/;
146
+ // A value read after ":" that is a type and nothing else (`string`, `str[]`, `Option<String>`).
147
+ const TYPE_ONLY_RE = /^(?:(?:string|str|String|number|boolean|bytes|SecretStr|SecureString|any)(?:\[\])*|Option<[\w<>]*>|Optional\[[\w[\]]*\])$/;
148
+ const WHOLE_LINE_KEY_RE = /(?:^|_)(?:AUTHORIZATION|COOKIE)$/i;
149
+ // Each identifier is read once from its front; its value is read only when the key is
150
+ // sensitive, so an ordinary key never swallows the next one and matching stays linear.
151
+ const IDENTIFIER_RE = /(?<![A-Za-z0-9_])[A-Za-z0-9_][A-Za-z0-9_-]*/g;
152
+ // The rest of a dotted flag name after its first part (`.token` in `--docs.token`). Each part
153
+ // starts with a letter, digit or "_", so a sentence's closing dot isn't one.
154
+ const DOTTED_NAME_RE = /(?:\.[A-Za-z0-9_][A-Za-z0-9_-]*)+/y;
155
+ const HSPACE = String.raw`[^\S\r\n]*`;
156
+ // After the key: its closing quote (escaped or not), a Go type (`var password string =`),
157
+ // an optional `?` (`password?: string`), the separator, and a type name between ":" and
158
+ // "=" when there is one (`password: string = "…"`).
159
+ const TYPE_NAME = String.raw`(?:string|str|String|bytes|\[\]byte|SecretStr|SecureString|any|&str|char\s?\*)`;
160
+ const FIELD_SEP_RE = new RegExp(String.raw`\\*["']?(?:[^\S\r\n]+${TYPE_NAME}(?=${HSPACE}=))?${HSPACE}\??(===|=>|:=|\?=|==|=|:)${HSPACE}(?:${TYPE_NAME}${HSPACE}=${HSPACE})?`, 'y');
161
+ // The next `key: ` on a joined line: a colon followed by a space or a quote, so a credential
162
+ // like `AWS id:signature`, a web address or a time doesn't count.
163
+ const NEXT_KEY_RE = /[A-Za-z_][A-Za-z0-9_-]*["']?[ \t]*:(?=[ \t"'])/y;
164
+ // A quoted value whose would-be closing quote opens a sensitive key's own value instead
165
+ // (`token: "Ab3d, api_key: "Xk9…"`) was left unclosed: it ends before that key.
166
+ const KEY_BEFORE_QUOTE_RE = /([A-Za-z_][A-Za-z0-9_-]*)\\*["']?[ \t]*(?::=|\?=|===|=>|==|=|:)[ \t]*\\*$/;
167
+ // After a --flag or -Parameter: spaces, then a value that isn't another flag.
168
+ const FLAG_SEP_RE = new RegExp(String.raw`[^\S\r\n]+(?=[^\s-])`, 'y');
169
+ const SCHEME_RE = /(?:bearer|basic|token|digest)[^\S\r\n]+/iy;
170
+ /** Bearer and Basic credentials: groups are the scheme, the gap, then the credential. The
171
+ * scheme and one credential character are exported for a scrubber that reads the
172
+ * credential across its own placeholders. */
173
+ export const BEARER_SCHEME = 'bearer|basic';
174
+ export const BEARER_CHAR = '[A-Za-z0-9._~+/=-]';
175
+ export const BEARER_RE = new RegExp(String.raw`\b(${BEARER_SCHEME})([ \t]+)${BEARER_CHAR}{8,}`, 'gi');
176
+ /** True when what BEARER_RE took for a credential is a plain word (`basic validation`,
177
+ * `Bearer authentication`, `basic end-to-end`): lowercase, Capitalized or all-caps parts
178
+ * joined by "-", maybe ending a sentence. The published scrubber and the prompt audit
179
+ * leave those; a real credential mixes cases inside a part, or holds a digit or a symbol. */
180
+ export function plainWords(credential) {
181
+ let end = credential.length;
182
+ while (end > 0 && credential[end - 1] === '.') end -= 1;
183
+ return credential.slice(0, end).split('-').every((p) => /^(?:[a-z]+|[A-Z][a-z]*|[A-Z]+)$/.test(p));
184
+ }
185
+ /** scheme://user:password@host. Group 1 runs through the ":" before the password. The user
186
+ * may be empty (redis://:password@host); the password may hold an "@" itself and ends at
187
+ * the last one. */
188
+ export const URL_PASSWORD_RE = /(?<![A-Za-z0-9+.-])([A-Za-z][A-Za-z0-9+.-]*:\/\/[^\s:/@"']*:)[^\s/"']+(?=@)/g;
189
+ /** curl and wget basic auth: -u user:password, --user user:password, --user=user:password.
190
+ * Groups are the flag, the user through its ":", then the password. The user part is
191
+ * capped, so a long run with no ":" is scanned once per flag (linear). */
192
+ export const USER_PASSWORD_RE = /(?<![\w-])(-u|--user)([ \t=]+["']?[^\s:"']{1,256}:)([^\s"']+)/g;
193
+ /** A PowerShell SecureString literal: groups are the command (with its -String,
194
+ * -AsPlainText and -Force switches), the quote, then the value. */
195
+ export const SECURE_STRING_RE = /(ConvertTo-SecureString(?:[ \t]+-(?:String|AsPlainText|Force)(?![\w-]))*[ \t]+)(["'])([^"'\n]*)\2/gi;
196
+ /** A literal piped into ConvertTo-SecureString (`"…" | ConvertTo-SecureString`): groups are
197
+ * the quote, the value, then the pipe and the command. */
198
+ export const SECURE_PIPE_RE = /(["'])([^"'\n]*)\1([ \t]*\|[ \t]*ConvertTo-SecureString\b)/gi;
199
+ /** An XML element whose tag names a secret (`<password>…</password>`): groups are the opening
200
+ * tag, its name, the value, then the closing tag. Names and values are capped, so a long run
201
+ * is scanned once per "<". */
202
+ export const XML_FIELD_RE = /(<((?:[A-Za-z_][\w.-]{0,63}:)?[A-Za-z_][\w.-]{0,63})(?:[ \t][^<>\n]{0,256})?>)([^<\n]{1,4096})(<\/\2>)/g;
203
+ /** True when an XML tag's name (without a namespace prefix) names a secret. */
204
+ export const xmlTagIsSecret = (name) => keyIsSecret(name.replace(/^.*:/, ''));
205
+
206
+ /** Where a value starting at `p` ends, and the part of it to hide. A double-quoted string
207
+ * opened by a quote escaped with `n` backslashes (n = 0 in plain JSON, 1 in JSON inside a
208
+ * JSON string, 3 a level deeper) closes at the next quote whose backslash count is n, or n
209
+ * plus a multiple of 2(n + 1): an escaped backslash right before the closing quote doesn't
210
+ * hide it, and an escaped quote inside the value doesn't end it. A quoted value cut off
211
+ * before its close runs to the end of the line. `mode` sets where an unquoted value ends:
212
+ * 'line' at the end of the line (a password line, a header), 'word' at the next space
213
+ * (after "="), 'tight' also at , ; and at ) } ] that close the value (after ":", where JSON
214
+ * and code go on). `lineEnd(p)` gives the end of the line holding p. */
215
+ function readValue(s, p, mode, lineEnd) {
216
+ const open = /\\*"/y;
217
+ open.lastIndex = p;
218
+ const o = open.exec(s);
219
+ if (o) {
220
+ const nl = lineEnd(p);
221
+ const n = o[0].length - 1;
222
+ const step = ((n + 1) & n) === 0 ? 2 * (n + 1) : 0;
223
+ for (let i = p + o[0].length; i < nl; ) {
224
+ const q = s.indexOf('"', i);
225
+ if (q < 0 || q > nl) break;
226
+ let k = 0;
227
+ while (k < q - i && s[q - 1 - k] === '\\') k += 1;
228
+ if (k === n || (step && k > n && (k - n) % step === 0)) {
229
+ const lead = s.slice(Math.max(p + o[0].length, q - 96), q);
230
+ const km = KEY_BEFORE_QUOTE_RE.exec(lead);
231
+ if (km && keyIsSecret(km[1]) && !/[A-Za-z0-9_-]/.test(lead[km.index - 1] ?? s[q - lead.length - 1] ?? '')) {
232
+ let end = q - lead.length + km.index;
233
+ while (end > p + o[0].length && /\s/.test(s[end - 1])) end -= 1;
234
+ if (end > p + o[0].length) return { start: p + o[0].length, end, after: q - lead.length + km.index };
235
+ // Nothing before the key: this quote opens that key's value, so read on to the next close.
236
+ i = q + 1;
237
+ continue;
238
+ }
239
+ return { start: p + o[0].length, end: q - n, after: q + 1 };
240
+ }
241
+ i = q + 1;
242
+ }
243
+ let end = nl;
244
+ while (end > p && /\s/.test(s[end - 1])) end -= 1;
245
+ return end > p + o[0].length ? { start: p + o[0].length, end, after: end } : null;
246
+ }
247
+ if (s[p] === "'") {
248
+ const nl1 = lineEnd(p);
249
+ const first = s.indexOf("'", p + 1);
250
+ for (let q = first; q > p && q < nl1; q = s.indexOf("'", q + 1)) {
251
+ const lead = s.slice(Math.max(p + 1, q - 96), q);
252
+ const km = KEY_BEFORE_QUOTE_RE.exec(lead);
253
+ if (km && keyIsSecret(km[1]) && !/[A-Za-z0-9_-]/.test(lead[km.index - 1] ?? s[q - lead.length - 1] ?? '')) {
254
+ let end = q - lead.length + km.index;
255
+ while (end > p + 1 && /\s/.test(s[end - 1])) end -= 1;
256
+ if (end > p + 1) return { start: p + 1, end, after: q - lead.length + km.index };
257
+ continue;
258
+ }
259
+ return { start: p + 1, end: q, after: q + 1 };
260
+ }
261
+ if (first > p && first < nl1) return { start: p + 1, end: first, after: first + 1 };
262
+ }
263
+ let end = p;
264
+ // A quote with a letter, digit or "_" on both sides is part of the value (`xk9'mp2qrz`), not
265
+ // the end of the text around it.
266
+ const quoteEnds = (i) => !(/\w/.test(s[i - 1] ?? '') && /\w/.test(s[i + 1] ?? ''));
267
+ if (mode === 'line') {
268
+ // The line ends at a newline, a quote, or the next `key:`, since the engine joins an
269
+ // excerpt's lines into one and a later key must be read on its own.
270
+ const nl = lineEnd(p);
271
+ while (end < nl && s[end] !== '\r' && !('"\''.includes(s[end]) && quoteEnds(end))) {
272
+ if (end > p && /\s/.test(s[end - 1]) && /[A-Za-z_]/.test(s[end])) {
273
+ NEXT_KEY_RE.lastIndex = end;
274
+ if (NEXT_KEY_RE.test(s)) break;
275
+ }
276
+ end += 1;
277
+ }
278
+ while (end > p && /\s/.test(s[end - 1])) end -= 1;
279
+ } else {
280
+ SCHEME_RE.lastIndex = p;
281
+ if (SCHEME_RE.exec(s)) end = SCHEME_RE.lastIndex;
282
+ const closes = (i) => mode === 'tight' && ')}]'.includes(s[i]) && (i + 1 >= s.length || /[\s"',;:)}\].?!]/.test(s[i + 1]));
283
+ const stops = mode === 'tight' ? (i) => /[\s,;]/.test(s[i]) || ('"\''.includes(s[i]) && quoteEnds(i)) : (i) => /\s/.test(s[i]);
284
+ while (end < s.length && !stops(end) && !closes(end)) end += 1;
285
+ }
286
+ return end > p ? { start: p, end, after: end } : null;
287
+ }
288
+
289
+ /** A sensitive key's value that isn't a secret: a flag value (`requiresAuth: true`) or a
290
+ * test count (`pass: 793`). */
291
+ export const notASecret = (key, value) => /^(?:true|false|null|undefined|none|nil)$/i.test(value) || (/^pass$/i.test(key) && /^\d+$/.test(value));
292
+
293
+ /** Hide the value of every sensitive field and flag in `s`. `hide(text, start, end)` gets
294
+ * one value and where it sits in `s` (UTF-16 indexes, `end` exclusive), and returns its
295
+ * replacement, or null to leave it. */
296
+ export function secretFields(s, hide) {
297
+ let out = '';
298
+ let at = 0;
299
+ const ids = new RegExp(IDENTIFIER_RE.source, 'g');
300
+ let nlAt = -1;
301
+ const lineEnd = (p) => {
302
+ if (nlAt < p) {
303
+ nlAt = s.indexOf('\n', p);
304
+ if (nlAt < 0) nlAt = s.length;
305
+ }
306
+ return nlAt;
307
+ };
308
+ for (let m = ids.exec(s); m; m = ids.exec(s)) {
309
+ // A flag's name may hold dots (`--docs.token`, `-Docs.Token`): it's read whole, each dot
310
+ // standing for "_", as `--docs-token` is. A name that isn't sensitive whole is left, and
311
+ // its later parts are read on their own as before.
312
+ let name = m[0];
313
+ let end = m.index + m[0].length;
314
+ if (s[m.index - 1] === '-') {
315
+ DOTTED_NAME_RE.lastIndex = end;
316
+ const dotted = DOTTED_NAME_RE.exec(s);
317
+ if (dotted) {
318
+ name = (m[0] + dotted[0]).replace(/\./g, '_');
319
+ end += dotted[0].length;
320
+ }
321
+ }
322
+ if (!keyIsSecret(name)) continue;
323
+ let v = null;
324
+ FIELD_SEP_RE.lastIndex = end;
325
+ const sep = FIELD_SEP_RE.exec(s);
326
+ const key = normalKey(name);
327
+ // A password line runs to its end unless what follows is a type (`password: string)`).
328
+ const typed = TYPE_AHEAD_RE.test(s.slice(FIELD_SEP_RE.lastIndex, FIELD_SEP_RE.lastIndex + 24));
329
+ // A type annotation has no value (`login(password: string, remember: boolean)`); one with
330
+ // a value after it (`password: string = "…"`) was read past by the separator. Hidden as a
331
+ // value, it made a second pass read the field another way (a placeholder isn't a type).
332
+ // Anything glued to the type word is read as a value, the way it would be without it. A
333
+ // header (Authorization, Cookie) always runs to the end of its line.
334
+ if (sep && sep[1] === ':' && typed && !WHOLE_LINE_KEY_RE.test(key)) {
335
+ const t = readValue(s, FIELD_SEP_RE.lastIndex, 'tight', lineEnd);
336
+ if (t && TYPE_ONLY_RE.test(s.slice(t.start, t.end))) continue;
337
+ }
338
+ if (sep) v = readValue(s, FIELD_SEP_RE.lastIndex, WHOLE_LINE_KEY_RE.test(key) || (sep[1] === ':' && PASSWORD_KEY_RE.test(key)) ? 'line' : sep[1] === ':' ? 'tight' : 'word', lineEnd);
339
+ else if (s[m.index - 1] === '-') {
340
+ FLAG_SEP_RE.lastIndex = end;
341
+ if (FLAG_SEP_RE.exec(s)) v = readValue(s, FLAG_SEP_RE.lastIndex, 'word', lineEnd);
342
+ }
343
+ if (!v) continue;
344
+ const text = s.slice(v.start, v.end);
345
+ if (notASecret(name, text)) continue;
346
+ const hidden = hide(text, v.start, v.end);
347
+ if (hidden === null) continue;
348
+ out += s.slice(at, v.start) + hidden;
349
+ at = v.end;
350
+ ids.lastIndex = Math.max(v.after, v.end);
351
+ }
352
+ return out + s.slice(at);
353
+ }
354
+
355
+ /** True when a field's value is only `*` and `_`: the close of a Markdown bold label
356
+ * (`**Auth:** …`) or a value already masked (`password: ****`), never a secret. */
357
+ export const markupOnly = (value) => /^[*_]+$/.test(value);
358
+
359
+ // Set-aside parts are written as tokens: a delimiter of Private-Use-Area characters that the
360
+ // text doesn't hold, a number, the delimiter again.
361
+ const SENTINEL = String.fromCharCode(0xe000);
362
+ /** A delimiter as a pattern: a run of one character as a counted repeat, so a long run (the
363
+ * text held a long run of U+E000) doesn't make a pattern too large to build. */
364
+ const delimiterSource = (d) => (d.length > 1 && [...d].every((c) => c === d[0]) ? `${d[0]}{${d.length}}` : d);
365
+ /** U+E000, repeated once more than its longest run in `text` (found in one pass). */
366
+ function sentinelFor(text) {
367
+ let longest = 0;
368
+ for (let i = 0, run = 0; i < text.length; i += 1) {
369
+ run = text.charCodeAt(i) === 0xe000 ? run + 1 : 0;
370
+ if (run > longest) longest = run;
371
+ }
372
+ return SENTINEL.repeat(longest + 1);
373
+ }
374
+ /** The delimiter for the tokens the scrubbers and the audit set parts aside as: one
375
+ * Private-Use-Area character `text` doesn't hold (U+E000 unless the text has it), found in
376
+ * one pass, so authored text can never impersonate a token. When the text holds every one of
377
+ * them, a run of U+E000 longer than any in the text. */
378
+ export function freshDelimiter(text) {
379
+ const seen = new Uint8Array(0xf900 - 0xe000);
380
+ for (let i = 0; i < text.length; i += 1) {
381
+ const c = text.charCodeAt(i);
382
+ if (c >= 0xe000 && c < 0xf900) seen[c - 0xe000] = 1;
383
+ }
384
+ const free = seen.indexOf(0);
385
+ return free >= 0 ? String.fromCharCode(0xe000 + free) : sentinelFor(text);
386
+ }
387
+ /** The pattern for one token, `delimiter digits delimiter`, with the digits as group 1. */
388
+ export const tokenSource = (delimiter) => `${delimiterSource(delimiter)}(\\d+)${delimiterSource(delimiter)}`;
389
+
390
+ /** The published scrubber's secret-field step, shared with the prompt audit: the password in
391
+ * a web address and after curl's -u, a SecureString literal, the value of every sensitive
392
+ * field and flag, and a Bearer or Basic credential that isn't a plain word. It reads `s`, a
393
+ * text in which the caller set its placeholders aside as tokens (`delimiter`, a number,
394
+ * `delimiter`); a value made only of those is left as it is. `partText(n)` gives token n's
395
+ * text. `hide(value)` returns the token to put in place of one value, which later rules here
396
+ * read as a placeholder. Returns the new text. */
397
+ export function hideSecretFields(s, { delimiter, partText, hide }) {
398
+ const d = delimiterSource(delimiter);
399
+ const tokens = new RegExp(tokenSource(delimiter), 'g');
400
+ const held = (v) => v.replace(tokens, '').trim() === '';
401
+ const restore = (v) => v.replace(tokens, (m, n) => partText(Number(n)) ?? m);
402
+ let out = s.replace(URL_PASSWORD_RE, (m, head) => (held(m.slice(head.length)) ? m : head + hide(m.slice(head.length))));
403
+ out = out.replace(USER_PASSWORD_RE, (m, flag, user, value) => (held(value) ? m : flag + user + hide(value)));
404
+ out = out.replace(SECURE_STRING_RE, (m, head, q, value) => (value && !held(value) ? `${head}${q}${hide(value)}${q}` : m));
405
+ out = out.replace(SECURE_PIPE_RE, (m, q, value, tail) => (value && !held(value) ? `${q}${hide(value)}${q}${tail}` : m));
406
+ out = out.replace(XML_FIELD_RE, (m, open, name, value, close) => (xmlTagIsSecret(name) && value.trim() && !held(value) && !markupOnly(value.trim()) && !notASecret(name.replace(/^.*:/, ''), value.trim()) ? `${open}${hide(value)}${close}` : m));
407
+ out = secretFields(out, (value) => (held(value) || markupOnly(value) ? null : hide(value)));
408
+ // A credential may run into a placeholder (`Bearer abc[redacted:term]def`); it's hidden whole.
409
+ const bearer = new RegExp(String.raw`\b(${BEARER_SCHEME})([ \t]+)((?:${BEARER_CHAR}|${d}\d+${d})+)`, 'gi');
410
+ out = out.replace(bearer, (m, scheme, gap, credential) => {
411
+ const text = restore(credential);
412
+ return held(credential) || text.length < 8 || plainWords(text) ? m : `${scheme}${gap}${hide(credential)}`;
413
+ });
414
+ // KEY=VALUE once more, on the text as it now reads. A sensitive key the scrubber's step 2 read
415
+ // as part of another key's value (`x = --token=**`, until `x` was hidden here) is read for
416
+ // itself now, as a second pass would read it. Only the value (or a comparison's operand)
417
+ // goes; one that is already a placeholder, maybe quoted, stays.
418
+ const leadToken = new RegExp(`^${tokenSource(delimiter)}`);
419
+ const settled = (v) => {
420
+ const w = v.replace(/^["']/, '');
421
+ if (w.trim() === '') return w !== '';
422
+ const t = leadToken.exec(w);
423
+ return t !== null && (w.length === t[0].length || w[t[0].length] === '"' || w[t[0].length] === "'");
424
+ };
425
+ const re = new RegExp(KV_AGAIN.source, 'g');
426
+ const rest = /\S*/y;
427
+ let result = '';
428
+ let at = 0;
429
+ for (let m = re.exec(out); m !== null; m = re.exec(out)) {
430
+ const [whole, key, value] = m;
431
+ if (!SENSITIVE_KV_KEY.test(key)) continue;
432
+ const operand = value.replace(/^=+/, '');
433
+ if (operand === '' || settled(operand)) continue;
434
+ // What is glued after the value, read only for a value that is hidden: read at every
435
+ // KEY=VALUE, it made a long run with no space in it (`x='a='x='a='…`) quadratic.
436
+ rest.lastIndex = m.index + whole.length;
437
+ const tail = rest.exec(out)[0];
438
+ result += out.slice(at, m.index) + whole.slice(0, whole.length - operand.length) + hide(operand + tail);
439
+ at = m.index + whole.length + tail.length;
440
+ re.lastIndex = at;
441
+ }
442
+ return result + out.slice(at);
443
+ }
444
+ const KV_AGAIN = new RegExp(REDACTION_SOURCES.keyValue, 'g');
445
+ const SENSITIVE_KV_KEY = new RegExp(REDACTION_SOURCES.sensitiveKey, 'i');
446
+
447
+ // A record read back as JSON loses the text around its values, so a value is hidden when
448
+ // its own key is sensitive ({"password": "…"}, {"x-api-key": "…"}). Only a key shaped like
449
+ // an identifier counts: an answer keyed by a question's text keeps its value.
450
+ export const IDENTIFIER_KEY = /^[A-Za-z_$][\w$.-]{0,63}$/;
451
+ // A long number under a password-like key ({"password": 12345678}) is hidden too.
452
+ export const NUMERIC_SECRET_KEY_RE = /(?:PASSWORD|PASSWD|PASSPHRASE|SECRET|(?<![A-Za-z])PIN)(?![A-Za-z])/i;
453
+ const ANY_PLACEHOLDER_RE = /\[redacted:\w+\]/g;
454
+ /** True when `value` holds a placeholder and nothing else but whitespace. Checked by removing
455
+ * them, not by one pattern with nested repeats, which backtracks exponentially when
456
+ * whitespace between many placeholders is followed by other text. */
457
+ export function onlyPlaceholders(value) {
458
+ const rest = value.replace(ANY_PLACEHOLDER_RE, '');
459
+ return rest.length < value.length && rest.trim() === '';
460
+ }
461
+
462
+ /** A term's words, split the way termMatchers splits them: each word is matched whole and in
463
+ * order, with only separators or nothing between them, so a text that matches a term holds
464
+ * every one of its words. The scrubber reads the same words to skip a term whose words
465
+ * aren't all in a text. */
466
+ export const termWords=(t)=>t.trim().split(/\s+/);
467
+ /** The characters that lowercase onto a plain English letter, or match one when a pattern
468
+ * ignores case (/iu): dotted capital I, long s and the Kelvin sign. A text holding one runs
469
+ * every term's patterns, whatever its words. */
470
+ export const FOLDS_ONTO_ASCII=/[\u0130\u017F\u212A]/;
471
+ /** `text` lowercased for the term check, with those three characters folded onto the letter
472
+ * they could stand for (i, s, k), so the check still finds every word a pattern could match
473
+ * and the fast path stays on. */
474
+ export const lowerForTerms=(text)=>(FOLDS_ONTO_ASCII.test(text)?text.replace(/\u0130/g,'i').replace(/\u017F/g,'s').replace(/\u212A/g,'k'):text).toLowerCase();
475
+ /** A term in each Unicode spelling it may be written in (as configured, composed NFC and
476
+ * decomposed NFD), so `café` listed one way still matches text written the other way. */
477
+ export const termSpellings=(t)=>[...new Set([t,t.normalize('NFC'),t.normalize('NFD')])];
478
+ /** The lowercased words the scrubber looks for before it runs a term's patterns, or null for
479
+ * a term with any character outside plain ASCII, which always runs. It's safe because
480
+ * anyCase spells each plain English letter as a class of its two cases, so no other
481
+ * character can match one. */
482
+ export const skipWords=(t)=>/^[\x00-\x7f]*$/.test(t)?termWords(t).map((w)=>w.toLowerCase()):null;
483
+ /** termMatchers(terms, names = []): `names` are the person names among `terms`. */
484
+ export function termMatchers(terms,names=[]){const people=new Set(names.filter((t)=>typeof t==='string').flatMap(termSpellings).map((t)=>t.trim().toLowerCase()));return terms.filter((t)=>typeof t==='string'&&t.trim()).flatMap(termSpellings).sort((x,y)=>y.trim().length-x.trim().length).flatMap((t)=>{const words=termWords(t);const cs=[...t.trim()];const word=new RegExp(`${edge(cs[0],'start')}${words.map(anyCase).join('(?:\\s+|[-_.]?)')}${edge(cs.at(-1),'end')}`,'gu');if(t.includes('.')||(t.match(/\p{L}/gu)??[]).length<4)return [word];const glued=words.map(anyCase).join('[-_]?');const part=`(?<!${PART})(?=${PART}*?${START_EDGE}${glued})(?=(${PART}+))\\1`;const dotless=across(new RegExp(`${AFTER_SCHEME}${part}(?!\\.[\\p{L}\\p{N}])`,'gu'));if(people.has(t.trim().toLowerCase()))return [across(new RegExp(`(?:${AFTER_SCHEME}|(?<=@)|(?<=(?<![\\p{L}\\p{N}_-])www\\.))${part}(?=(?:\\.${PIECE}){0,8}${ENDING})`,'gu')),across(new RegExp(`${part}(?=(?:\\.${PIECE}){0,8}${WEB_ENDING})`,'gu')),dotless,word];return [across(new RegExp(`${part}(?=(?:\\.${PIECE}){0,8}${ENDING})`,'gu')),dotless,word];});}
485
+
486
+ // --- fields read after the scrubbers' own rules ---------------------------------------
487
+ //
488
+ // Three more spellings of a secret field: a Java or Gradle property (`-Dapi.key=…`,
489
+ // `-Psigning.password=…`), a dotted key whose secret name runs across its last dot
490
+ // (`api.key=…`, `secret.key: …`), and a value after ":" whose opening single quote never
491
+ // closes on its line (`password: '…`).
492
+ //
493
+ // They are read in a pass of their own, over the text a scrubber has finished with, and never
494
+ // inside its rules. The pass can only put more placeholders down: a placeholder already there
495
+ // is opaque to it (never read as a name, never removed, split or moved), and no character
496
+ // outside the spans it hides changes. So whatever the rules above hide stays hidden, whatever
497
+ // this pass reads, and nothing it does can change what those rules saw.
498
+
499
+ const LATER_MASK = String.fromCharCode(0xe000);
500
+ const DOT_KEY_ALL_RE = new RegExp(SECRET_KEY_RE.source, 'gi');
501
+ /** True when a secret's name runs across the last dot of a dotted word (`api.key`,
502
+ * `secret.key`, `aws.secret.access.key`, `x.pwd`) and its last part alone doesn't name one
503
+ * (`db.password`, which the rules above read by its last part). `password.length` and
504
+ * `auth.ts` are a property and a file name. Only the last two parts are read. */
505
+ function secretAcrossDot(name) {
506
+ const cut = name.lastIndexOf('.');
507
+ const last = name.slice(cut + 1);
508
+ if (cut < 0 || keyIsSecret(last)) return false;
509
+ const head = normalKey(name.slice(name.lastIndexOf('.', cut - 1) + 1, cut));
510
+ const joined = `${head}_${normalKey(last)}`;
511
+ DOT_KEY_ALL_RE.lastIndex = 0;
512
+ for (let m = DOT_KEY_ALL_RE.exec(joined); m; m = DOT_KEY_ALL_RE.exec(joined)) {
513
+ if (m.index + m[0].length > head.length + 1) return true;
514
+ }
515
+ return false;
516
+ }
517
+ /** Whether a record's own key names a secret, for deepRedact and the view's leak count: a key
518
+ * the rules above call sensitive, or a dotted one whose secret name runs across its last dot
519
+ * (`api.key`). */
520
+ export const recordKeyIsSecret = (key) => keyIsSecret(key) || secretAcrossDot(key);
521
+
522
+ /** The spans of `s` the later pass hides, as [start, end) pairs in UTF-16 units, in order and
523
+ * apart. `s` is a scrubber's finished text. Each identifier is read once from its front and a
524
+ * value read is skipped past, as in secretFields, so the pass is linear. A value is read the
525
+ * way secretFields reads one (readValue), over a copy of the text with every placeholder
526
+ * masked, and the placeholders inside it are left out of what is hidden. */
527
+ export function laterFieldSpans(s) {
528
+ const marks = [];
529
+ const masked = s.replace(ANY_PLACEHOLDER_RE, (m, at) => {
530
+ marks.push([at, at + m.length]);
531
+ return LATER_MASK.repeat(m.length);
532
+ });
533
+ const spans = [];
534
+ let mark = 0;
535
+ // The parts of a value from `start` to `end` that aren't placeholders, trimmed of spaces.
536
+ const keep = (start, end) => {
537
+ while (mark < marks.length && marks[mark][1] <= start) mark += 1;
538
+ let from = start;
539
+ for (let k = mark; from < end; k += 1) {
540
+ const to = k < marks.length && marks[k][0] < end ? marks[k][0] : end;
541
+ let a = from;
542
+ let b = to;
543
+ while (a < b && /\s/.test(s[a])) a += 1;
544
+ while (b > a && /\s/.test(s[b - 1])) b -= 1;
545
+ if (b > a) spans.push([a, b]);
546
+ if (to === end) break;
547
+ from = Math.min(end, marks[k][1]);
548
+ }
549
+ };
550
+ const ids = new RegExp(IDENTIFIER_RE.source, 'g');
551
+ let nlAt = -1;
552
+ const lineEnd = (p) => {
553
+ if (nlAt < p) {
554
+ nlAt = masked.indexOf('\n', p);
555
+ if (nlAt < 0) nlAt = masked.length;
556
+ }
557
+ return nlAt;
558
+ };
559
+ // Where the last dotted word read ends: an identifier before it is one of its later parts.
560
+ let dottedEnd = 0;
561
+ for (let m = ids.exec(masked); m; m = ids.exec(masked)) {
562
+ const flag = masked[m.index - 1] === '-';
563
+ let rest = '';
564
+ if (flag || m.index >= dottedEnd) {
565
+ DOTTED_NAME_RE.lastIndex = m.index + m[0].length;
566
+ rest = DOTTED_NAME_RE.exec(masked)?.[0] ?? '';
567
+ }
568
+ const whole = m[0] + rest;
569
+ const wholeEnd = m.index + whole.length;
570
+ if (wholeEnd > dottedEnd) dottedEnd = wholeEnd;
571
+ // A name only this pass reads (`added`), or one the rules above read too: for that one,
572
+ // only a value they left because its single quote never closes is read here.
573
+ let name = m[0];
574
+ let end = m.index + m[0].length;
575
+ let added = false;
576
+ if (flag) {
577
+ const dotted = whole.replace(/\./g, '_');
578
+ const property = masked[m.index - 2] !== '-' && /^[DP][A-Za-z0-9_]/.test(whole) && masked[wholeEnd] === '=';
579
+ if (keyIsSecret(dotted)) [name, end] = [dotted, wholeEnd];
580
+ else if (property && keyIsSecret(dotted.slice(1))) [name, end, added] = [dotted.slice(1), wholeEnd, true];
581
+ else continue;
582
+ } else if (rest && secretAcrossDot(whole)) [name, end, added] = [whole.replace(/\./g, '_'), wholeEnd, true];
583
+ else if (!keyIsSecret(name)) continue;
584
+ FIELD_SEP_RE.lastIndex = end;
585
+ const sep = FIELD_SEP_RE.exec(masked);
586
+ if (!sep) continue;
587
+ const p = FIELD_SEP_RE.lastIndex;
588
+ const key = normalKey(name);
589
+ if (sep[1] === ':' && TYPE_AHEAD_RE.test(masked.slice(p, p + 24)) && !WHOLE_LINE_KEY_RE.test(key)) {
590
+ const t = readValue(masked, p, 'tight', lineEnd);
591
+ if (t && TYPE_ONLY_RE.test(masked.slice(t.start, t.end))) continue;
592
+ }
593
+ const mode = WHOLE_LINE_KEY_RE.test(key) || (sep[1] === ':' && PASSWORD_KEY_RE.test(key)) ? 'line' : sep[1] === ':' ? 'tight' : 'word';
594
+ let v = readValue(masked, p, mode, lineEnd);
595
+ // The rules above read this value themselves, and hid it or left it for a reason.
596
+ if (v && !added) continue;
597
+ // After ":", a single quote that opens no closed value: read on from after it, the way the
598
+ // value would be read without it. A quote followed by a space, the end, or , ; ) ] } . +
599
+ // closes the text the key sits in (`'Password:',`), and opens no value.
600
+ if (!v && sep[1] === ':' && masked[p] === "'" && /[^\s,;)\]}.+]/.test(masked[p + 1] ?? ' ')) v = readValue(masked, p + 1, mode, lineEnd);
601
+ if (!v) continue;
602
+ const text = s.slice(v.start, v.end);
603
+ if (!notASecret(name, text) && !markupOnly(text)) keep(v.start, v.end);
604
+ ids.lastIndex = Math.max(v.after, v.end, ids.lastIndex);
605
+ }
606
+ return spans;
607
+ }
608
+
609
+ /** `s` with the later pass's spans replaced by `placeholder`; `counted()` is called once per span. */
610
+ export function hideLaterFields(s, placeholder, counted = () => {}) {
611
+ const spans = laterFieldSpans(s);
612
+ if (!spans.length) return s;
613
+ let out = '';
614
+ let at = 0;
615
+ for (const [start, end] of spans) {
616
+ counted();
617
+ out += s.slice(at, start) + placeholder;
618
+ at = end;
619
+ }
620
+ return out + s.slice(at);
621
+ }
622
+
623
+ // --- fields read from the text as it was written ---------------------------------------
624
+ //
625
+ // The rules above read a field's value and move on past it. When that value is itself a name
626
+ // (`api_key=token: …`, `password="token": "…"`, `token: "Ab3d, api.key: "…"`), they hide the
627
+ // name and the real value after it is left. And the KEY=VALUE rule reads a key only where an
628
+ // earlier `name=` hasn't taken it for a value (`x=` on one line, `apiKey` on the next, `=…`
629
+ // on the third).
630
+ //
631
+ // Those values are found here, in the text as it was written: by then the scrubbers have put a
632
+ // placeholder where the inner name was, so their finished text no longer says there was one.
633
+ // What is found is hidden in addition to everything the scrubbers hide. hideSourceFields takes
634
+ // a scrubber's finished text and only swaps more of its shown characters for a placeholder,
635
+ // and the audit only adds spans between the ones it has. So nothing the rules above hide can
636
+ // show again, whatever is read here.
637
+
638
+ // What the KEY=VALUE rule reads after its key, from a fixed start.
639
+ const KV_REST_RE = /\s*=\s*("[^"]*"|'[^']*'|\S+)/y;
640
+ // A word every sensitive key holds: each name SECRET_KEY_RE matches has one of these in it, and
641
+ // normalKey only adds "_" to a name, so a text with none of them has no sensitive key at all.
642
+ const SECRET_NAME_HINT_RE = /KEY|PASS|AUTH|COOKIE|CREDENTIAL|SIG|TOKEN|SECRET|PWD/i;
643
+ const TAIL_SEP_CHAR = /[\\"' \t:=?]/;
644
+ const NAME_CHAR = /[A-Za-z0-9_.-]/;
645
+ // What is trimmed from each end of a piece hidden here: spaces, and quotes with their escapes.
646
+ const EDGE_CHAR = /[\s\\"']/;
647
+
648
+ /** The spans of `s`, a text as it was written, that hold a value the rules above leave shown
649
+ * because of what sits before it, as [start, end) pairs in UTF-16 units, in order and apart:
650
+ * - the value after a name that ends a secret field's own value, or is all of it in quotes
651
+ * (`api_key=token: …`, `password="token": "…"`, `token: "see api.key: "…"`), any number of
652
+ * names deep. The inner name must name a secret itself, or be one quoted word;
653
+ * - the value of a KEY=VALUE key split from its "=" or its value by a line break, which the
654
+ * rule misses when an earlier `name=` took the key for its value.
655
+ * Placeholders already in `s` are masked and left out, as in laterFieldSpans. A field is read
656
+ * the way secretFields reads it and is skipped past, and each inner value starts at or after
657
+ * the end of the one before it, so the pass is linear. */
658
+ export function sourceFieldSpans(s) {
659
+ // Every span here follows a sensitive key. A text that names none is done in one search,
660
+ // so the audit, which reads its source and then its rendition, pays for neither.
661
+ if (!SECRET_NAME_HINT_RE.test(s)) return [];
662
+ const marks = [];
663
+ const masked = s.replace(ANY_PLACEHOLDER_RE, (m, at) => {
664
+ marks.push([at, at + m.length]);
665
+ return LATER_MASK.repeat(m.length);
666
+ });
667
+ const spans = [];
668
+ let mark = 0;
669
+ // The parts of a value from `start` to `end` that aren't placeholders, trimmed of spaces and of
670
+ // quotes: a quote is where the rules stop reading a line, and one left in place stops a second
671
+ // pass where it stopped the first. What is left of a value next to a placeholder inside it is
672
+ // kept only when it holds a letter or a digit (`":` between two placeholders isn't a value).
673
+ const keep = (start, end) => {
674
+ while (mark < marks.length && marks[mark][1] <= start) mark += 1;
675
+ let from = start;
676
+ for (let k = mark; from < end; k += 1) {
677
+ const to = k < marks.length && marks[k][0] < end ? marks[k][0] : end;
678
+ let a = from;
679
+ let b = to;
680
+ while (a < b && EDGE_CHAR.test(s[a])) a += 1;
681
+ while (b > a && EDGE_CHAR.test(s[b - 1])) b -= 1;
682
+ if (b > a && ((from === start && to === end) || /[\p{L}\p{N}]/u.test(s.slice(a, b)))) spans.push([a, b]);
683
+ if (to === end) break;
684
+ from = Math.min(end, marks[k][1]);
685
+ }
686
+ };
687
+ let nlAt = -1;
688
+ const lineEnd = (p) => {
689
+ if (nlAt < p) {
690
+ nlAt = masked.indexOf('\n', p);
691
+ if (nlAt < 0) nlAt = masked.length;
692
+ }
693
+ return nlAt;
694
+ };
695
+ // The value of the field whose name ends at `end`, read as secretFields reads it.
696
+ const valueOf = (name, end, flag) => {
697
+ const key = normalKey(name);
698
+ FIELD_SEP_RE.lastIndex = end;
699
+ const sep = FIELD_SEP_RE.exec(masked);
700
+ if (!sep) {
701
+ if (!flag) return null;
702
+ FLAG_SEP_RE.lastIndex = end;
703
+ return FLAG_SEP_RE.exec(masked) ? readValue(masked, FLAG_SEP_RE.lastIndex, 'word', lineEnd) : null;
704
+ }
705
+ const p = FIELD_SEP_RE.lastIndex;
706
+ const wholeLine = WHOLE_LINE_KEY_RE.test(key);
707
+ if (sep[1] === ':' && !wholeLine && TYPE_AHEAD_RE.test(masked.slice(p, p + 24))) {
708
+ const t = readValue(masked, p, 'tight', lineEnd);
709
+ if (t && TYPE_ONLY_RE.test(masked.slice(t.start, t.end))) return null;
710
+ }
711
+ const mode = wholeLine || (sep[1] === ':' && PASSWORD_KEY_RE.test(key)) ? 'line' : sep[1] === ':' ? 'tight' : 'word';
712
+ let v = readValue(masked, p, mode, lineEnd);
713
+ // A single quote that never closes, read the way laterFieldSpans reads it.
714
+ if (!v && sep[1] === ':' && masked[p] === "'" && /[^\s,;)\]}.+]/.test(masked[p + 1] ?? ' ')) v = readValue(masked, p + 1, mode, lineEnd);
715
+ return v && { ...v, line: mode === 'line' };
716
+ };
717
+ // The name a value ends with, when its separator runs to the value's end or past it: the
718
+ // value was a name, and the real one comes next. Null when there is none.
719
+ const innerName = (v) => {
720
+ let ne = v.end;
721
+ while (ne > v.start && TAIL_SEP_CHAR.test(masked[ne - 1])) ne -= 1;
722
+ let ns = ne;
723
+ while (ns > v.start && ne - ns <= 96 && NAME_CHAR.test(masked[ns - 1])) ns -= 1;
724
+ if (ne - ns > 96) return null;
725
+ while (ns < ne && /[.-]/.test(masked[ns])) ns += 1;
726
+ if (ns === ne) return null;
727
+ FIELD_SEP_RE.lastIndex = ne;
728
+ if (!FIELD_SEP_RE.exec(masked) || FIELD_SEP_RE.lastIndex < v.end) return null;
729
+ const name = masked.slice(ns, ne);
730
+ const quotedWord = ns === v.start && ne === v.end && /["']/.test(masked[ns - 1] ?? '') && /["'\\]/.test(masked[ne] ?? '');
731
+ const secret = keyIsSecret(name.slice(name.lastIndexOf('.') + 1)) || secretAcrossDot(name) || (masked[ns - 1] === '-' && keyIsSecret(name.replace(/\./g, '_')));
732
+ return secret || quotedWord ? { name: name.replace(/\./g, '_'), end: ne } : null;
733
+ };
734
+ const ids = new RegExp(IDENTIFIER_RE.source, 'g');
735
+ // Where the last dotted word read ends: an identifier before it is one of its later parts.
736
+ let dottedEnd = 0;
737
+ // Where the last field's own value ends.
738
+ let ownEnd = 0;
739
+ for (let m = ids.exec(masked); m; m = ids.exec(masked)) {
740
+ const flag = masked[m.index - 1] === '-';
741
+ let rest = '';
742
+ if (flag || m.index >= dottedEnd) {
743
+ DOTTED_NAME_RE.lastIndex = m.index + m[0].length;
744
+ rest = DOTTED_NAME_RE.exec(masked)?.[0] ?? '';
745
+ }
746
+ const whole = m[0] + rest;
747
+ const wholeEnd = m.index + whole.length;
748
+ if (wholeEnd > dottedEnd) dottedEnd = wholeEnd;
749
+ let name = m[0];
750
+ let end = m.index + m[0].length;
751
+ if (flag) {
752
+ const dotted = whole.replace(/\./g, '_');
753
+ const property = masked[m.index - 2] !== '-' && /^[DP][A-Za-z0-9_]/.test(whole) && masked[wholeEnd] === '=';
754
+ if (keyIsSecret(dotted)) [name, end] = [dotted, wholeEnd];
755
+ else if (property && keyIsSecret(dotted.slice(1))) [name, end] = [dotted.slice(1), wholeEnd];
756
+ else continue;
757
+ } else if (rest && secretAcrossDot(whole)) [name, end] = [whole.replace(/\./g, '_'), wholeEnd];
758
+ else if (!keyIsSecret(name)) continue;
759
+ // The field's own value is the rules' to hide; only what follows a name in it is read here.
760
+ // A name inside the value of the field before it isn't read as a field again (that would
761
+ // read a long value once per name in it); only a key split from its "=" is read there.
762
+ let v = m.index < ownEnd ? null : valueOf(name, end, flag);
763
+ const own = v;
764
+ if (own) ownEnd = Math.max(ownEnd, own.end);
765
+ if (!v) {
766
+ // No value on the key's own line: the KEY=VALUE rule's reading, across the line break.
767
+ const kvKey = m[0].slice(m[0].lastIndexOf('-') + 1);
768
+ if (!/^[A-Za-z_]/.test(kvKey) || !SENSITIVE_KV_KEY.test(kvKey)) continue;
769
+ KV_REST_RE.lastIndex = m.index + m[0].length;
770
+ const kv = KV_REST_RE.exec(masked);
771
+ if (!kv) continue;
772
+ const after = KV_REST_RE.lastIndex;
773
+ let from = after - kv[1].length;
774
+ let to = after;
775
+ // The rest of an == or === comparison: its operand.
776
+ while (from < to && masked[from] === '=') from += 1;
777
+ if (to - from >= 2 && '"\''.includes(masked[from]) && masked[to - 1] === masked[from]) [from, to] = [from + 1, to - 1];
778
+ if (to <= from) continue;
779
+ keep(from, to);
780
+ v = { start: from, end: to, after };
781
+ }
782
+ let hidden = 0;
783
+ for (let inner = innerName(v); ; inner = innerName(v)) {
784
+ // A value hidden here is skipped past; the field's own is read on through.
785
+ if (v !== own) ids.lastIndex = Math.max(v.after, v.end, ids.lastIndex);
786
+ if (!inner) break;
787
+ const next = valueOf(inner.name, inner.end, false);
788
+ if (!next) break;
789
+ v = next;
790
+ const text = s.slice(v.start, v.end);
791
+ if (notASecret(inner.name, text) || markupOnly(text)) continue;
792
+ keep(v.start, v.end);
793
+ hidden += 1;
794
+ }
795
+ // A password line or a header hides to the end of its line, or to the next `name:` on it.
796
+ // With that name's value now hidden, the rest of the line is the field's own, as a second
797
+ // pass would read it.
798
+ if (own?.line && hidden && !/["'\\]/.test(masked[v.end] ?? '"')) {
799
+ const more = readValue(masked, v.end, 'line', lineEnd);
800
+ if (more) {
801
+ keep(more.start, more.end);
802
+ ids.lastIndex = Math.max(more.after, more.end, ids.lastIndex);
803
+ }
804
+ }
805
+ }
806
+ return spans;
807
+ }
808
+
809
+ /** Where each run of text between the placeholders of `redacted` sits in `original`, the text
810
+ * it was made from: { at, length } in `redacted`, and the earliest and latest place it can
811
+ * start in `original` (`lo`, `hi`), found from the front and from the back. A scrubber's
812
+ * finished text is its input with spans swapped for placeholders, so the runs appear in the
813
+ * input in order, the first at its start and the last at its end. A run whose `lo` and `hi`
814
+ * agree sits exactly there. The KEY=VALUE rule writes its own "=" before its placeholder
815
+ * whatever spacing the input had, so a run's "=" right before a placeholder isn't matched
816
+ * (`equals` marks a run that has one).
817
+ * Null when the runs don't fit `original` that way. */
818
+ export function alignRedacted(original, redacted) {
819
+ const chunks = [];
820
+ let at = 0;
821
+ for (const m of redacted.matchAll(ANY_PLACEHOLDER_RE)) {
822
+ const equals = m.index > at && redacted[m.index - 1] === '=';
823
+ chunks.push({ at, length: m.index - at - (equals ? 1 : 0), equals, lo: 0, hi: 0 });
824
+ at = m.index + m[0].length;
825
+ }
826
+ chunks.push({ at, length: redacted.length - at, equals: false, lo: 0, hi: 0 });
827
+ const text = (c) => redacted.slice(c.at, c.at + c.length);
828
+ const first = chunks[0];
829
+ const last = chunks[chunks.length - 1];
830
+ if (!original.startsWith(text(first))) return null;
831
+ if (chunks.length === 1) return original.length === first.length ? chunks : null;
832
+ last.lo = original.length - last.length;
833
+ last.hi = last.lo;
834
+ if (last.lo < first.length || !original.endsWith(text(last))) return null;
835
+ let pos = first.length;
836
+ for (let i = 1; i < chunks.length - 1; i += 1) {
837
+ const c = chunks[i];
838
+ c.lo = original.indexOf(text(c), pos);
839
+ if (c.lo < 0 || c.lo + c.length > last.lo) return null;
840
+ pos = c.lo + c.length;
841
+ }
842
+ let limit = last.lo;
843
+ for (let i = chunks.length - 2; i > 0; i -= 1) {
844
+ const c = chunks[i];
845
+ c.hi = original.lastIndexOf(text(c), limit - c.length);
846
+ limit = c.hi;
847
+ }
848
+ return chunks;
849
+ }
850
+
851
+ /** `redacted`, a scrubber's finished text for `original`, with what sourceFieldSpans finds in
852
+ * `original` hidden too. Only characters `redacted` still shows are replaced, each run by
853
+ * `placeholder`; no placeholder in it is touched. A run of text that could sit in more than one
854
+ * place in `original` is hidden whole when a span reaches any of those places, and so is every
855
+ * run when the two texts don't line up, so a value found is never left for want of its place.
856
+ * `counted()` is called once per run hidden. */
857
+ export function hideSourceFields(original, redacted, placeholder, counted = () => {}) {
858
+ const spans = sourceFieldSpans(original);
859
+ if (!spans.length) return redacted;
860
+ const chunks = alignRedacted(original, redacted);
861
+ const cuts = [];
862
+ if (!chunks) {
863
+ let at = 0;
864
+ for (const m of redacted.matchAll(ANY_PLACEHOLDER_RE)) {
865
+ cuts.push([at, m.index]);
866
+ at = m.index + m[0].length;
867
+ }
868
+ cuts.push([at, redacted.length]);
869
+ } else {
870
+ let k = 0;
871
+ for (const [a, b] of spans) {
872
+ while (k < chunks.length && chunks[k].hi + chunks[k].length <= a) k += 1;
873
+ for (let j = k; j < chunks.length && chunks[j].lo < b; j += 1) {
874
+ const c = chunks[j];
875
+ if (c.lo === c.hi) {
876
+ const from = Math.max(a, c.lo);
877
+ const to = Math.min(b, c.lo + c.length);
878
+ if (to <= from) continue;
879
+ // The run's own "=" goes too when the span runs on over the "=" in `original`.
880
+ let q = to;
881
+ if (c.equals && to === c.lo + c.length) while (q < b && /\s/.test(original[q])) q += 1;
882
+ const equals = c.equals && to === c.lo + c.length && q < b && original[q] === '=';
883
+ cuts.push([c.at + from - c.lo, c.at + to - c.lo + (equals ? 1 : 0)]);
884
+ } else if (c.length && a < c.hi + c.length) cuts.push([c.at, c.at + c.length]);
885
+ }
886
+ }
887
+ }
888
+ let out = '';
889
+ let at = 0;
890
+ for (const cut of cuts) {
891
+ let start = Math.max(cut[0], at);
892
+ let end = cut[1];
893
+ while (start < end && /\s/.test(redacted[start])) start += 1;
894
+ while (end > start && /\s/.test(redacted[end - 1])) end -= 1;
895
+ if (end <= start) continue;
896
+ counted();
897
+ out += redacted.slice(at, start) + placeholder;
898
+ at = end;
899
+ }
900
+ return out + redacted.slice(at);
901
+ }
902
+
903
+ // --- last, over the finished text ------------------------------------------------------
904
+ //
905
+ // Two things the rules above leave, read once a scrubber is otherwise done, so they can only
906
+ // ever hide more: no character outside the spans they hide changes, and no placeholder moves.
907
+ // - A header (Authorization, Cookie) whose value is only placeholders and stops at a quote
908
+ // with text glued after it (`Authorization=[redacted:secret]'xk9…`, issue 80): the rest of
909
+ // the line, from the quote, is the header's too.
910
+ // - A `/root/…` home path (ROOT_PATH_SOURCE, issue 85), each part between placeholders.
911
+
912
+ const ROOT_PATH_RE = new RegExp(ROOT_PATH_SOURCE, 'g');
913
+ const onlyMasks = (text) => text.includes(LATER_MASK) && text.replace(/\s/g, '').split(LATER_MASK).join('') === '';
914
+
915
+ /** The spans of `s` hidden last, as [start, end, kind) triples in UTF-16 units, in order and
916
+ * apart: kind is 'secret' for a header's glued text and 'path' for a `/root/…` path, left out
917
+ * when `paths` is false. `s` is a scrubber's finished text; spans hold no placeholder. */
918
+ export function finishedSpans(s, { paths = true } = {}) {
919
+ if (!s.includes('/root/') && !SECRET_NAME_HINT_RE.test(s)) return [];
920
+ const marks = [];
921
+ const masked = s.replace(ANY_PLACEHOLDER_RE, (m, at) => {
922
+ marks.push([at, at + m.length]);
923
+ return LATER_MASK.repeat(m.length);
924
+ });
925
+ // The parts of [start, end) between placeholders, trimmed of spaces, as spans of `kind`.
926
+ const pieces = (start, end, kind, out) => {
927
+ let k = 0;
928
+ while (k < marks.length && marks[k][1] <= start) k += 1;
929
+ for (let from = start; from < end; k += 1) {
930
+ const to = k < marks.length && marks[k][0] < end ? marks[k][0] : end;
931
+ let a = from;
932
+ let b = to;
933
+ while (a < b && /\s/.test(s[a])) a += 1;
934
+ while (b > a && /\s/.test(s[b - 1])) b -= 1;
935
+ if (b > a) out.push([a, b, kind]);
936
+ if (to === end) break;
937
+ from = Math.min(end, marks[k][1]);
938
+ }
939
+ };
940
+ const headers = [];
941
+ let nlAt = -1;
942
+ const lineEnd = (p) => {
943
+ if (nlAt < p) {
944
+ nlAt = masked.indexOf('\n', p);
945
+ if (nlAt < 0) nlAt = masked.length;
946
+ }
947
+ return nlAt;
948
+ };
949
+ const ids = new RegExp(IDENTIFIER_RE.source, 'g');
950
+ for (let m = ids.exec(masked); m; m = ids.exec(masked)) {
951
+ if (!WHOLE_LINE_KEY_RE.test(normalKey(m[0])) || !keyIsSecret(m[0])) continue;
952
+ FIELD_SEP_RE.lastIndex = m.index + m[0].length;
953
+ if (!FIELD_SEP_RE.exec(masked)) continue;
954
+ const v = readValue(masked, FIELD_SEP_RE.lastIndex, 'line', lineEnd);
955
+ if (!v || !onlyMasks(masked.slice(v.start, v.end))) continue;
956
+ // What it hides can end at another quote with text glued after it: that's the header's too.
957
+ let end = v.end;
958
+ for (let more; `"'`.includes(masked[end] ?? '') && /\w/.test(masked[end + 1] ?? '') && (more = readValue(masked, end + 1, 'line', lineEnd)); end = more.end) {
959
+ pieces(end, more.end, 'secret', headers);
960
+ ids.lastIndex = Math.max(more.after, more.end);
961
+ }
962
+ }
963
+ const rootPaths = [];
964
+ if (paths) {
965
+ for (const r of masked.matchAll(ROOT_PATH_RE)) pieces(r.index, pathMatchEnd(masked, r.index, r.index + r[0].length), 'path', rootPaths);
966
+ }
967
+ // Spans that overlap become one, a path only when both were.
968
+ const out = [];
969
+ for (const span of [...headers, ...rootPaths].sort((x, y) => x[0] - y[0])) {
970
+ const last = out[out.length - 1];
971
+ if (last && span[0] < last[1]) {
972
+ last[1] = Math.max(last[1], span[1]);
973
+ if (span[2] !== 'path') last[2] = 'secret';
974
+ } else out.push([...span]);
975
+ }
976
+ return out;
977
+ }
978
+
979
+ /** `s` with its finishedSpans replaced: a header's glued text by `placeholder`, a path by
980
+ * `pathPlaceholder`, or left shown when that's null. `counted()` is called once per span. */
981
+ export function hideFinishedSpans(s, placeholder, counted = () => {}, pathPlaceholder = '[redacted:path]') {
982
+ if (typeof s !== 'string' || s.length === 0) return s;
983
+ const spans = finishedSpans(s, { paths: pathPlaceholder != null });
984
+ if (!spans.length) return s;
985
+ let out = '';
986
+ let at = 0;
987
+ for (const [start, end, kind] of spans) {
988
+ counted();
989
+ out += s.slice(at, start) + (kind === 'path' ? pathPlaceholder : placeholder);
990
+ at = end;
991
+ }
992
+ return out + s.slice(at);
993
+ }