@agentdocstore/core 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +55 -0
- package/dist/authz.d.ts +76 -0
- package/dist/authz.js +106 -0
- package/dist/diff.d.ts +31 -0
- package/dist/diff.js +57 -0
- package/dist/errors.d.ts +54 -0
- package/dist/errors.js +75 -0
- package/dist/id.d.ts +18 -0
- package/dist/id.js +26 -0
- package/dist/index.d.ts +15 -0
- package/dist/index.js +15 -0
- package/dist/model/comment.d.ts +18 -0
- package/dist/model/comment.js +2 -0
- package/dist/model/document.d.ts +41 -0
- package/dist/model/document.js +32 -0
- package/dist/model/edit-message.d.ts +8 -0
- package/dist/model/edit-message.js +20 -0
- package/dist/model/limits.d.ts +42 -0
- package/dist/model/limits.js +43 -0
- package/dist/model/title.d.ts +8 -0
- package/dist/model/title.js +17 -0
- package/dist/model/version.d.ts +20 -0
- package/dist/model/version.js +2 -0
- package/dist/offline.d.ts +57 -0
- package/dist/offline.js +58 -0
- package/dist/scanner.d.ts +47 -0
- package/dist/scanner.js +294 -0
- package/dist/search/CoreSearchIndex.d.ts +21 -0
- package/dist/search/CoreSearchIndex.js +87 -0
- package/dist/spi/capabilities.d.ts +28 -0
- package/dist/spi/capabilities.js +2 -0
- package/dist/spi/comments.d.ts +14 -0
- package/dist/spi/comments.js +2 -0
- package/dist/spi/identity.d.ts +13 -0
- package/dist/spi/identity.js +2 -0
- package/dist/spi/index.d.ts +8 -0
- package/dist/spi/index.js +8 -0
- package/dist/spi/module.d.ts +54 -0
- package/dist/spi/module.js +57 -0
- package/dist/spi/provider.d.ts +33 -0
- package/dist/spi/provider.js +2 -0
- package/dist/spi/repository.d.ts +64 -0
- package/dist/spi/repository.js +2 -0
- package/dist/spi/search.d.ts +53 -0
- package/dist/spi/search.js +2 -0
- package/package.json +58 -0
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Size caps enforced across the app. Content is capped at write time (never
|
|
3
|
+
* stored oversized); the diff-input cap bounds the diff engine's inputs.
|
|
4
|
+
*/
|
|
5
|
+
export declare const LIMITS: {
|
|
6
|
+
/** Max doc title length in UTF-8 bytes. */
|
|
7
|
+
readonly MAX_TITLE_BYTES: 300;
|
|
8
|
+
/** Max stored content per version in UTF-8 bytes (write-time reject). */
|
|
9
|
+
readonly MAX_CONTENT_BYTES: number;
|
|
10
|
+
/**
|
|
11
|
+
* Max JSON request body in UTF-8 bytes. JSON writes a quote, backslash,
|
|
12
|
+
* newline or tab as two bytes, so the body of a write can be twice its
|
|
13
|
+
* content; this leaves room for that at MAX_CONTENT_BYTES, plus 1 MiB for
|
|
14
|
+
* the other fields. (Other control characters take six bytes each; text
|
|
15
|
+
* made of those can still pass it.)
|
|
16
|
+
*/
|
|
17
|
+
readonly MAX_REQUEST_BYTES: number;
|
|
18
|
+
/** Max input size (per side) accepted by the diff engine in UTF-8 bytes. */
|
|
19
|
+
readonly MAX_DIFF_INPUT_BYTES: number;
|
|
20
|
+
/**
|
|
21
|
+
* Max lines a diff may add or remove. Finding a diff costs time that grows
|
|
22
|
+
* with the square of the lines that differ, and it runs on the server's
|
|
23
|
+
* only thread, so versions that share few lines could stall every other
|
|
24
|
+
* request for minutes; past this the diff engine stops and refuses.
|
|
25
|
+
*/
|
|
26
|
+
readonly MAX_DIFF_CHANGED_LINES: 5000;
|
|
27
|
+
/** Max comment body length in UTF-8 bytes. */
|
|
28
|
+
readonly MAX_COMMENT_BYTES: 10000;
|
|
29
|
+
/** Max edit message length in characters, after trimming. */
|
|
30
|
+
readonly MAX_EDIT_MESSAGE_CHARS: 500;
|
|
31
|
+
/** Longest expiry, in days from now (about 100 years). Much further is not a date. */
|
|
32
|
+
readonly MAX_EXPIRY_DAYS: 36500;
|
|
33
|
+
};
|
|
34
|
+
/** The shape of {@link LIMITS}. */
|
|
35
|
+
export type Limits = typeof LIMITS;
|
|
36
|
+
/**
|
|
37
|
+
* The detail for a size-limit error: "10976 bytes; the limit is 10000 bytes".
|
|
38
|
+
* Size errors carry it so a REST or MCP client learns how much to cut, not
|
|
39
|
+
* only that the value was too big.
|
|
40
|
+
*/
|
|
41
|
+
export declare function sizeOverLimit(size: number, limit: number): string;
|
|
42
|
+
//# sourceMappingURL=limits.d.ts.map
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
const MAX_CONTENT_BYTES = 5 * 1024 * 1024;
|
|
2
|
+
/**
|
|
3
|
+
* Size caps enforced across the app. Content is capped at write time (never
|
|
4
|
+
* stored oversized); the diff-input cap bounds the diff engine's inputs.
|
|
5
|
+
*/
|
|
6
|
+
export const LIMITS = {
|
|
7
|
+
/** Max doc title length in UTF-8 bytes. */
|
|
8
|
+
MAX_TITLE_BYTES: 300,
|
|
9
|
+
/** Max stored content per version in UTF-8 bytes (write-time reject). */
|
|
10
|
+
MAX_CONTENT_BYTES,
|
|
11
|
+
/**
|
|
12
|
+
* Max JSON request body in UTF-8 bytes. JSON writes a quote, backslash,
|
|
13
|
+
* newline or tab as two bytes, so the body of a write can be twice its
|
|
14
|
+
* content; this leaves room for that at MAX_CONTENT_BYTES, plus 1 MiB for
|
|
15
|
+
* the other fields. (Other control characters take six bytes each; text
|
|
16
|
+
* made of those can still pass it.)
|
|
17
|
+
*/
|
|
18
|
+
MAX_REQUEST_BYTES: 2 * MAX_CONTENT_BYTES + 1024 * 1024,
|
|
19
|
+
/** Max input size (per side) accepted by the diff engine in UTF-8 bytes. */
|
|
20
|
+
MAX_DIFF_INPUT_BYTES: 2 * 1024 * 1024,
|
|
21
|
+
/**
|
|
22
|
+
* Max lines a diff may add or remove. Finding a diff costs time that grows
|
|
23
|
+
* with the square of the lines that differ, and it runs on the server's
|
|
24
|
+
* only thread, so versions that share few lines could stall every other
|
|
25
|
+
* request for minutes; past this the diff engine stops and refuses.
|
|
26
|
+
*/
|
|
27
|
+
MAX_DIFF_CHANGED_LINES: 5_000,
|
|
28
|
+
/** Max comment body length in UTF-8 bytes. */
|
|
29
|
+
MAX_COMMENT_BYTES: 10_000,
|
|
30
|
+
/** Max edit message length in characters, after trimming. */
|
|
31
|
+
MAX_EDIT_MESSAGE_CHARS: 500,
|
|
32
|
+
/** Longest expiry, in days from now (about 100 years). Much further is not a date. */
|
|
33
|
+
MAX_EXPIRY_DAYS: 36_500,
|
|
34
|
+
};
|
|
35
|
+
/**
|
|
36
|
+
* The detail for a size-limit error: "10976 bytes; the limit is 10000 bytes".
|
|
37
|
+
* Size errors carry it so a REST or MCP client learns how much to cut, not
|
|
38
|
+
* only that the value was too big.
|
|
39
|
+
*/
|
|
40
|
+
export function sizeOverLimit(size, limit) {
|
|
41
|
+
return `${size} bytes; the limit is ${limit} bytes`;
|
|
42
|
+
}
|
|
43
|
+
//# sourceMappingURL=limits.js.map
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reject a title the built-in providers would refuse: blank, or longer than
|
|
3
|
+
* {@link LIMITS.MAX_TITLE_BYTES} bytes of UTF-8. REST and MCP call this before
|
|
4
|
+
* their first write, so an update with a bad title saves nothing — the
|
|
5
|
+
* provider's own check would only fire after the new version was stored.
|
|
6
|
+
*/
|
|
7
|
+
export declare function validateTitle(title: string): void;
|
|
8
|
+
//# sourceMappingURL=title.d.ts.map
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { ValidationError } from '../errors.js';
|
|
2
|
+
import { LIMITS, sizeOverLimit } from './limits.js';
|
|
3
|
+
/**
|
|
4
|
+
* Reject a title the built-in providers would refuse: blank, or longer than
|
|
5
|
+
* {@link LIMITS.MAX_TITLE_BYTES} bytes of UTF-8. REST and MCP call this before
|
|
6
|
+
* their first write, so an update with a bad title saves nothing — the
|
|
7
|
+
* provider's own check would only fire after the new version was stored.
|
|
8
|
+
*/
|
|
9
|
+
export function validateTitle(title) {
|
|
10
|
+
if (title.trim().length === 0)
|
|
11
|
+
throw new ValidationError('Title must not be empty');
|
|
12
|
+
const size = Buffer.byteLength(title, 'utf8');
|
|
13
|
+
if (size > LIMITS.MAX_TITLE_BYTES) {
|
|
14
|
+
throw new ValidationError(`Title exceeds maximum length (${sizeOverLimit(size, LIMITS.MAX_TITLE_BYTES)})`);
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
//# sourceMappingURL=title.js.map
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* An immutable version of a {@link Document}'s content. Versions are append-only and
|
|
3
|
+
* addressed by a 1-based monotonic number; the doc's `latestVersion` points at
|
|
4
|
+
* the newest one.
|
|
5
|
+
*/
|
|
6
|
+
export interface DocumentVersion {
|
|
7
|
+
/** Owning doc id. */
|
|
8
|
+
readonly documentId: string;
|
|
9
|
+
/** 1-based monotonic version number. */
|
|
10
|
+
readonly version: number;
|
|
11
|
+
/** Raw content of this version. */
|
|
12
|
+
readonly content: string;
|
|
13
|
+
/** Identity that authored this version. */
|
|
14
|
+
readonly createdBy: string;
|
|
15
|
+
/** ISO-8601 creation timestamp. */
|
|
16
|
+
readonly createdAt: string;
|
|
17
|
+
/** Optional note from the editor describing this change. */
|
|
18
|
+
readonly message?: string;
|
|
19
|
+
}
|
|
20
|
+
//# sourceMappingURL=version.d.ts.map
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runtime mode policy.
|
|
3
|
+
*
|
|
4
|
+
* AgentDocStore is offline by default and you opt OUT, never in. `offline` is not
|
|
5
|
+
* a decoration: this module is the single place that decides whether a given
|
|
6
|
+
* (mode, host, provider) combination is allowed, and the CLI refuses to start
|
|
7
|
+
* when it says no. The complementary runtime guard — severing actual outbound
|
|
8
|
+
* sockets — lives in the CLI, because it patches process globals; this module
|
|
9
|
+
* stays pure so it can be unit-tested and reused by a fork's own host process.
|
|
10
|
+
*/
|
|
11
|
+
import type { Capabilities } from './spi/capabilities.js';
|
|
12
|
+
/**
|
|
13
|
+
* `offline` — loopback only, local provider only, outbound egress fused.
|
|
14
|
+
* `networked` — the operator has explicitly accepted egress and remote binding.
|
|
15
|
+
*/
|
|
16
|
+
export type RuntimeMode = 'offline' | 'networked';
|
|
17
|
+
export declare const RUNTIME_MODES: readonly RuntimeMode[];
|
|
18
|
+
export declare function isRuntimeMode(value: unknown): value is RuntimeMode;
|
|
19
|
+
/**
|
|
20
|
+
* True when a host names this machine and nothing else.
|
|
21
|
+
*
|
|
22
|
+
* `localhost` is included because that is what an operator types, even though it
|
|
23
|
+
* resolves through DNS. Deliberately excluded: `0.0.0.0` and `::` — as a bind
|
|
24
|
+
* address they mean *every* interface, which is exactly what offline mode is
|
|
25
|
+
* meant to prevent.
|
|
26
|
+
*/
|
|
27
|
+
export declare function isLoopbackHost(host: string): boolean;
|
|
28
|
+
export interface OfflinePolicyInput {
|
|
29
|
+
mode: RuntimeMode;
|
|
30
|
+
/** The address the server will bind, when it will bind one. */
|
|
31
|
+
host?: string;
|
|
32
|
+
/**
|
|
33
|
+
* The operator has explicitly accepted INBOUND exposure on a non-loopback
|
|
34
|
+
* address. Orthogonal to {@link RuntimeMode}: a container binding `0.0.0.0`
|
|
35
|
+
* inside its own network namespace is reachable only through a port mapping
|
|
36
|
+
* the operator chose, and says nothing about egress. Keeping the two separate
|
|
37
|
+
* is what lets a Docker deployment stay offline (fused egress, local store)
|
|
38
|
+
* while still serving.
|
|
39
|
+
*/
|
|
40
|
+
expose?: boolean;
|
|
41
|
+
/** Capabilities of the provider that is about to be used. */
|
|
42
|
+
capabilities?: Pick<Capabilities, 'requiresNetwork'>;
|
|
43
|
+
/** Provider name, for a message that points at the actual culprit. */
|
|
44
|
+
providerName?: string;
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* Throw {@link OfflineViolationError} when the requested configuration would
|
|
48
|
+
* break the offline guarantee. A no-op in `networked` mode — that mode's whole
|
|
49
|
+
* purpose is to be permitted.
|
|
50
|
+
*
|
|
51
|
+
* Two independent clauses:
|
|
52
|
+
* - the bind address, unless `expose` was set (inbound exposure);
|
|
53
|
+
* - the provider's own network requirement (outbound egress), which `expose`
|
|
54
|
+
* deliberately does NOT waive.
|
|
55
|
+
*/
|
|
56
|
+
export declare function assertOfflinePolicy(input: OfflinePolicyInput): void;
|
|
57
|
+
//# sourceMappingURL=offline.d.ts.map
|
package/dist/offline.js
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runtime mode policy.
|
|
3
|
+
*
|
|
4
|
+
* AgentDocStore is offline by default and you opt OUT, never in. `offline` is not
|
|
5
|
+
* a decoration: this module is the single place that decides whether a given
|
|
6
|
+
* (mode, host, provider) combination is allowed, and the CLI refuses to start
|
|
7
|
+
* when it says no. The complementary runtime guard — severing actual outbound
|
|
8
|
+
* sockets — lives in the CLI, because it patches process globals; this module
|
|
9
|
+
* stays pure so it can be unit-tested and reused by a fork's own host process.
|
|
10
|
+
*/
|
|
11
|
+
import { OfflineViolationError } from './errors.js';
|
|
12
|
+
export const RUNTIME_MODES = ['offline', 'networked'];
|
|
13
|
+
export function isRuntimeMode(value) {
|
|
14
|
+
return value === 'offline' || value === 'networked';
|
|
15
|
+
}
|
|
16
|
+
/** IPv4 loopback is the whole 127.0.0.0/8 block, not just 127.0.0.1. */
|
|
17
|
+
const IPV4_LOOPBACK = /^127(?:\.(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)){3}$/;
|
|
18
|
+
/**
|
|
19
|
+
* True when a host names this machine and nothing else.
|
|
20
|
+
*
|
|
21
|
+
* `localhost` is included because that is what an operator types, even though it
|
|
22
|
+
* resolves through DNS. Deliberately excluded: `0.0.0.0` and `::` — as a bind
|
|
23
|
+
* address they mean *every* interface, which is exactly what offline mode is
|
|
24
|
+
* meant to prevent.
|
|
25
|
+
*/
|
|
26
|
+
export function isLoopbackHost(host) {
|
|
27
|
+
const h = host
|
|
28
|
+
.trim()
|
|
29
|
+
.toLowerCase()
|
|
30
|
+
.replace(/^\[|\]$/g, '');
|
|
31
|
+
if (h === 'localhost' || h === '::1' || h === '0:0:0:0:0:0:0:1')
|
|
32
|
+
return true;
|
|
33
|
+
return IPV4_LOOPBACK.test(h);
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Throw {@link OfflineViolationError} when the requested configuration would
|
|
37
|
+
* break the offline guarantee. A no-op in `networked` mode — that mode's whole
|
|
38
|
+
* purpose is to be permitted.
|
|
39
|
+
*
|
|
40
|
+
* Two independent clauses:
|
|
41
|
+
* - the bind address, unless `expose` was set (inbound exposure);
|
|
42
|
+
* - the provider's own network requirement (outbound egress), which `expose`
|
|
43
|
+
* deliberately does NOT waive.
|
|
44
|
+
*/
|
|
45
|
+
export function assertOfflinePolicy(input) {
|
|
46
|
+
if (input.mode === 'networked')
|
|
47
|
+
return;
|
|
48
|
+
if (input.host !== undefined && input.expose !== true && !isLoopbackHost(input.host)) {
|
|
49
|
+
throw new OfflineViolationError(`Offline mode refuses to bind '${input.host}': only loopback addresses are permitted.`, 'host', 'Pass --expose to accept inbound connections while staying offline, or ' +
|
|
50
|
+
'--networked to also allow egress.');
|
|
51
|
+
}
|
|
52
|
+
if (input.capabilities?.requiresNetwork === true) {
|
|
53
|
+
const name = input.providerName ?? 'the configured provider';
|
|
54
|
+
throw new OfflineViolationError(`Offline mode refuses provider '${name}': it declares requiresNetwork=true.`, 'provider', 'Use a local provider (fs, memory), or pass --networked to allow egress. ' +
|
|
55
|
+
'--expose does not waive this: it governs inbound only.');
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
//# sourceMappingURL=offline.js.map
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Dependency-free credential scanner.
|
|
3
|
+
*
|
|
4
|
+
* `scan` detects secrets and credentials in text content by matching known
|
|
5
|
+
* patterns (PEM keys, AWS keys, JWTs, bearer tokens, platform tokens, generic
|
|
6
|
+
* assignments, connection strings). `redact` replaces detected spans with
|
|
7
|
+
* placeholder tags.
|
|
8
|
+
*
|
|
9
|
+
* Design constraints:
|
|
10
|
+
* - Every regex quantifier is bounded to prevent catastrophic backtracking.
|
|
11
|
+
* - The generic-assignment pattern is tuned to skip obvious placeholders.
|
|
12
|
+
* - No external dependencies.
|
|
13
|
+
*/
|
|
14
|
+
/** A single credential finding within scanned content. */
|
|
15
|
+
export interface Finding {
|
|
16
|
+
/** Credential type identifier (e.g. `'pem-private-key'`, `'aws-access-key'`). */
|
|
17
|
+
readonly type: string;
|
|
18
|
+
/** 1-based line number where the finding starts. */
|
|
19
|
+
readonly line: number;
|
|
20
|
+
/** Absolute byte offset of the match start within `content`. */
|
|
21
|
+
readonly start: number;
|
|
22
|
+
/** Absolute byte offset of the match end (exclusive) within `content`. */
|
|
23
|
+
readonly end: number;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Scan `content` for credential-like patterns.
|
|
27
|
+
*
|
|
28
|
+
* Returns an immutable array of {@link Finding} objects sorted by `start`
|
|
29
|
+
* offset. Overlapping findings are deduplicated: when two patterns match
|
|
30
|
+
* overlapping spans, the longer match wins.
|
|
31
|
+
*
|
|
32
|
+
* @param content - The text to scan (source code, config files, etc.).
|
|
33
|
+
* @returns Detected credential findings, sorted by offset.
|
|
34
|
+
*/
|
|
35
|
+
export declare function scan(content: string): readonly Finding[];
|
|
36
|
+
/**
|
|
37
|
+
* Replace each finding span in `content` with `[REDACTED:<type>]`.
|
|
38
|
+
*
|
|
39
|
+
* Replacements are applied right-to-left so that earlier offsets remain valid
|
|
40
|
+
* even when replacement text differs in length from the original span.
|
|
41
|
+
*
|
|
42
|
+
* @param content - The original text.
|
|
43
|
+
* @param findings - Findings from {@link scan} (or a subset).
|
|
44
|
+
* @returns The redacted text.
|
|
45
|
+
*/
|
|
46
|
+
export declare function redact(content: string, findings: readonly Finding[]): string;
|
|
47
|
+
//# sourceMappingURL=scanner.d.ts.map
|
package/dist/scanner.js
ADDED
|
@@ -0,0 +1,294 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Dependency-free credential scanner.
|
|
3
|
+
*
|
|
4
|
+
* `scan` detects secrets and credentials in text content by matching known
|
|
5
|
+
* patterns (PEM keys, AWS keys, JWTs, bearer tokens, platform tokens, generic
|
|
6
|
+
* assignments, connection strings). `redact` replaces detected spans with
|
|
7
|
+
* placeholder tags.
|
|
8
|
+
*
|
|
9
|
+
* Design constraints:
|
|
10
|
+
* - Every regex quantifier is bounded to prevent catastrophic backtracking.
|
|
11
|
+
* - The generic-assignment pattern is tuned to skip obvious placeholders.
|
|
12
|
+
* - No external dependencies.
|
|
13
|
+
*/
|
|
14
|
+
/**
|
|
15
|
+
* Placeholder values that the generic-assignment pattern must skip.
|
|
16
|
+
* Lowercased for comparison.
|
|
17
|
+
*/
|
|
18
|
+
const PLACEHOLDER_VALUES = new Set([
|
|
19
|
+
'',
|
|
20
|
+
'changeme',
|
|
21
|
+
'xxx',
|
|
22
|
+
'xxxx',
|
|
23
|
+
'xxxxx',
|
|
24
|
+
'example',
|
|
25
|
+
'redacted',
|
|
26
|
+
'todo',
|
|
27
|
+
'fixme',
|
|
28
|
+
'placeholder',
|
|
29
|
+
'your-password',
|
|
30
|
+
'your-secret',
|
|
31
|
+
'your-token',
|
|
32
|
+
'your-api-key',
|
|
33
|
+
]);
|
|
34
|
+
/** Returns true when a value looks like a placeholder rather than a real secret. */
|
|
35
|
+
function isPlaceholder(value) {
|
|
36
|
+
const trimmed = value.replace(/^["'`]+|["'`]+$/g, '').trim();
|
|
37
|
+
if (trimmed.length === 0)
|
|
38
|
+
return true;
|
|
39
|
+
const lower = trimmed.toLowerCase();
|
|
40
|
+
if (PLACEHOLDER_VALUES.has(lower))
|
|
41
|
+
return true;
|
|
42
|
+
// Template / env-var references: ${VAR}, $VAR, {{VAR}}, <your-password>
|
|
43
|
+
if (/^\$\{[^}]{1,80}\}$/.test(trimmed))
|
|
44
|
+
return true;
|
|
45
|
+
if (/^\$[A-Z_][A-Z0-9_]{0,80}$/.test(trimmed))
|
|
46
|
+
return true;
|
|
47
|
+
if (/^\{\{[^}]{1,80}\}\}$/.test(trimmed))
|
|
48
|
+
return true;
|
|
49
|
+
if (/^<[^>]{1,80}>$/.test(trimmed))
|
|
50
|
+
return true;
|
|
51
|
+
return false;
|
|
52
|
+
}
|
|
53
|
+
const PATTERNS = [
|
|
54
|
+
{
|
|
55
|
+
type: 'aws-access-key',
|
|
56
|
+
// AWS access key IDs start with AKIA (long-term) or ASIA (temporary/STS).
|
|
57
|
+
// Followed by exactly 16 uppercase alphanumeric characters.
|
|
58
|
+
regex: /(?:^|[^A-Z0-9])(?:AKIA|ASIA)[A-Z0-9]{16}(?=[^A-Z0-9]|$)/gm,
|
|
59
|
+
comment: 'AWS access key IDs (AKIA/ASIA prefix + 16 uppercase alphanumerics)',
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
type: 'aws-secret-key',
|
|
63
|
+
// AWS secret access keys are 40 characters of base64-like characters.
|
|
64
|
+
// Anchored behind a keyword to reduce false positives.
|
|
65
|
+
regex: /(?:aws_secret_access_key|secret_access_key|SecretAccessKey)[\s]*[=:"']\s*[A-Za-z0-9/+=]{40}(?=[^A-Za-z0-9/+=]|$)/gi,
|
|
66
|
+
comment: 'AWS secret access key shapes (keyword + 40 base64 chars)',
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
type: 'jwt',
|
|
70
|
+
// JWTs are three dot-separated base64url segments. Header is short (< 512),
|
|
71
|
+
// payload moderate (< 4096), signature moderate (< 1024).
|
|
72
|
+
regex: /eyJ[A-Za-z0-9_-]{4,512}\.eyJ[A-Za-z0-9_-]{4,4096}\.[A-Za-z0-9_-]{4,1024}/g,
|
|
73
|
+
comment: 'JSON Web Tokens (three base64url dot-separated segments starting with eyJ)',
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
type: 'bearer-token',
|
|
77
|
+
// Bearer or OAuth tokens in Authorization-style headers.
|
|
78
|
+
regex: /[Bb]earer\s+[A-Za-z0-9_\-.~+/]{20,1024}/g,
|
|
79
|
+
comment: 'Bearer / OAuth tokens in authorization headers or config',
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
type: 'github-token',
|
|
83
|
+
// GitHub PAT / OAuth / user / server / refresh tokens.
|
|
84
|
+
regex: /gh[pousr]_[A-Za-z0-9_]{36,255}/g,
|
|
85
|
+
comment: 'GitHub tokens (ghp_, gho_, ghu_, ghs_, ghr_ prefixes)',
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
type: 'gitlab-token',
|
|
89
|
+
// GitLab personal access tokens.
|
|
90
|
+
regex: /glpat-[A-Za-z0-9_-]{20,255}/g,
|
|
91
|
+
comment: 'GitLab personal access tokens (glpat- prefix)',
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
type: 'slack-token',
|
|
95
|
+
// Slack bot, app, user, workspace, and refresh tokens.
|
|
96
|
+
regex: /xox[abprs]-[A-Za-z0-9-]{10,255}/g,
|
|
97
|
+
comment: 'Slack tokens (xoxb-, xoxa-, xoxp-, xoxr-, xoxs- prefixes)',
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
type: 'connection-string-password',
|
|
101
|
+
// scheme://user:password@host — captures the password portion.
|
|
102
|
+
// Password bounded to 256 chars; scheme, user, and host also bounded.
|
|
103
|
+
regex: /[a-zA-Z][a-zA-Z0-9+.-]{1,30}:\/\/[^\s:@]{1,128}:([^\s@]{1,256})@[^\s]{1,512}/g,
|
|
104
|
+
comment: 'Connection-string embedded passwords (scheme://user:password@host)',
|
|
105
|
+
},
|
|
106
|
+
];
|
|
107
|
+
/**
|
|
108
|
+
* PEM / OpenSSH private key blocks: a BEGIN line, 1 to 16,384 characters, then
|
|
109
|
+
* the first END line after them. The kind named on the two lines need not
|
|
110
|
+
* agree. This finds what
|
|
111
|
+
* `/-----BEGIN …PRIVATE KEY-----[\s\S]{1,16384}?-----END …PRIVATE KEY-----/g`
|
|
112
|
+
* finds, in one pass: that regex tried up to 16,384 positions for an END line
|
|
113
|
+
* after every BEGIN line, so text made of BEGIN lines alone took about a
|
|
114
|
+
* second per megabyte.
|
|
115
|
+
*/
|
|
116
|
+
const PEM_BEGIN_REGEX = /-----BEGIN (?:RSA |DSA |EC |OPENSSH |ENCRYPTED )?PRIVATE KEY-----/g;
|
|
117
|
+
const PEM_END_REGEX = /-----END (?:RSA |DSA |EC |OPENSSH |ENCRYPTED )?PRIVATE KEY-----/g;
|
|
118
|
+
const PEM_MAX_BODY = 16_384;
|
|
119
|
+
function pemPrivateKeyBlocks(content) {
|
|
120
|
+
// Every place an END line starts, in order; overlapping ones included.
|
|
121
|
+
const ends = [];
|
|
122
|
+
PEM_END_REGEX.lastIndex = 0;
|
|
123
|
+
for (let m = PEM_END_REGEX.exec(content); m !== null; m = PEM_END_REGEX.exec(content)) {
|
|
124
|
+
ends.push({ start: m.index, end: m.index + m[0].length });
|
|
125
|
+
PEM_END_REGEX.lastIndex = m.index + 1;
|
|
126
|
+
}
|
|
127
|
+
const blocks = [];
|
|
128
|
+
let next = 0; // the first END line that could still close a block
|
|
129
|
+
PEM_BEGIN_REGEX.lastIndex = 0;
|
|
130
|
+
for (let m = PEM_BEGIN_REGEX.exec(content); m !== null; m = PEM_BEGIN_REGEX.exec(content)) {
|
|
131
|
+
const bodyStart = m.index + m[0].length;
|
|
132
|
+
// The body has at least one character, so an END must start after it.
|
|
133
|
+
while (next < ends.length && (ends[next]?.start ?? Infinity) <= bodyStart)
|
|
134
|
+
next++;
|
|
135
|
+
const end = ends[next];
|
|
136
|
+
if (end !== undefined && end.start - bodyStart <= PEM_MAX_BODY) {
|
|
137
|
+
blocks.push({ start: m.index, end: end.end });
|
|
138
|
+
PEM_BEGIN_REGEX.lastIndex = end.end;
|
|
139
|
+
}
|
|
140
|
+
else {
|
|
141
|
+
// As the regex did: look for the next block from one character on.
|
|
142
|
+
PEM_BEGIN_REGEX.lastIndex = m.index + 1;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
return blocks;
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* Generic assignment pattern built separately because it requires a
|
|
149
|
+
* placeholder filter post-match.
|
|
150
|
+
*
|
|
151
|
+
* Matches: password = "value", api_key: 'value', token=value, and forms where
|
|
152
|
+
* the keyword is embedded in a longer identifier such as
|
|
153
|
+
* `aws_secret_access_key = ...`, `client_secret_id: ...` or `AUTH_TOKEN_V2=...`.
|
|
154
|
+
* The trailing `[A-Za-z0-9_.-]{0,64}` is what admits those: without it the
|
|
155
|
+
* keyword had to sit immediately before the separator, which missed the most
|
|
156
|
+
* common real-world config naming conventions. For a credential scanner a
|
|
157
|
+
* false negative is worse than a false positive, and the placeholder filter
|
|
158
|
+
* below absorbs the extra noise.
|
|
159
|
+
*
|
|
160
|
+
* That identifier tail is bounded because each keyword starts a match of its
|
|
161
|
+
* own: unbounded, every "token" in "tokentoken…" re-read the rest of the run,
|
|
162
|
+
* so the scan took time growing with the square of the input (256 KB took
|
|
163
|
+
* 13 s).
|
|
164
|
+
*
|
|
165
|
+
* The keywords and the value are captured so the placeholder filter can
|
|
166
|
+
* inspect the value group.
|
|
167
|
+
*/
|
|
168
|
+
const GENERIC_ASSIGNMENT_REGEX = /(?:password|passwd|secret|api_key|apikey|token)[A-Za-z0-9_.-]{0,64}\s*[=:]\s*["'`]?([^\s"'`]{1,512})["'`]?/gi;
|
|
169
|
+
// ---------------------------------------------------------------------------
|
|
170
|
+
// Implementation
|
|
171
|
+
// ---------------------------------------------------------------------------
|
|
172
|
+
/**
|
|
173
|
+
* The offset of every newline in `content`, in order. Found once per scan, so
|
|
174
|
+
* each finding's line is a binary search: counting from the start of the
|
|
175
|
+
* content for every finding made a finding on every line take time growing
|
|
176
|
+
* with the square of the content.
|
|
177
|
+
*/
|
|
178
|
+
function newlineOffsets(content) {
|
|
179
|
+
const offsets = [];
|
|
180
|
+
for (let i = content.indexOf('\n'); i !== -1; i = content.indexOf('\n', i + 1)) {
|
|
181
|
+
offsets.push(i);
|
|
182
|
+
}
|
|
183
|
+
return offsets;
|
|
184
|
+
}
|
|
185
|
+
/**
|
|
186
|
+
* The 1-based line number of `offset`: one more than the newlines before it.
|
|
187
|
+
*/
|
|
188
|
+
function lineNumberAt(newlines, offset) {
|
|
189
|
+
let lo = 0;
|
|
190
|
+
let hi = newlines.length;
|
|
191
|
+
while (lo < hi) {
|
|
192
|
+
const mid = (lo + hi) >>> 1;
|
|
193
|
+
if ((newlines[mid] ?? Infinity) < offset)
|
|
194
|
+
lo = mid + 1;
|
|
195
|
+
else
|
|
196
|
+
hi = mid;
|
|
197
|
+
}
|
|
198
|
+
return lo + 1;
|
|
199
|
+
}
|
|
200
|
+
/**
|
|
201
|
+
* Scan `content` for credential-like patterns.
|
|
202
|
+
*
|
|
203
|
+
* Returns an immutable array of {@link Finding} objects sorted by `start`
|
|
204
|
+
* offset. Overlapping findings are deduplicated: when two patterns match
|
|
205
|
+
* overlapping spans, the longer match wins.
|
|
206
|
+
*
|
|
207
|
+
* @param content - The text to scan (source code, config files, etc.).
|
|
208
|
+
* @returns Detected credential findings, sorted by offset.
|
|
209
|
+
*/
|
|
210
|
+
export function scan(content) {
|
|
211
|
+
const raw = [];
|
|
212
|
+
let newlines;
|
|
213
|
+
const lineOf = (offset) => lineNumberAt((newlines ??= newlineOffsets(content)), offset);
|
|
214
|
+
for (const block of pemPrivateKeyBlocks(content)) {
|
|
215
|
+
raw.push({ type: 'pem-private-key', line: lineOf(block.start), ...block });
|
|
216
|
+
}
|
|
217
|
+
// Run each fixed pattern.
|
|
218
|
+
for (const pat of PATTERNS) {
|
|
219
|
+
pat.regex.lastIndex = 0;
|
|
220
|
+
let m;
|
|
221
|
+
while ((m = pat.regex.exec(content)) !== null) {
|
|
222
|
+
let matchStart = m.index;
|
|
223
|
+
let matchStr = m[0];
|
|
224
|
+
// For aws-access-key, the leading non-alphanumeric char is a lookaround
|
|
225
|
+
// workaround. Trim it from the match span if present.
|
|
226
|
+
if (pat.type === 'aws-access-key' && matchStr.length > 20) {
|
|
227
|
+
matchStart += matchStr.length - 20;
|
|
228
|
+
matchStr = matchStr.slice(matchStr.length - 20);
|
|
229
|
+
}
|
|
230
|
+
// For connection-string-password, narrow the finding to the password
|
|
231
|
+
// capture group (group 1).
|
|
232
|
+
if (pat.type === 'connection-string-password' && m[1] !== undefined) {
|
|
233
|
+
const pwdOffset = m[0].indexOf(':' + m[1] + '@');
|
|
234
|
+
if (pwdOffset !== -1) {
|
|
235
|
+
matchStart = m.index + pwdOffset + 1; // skip the ':'
|
|
236
|
+
matchStr = m[1];
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
raw.push({
|
|
240
|
+
type: pat.type,
|
|
241
|
+
line: lineOf(matchStart),
|
|
242
|
+
start: matchStart,
|
|
243
|
+
end: matchStart + matchStr.length,
|
|
244
|
+
});
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
// Run the generic assignment pattern with placeholder filtering.
|
|
248
|
+
GENERIC_ASSIGNMENT_REGEX.lastIndex = 0;
|
|
249
|
+
let gm;
|
|
250
|
+
while ((gm = GENERIC_ASSIGNMENT_REGEX.exec(content)) !== null) {
|
|
251
|
+
const value = gm[1] ?? '';
|
|
252
|
+
if (isPlaceholder(value))
|
|
253
|
+
continue;
|
|
254
|
+
raw.push({
|
|
255
|
+
type: 'generic-secret',
|
|
256
|
+
line: lineOf(gm.index),
|
|
257
|
+
start: gm.index,
|
|
258
|
+
end: gm.index + gm[0].length,
|
|
259
|
+
});
|
|
260
|
+
}
|
|
261
|
+
// Sort by start offset ascending, then by length descending for overlap tie-breaking.
|
|
262
|
+
raw.sort((a, b) => a.start - b.start || b.end - b.start - (a.end - a.start));
|
|
263
|
+
// Deduplicate overlapping findings: keep the first (longest at each offset).
|
|
264
|
+
const deduped = [];
|
|
265
|
+
let lastEnd = -1;
|
|
266
|
+
for (const f of raw) {
|
|
267
|
+
if (f.start >= lastEnd) {
|
|
268
|
+
deduped.push(f);
|
|
269
|
+
lastEnd = f.end;
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
return deduped;
|
|
273
|
+
}
|
|
274
|
+
/**
|
|
275
|
+
* Replace each finding span in `content` with `[REDACTED:<type>]`.
|
|
276
|
+
*
|
|
277
|
+
* Replacements are applied right-to-left so that earlier offsets remain valid
|
|
278
|
+
* even when replacement text differs in length from the original span.
|
|
279
|
+
*
|
|
280
|
+
* @param content - The original text.
|
|
281
|
+
* @param findings - Findings from {@link scan} (or a subset).
|
|
282
|
+
* @returns The redacted text.
|
|
283
|
+
*/
|
|
284
|
+
export function redact(content, findings) {
|
|
285
|
+
// Work on a copy of findings sorted by start offset descending (right-to-left).
|
|
286
|
+
const sorted = [...findings].sort((a, b) => b.start - a.start);
|
|
287
|
+
let result = content;
|
|
288
|
+
for (const f of sorted) {
|
|
289
|
+
const tag = `[REDACTED:${f.type}]`;
|
|
290
|
+
result = result.slice(0, f.start) + tag + result.slice(f.end);
|
|
291
|
+
}
|
|
292
|
+
return result;
|
|
293
|
+
}
|
|
294
|
+
//# sourceMappingURL=scanner.js.map
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import type { SearchDoc, SearchIndex, SearchQueryOptions, SearchResults } from '../spi/search.js';
|
|
2
|
+
/**
|
|
3
|
+
* The default {@link SearchIndex} implementation, backed by MiniSearch. Used by
|
|
4
|
+
* any provider whose {@link Capabilities.search} is `'core-fallback'`.
|
|
5
|
+
*
|
|
6
|
+
* `query` returns PUBLIC hits plus the viewer's own PRIVATE hits — a PRIVATE
|
|
7
|
+
* doc owned by someone else is never returned.
|
|
8
|
+
*/
|
|
9
|
+
export declare class CoreSearchIndex implements SearchIndex {
|
|
10
|
+
private mini;
|
|
11
|
+
constructor();
|
|
12
|
+
add(doc: SearchDoc): void;
|
|
13
|
+
update(doc: SearchDoc): void;
|
|
14
|
+
remove(documentId: string): void;
|
|
15
|
+
query(q: string, viewer: string | null, opts?: SearchQueryOptions): SearchResults;
|
|
16
|
+
snapshot(): string;
|
|
17
|
+
restore(snapshot: string): void;
|
|
18
|
+
clear(): void;
|
|
19
|
+
size(): number;
|
|
20
|
+
}
|
|
21
|
+
//# sourceMappingURL=CoreSearchIndex.d.ts.map
|