@zerowidth/workbench-sdk 2.1.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/nodes/json-to-xml/json-to-xml.config.json +2 -2
- package/nodes/json-to-yaml/json-to-yaml.config.json +2 -2
- package/nodes/json-to-yaml/json-to-yaml.tests.json +9 -0
- package/nodes/message/message.config.json +1 -1
- package/nodes/message/message.process.js +25 -2
- package/nodes/message/message.tests.json +85 -0
- package/nodes/object-merge/object-merge.config.json +2 -2
- package/nodes/object-merge/object-merge.process.js +6 -2
- package/nodes/object-merge/object-merge.tests.json +12 -0
- package/nodes/prompt-injection-guard/prompt-injection-guard.config.json +39 -0
- package/nodes/prompt-injection-guard/prompt-injection-guard.process.js +269 -0
- package/nodes/prompt-injection-guard/prompt-injection-guard.tests.json +205 -0
- package/nodes/string-template/string-template.config.json +2 -2
- package/nodes/string-template/string-template.process.js +33 -5
- package/nodes/string-template/string-template.tests.json +52 -0
- package/nodes/system-prompt/system-prompt.config.json +1 -1
- package/nodes/system-prompt/system-prompt.process.js +25 -2
- package/nodes/system-prompt/system-prompt.tests.json +113 -0
- package/package.json +1 -1
- package/src/index.js +112 -5
- package/src/utilities/mcp.js +25 -5
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
{
|
|
9
9
|
"name": "json",
|
|
10
10
|
"display_name": "JSON",
|
|
11
|
-
"type": "object",
|
|
12
|
-
"description": "The JSON object to convert to XML",
|
|
11
|
+
"type": "object or array of objects",
|
|
12
|
+
"description": "The JSON object, or array of objects, to convert to XML",
|
|
13
13
|
"required": true
|
|
14
14
|
},
|
|
15
15
|
{
|
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
{
|
|
9
9
|
"name": "json",
|
|
10
10
|
"display_name": "JSON",
|
|
11
|
-
"type": "object",
|
|
12
|
-
"description": "The JSON object to convert to YAML",
|
|
11
|
+
"type": "object or array of objects",
|
|
12
|
+
"description": "The JSON object, or array of objects, to convert to YAML",
|
|
13
13
|
"required": true
|
|
14
14
|
},
|
|
15
15
|
{
|
|
@@ -97,5 +97,14 @@
|
|
|
97
97
|
"expected": {
|
|
98
98
|
"yaml": "null"
|
|
99
99
|
}
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
"description": "Top-level array of objects",
|
|
103
|
+
"inputs": {
|
|
104
|
+
"json": [{"name": "Alice"}, {"name": "Bob"}]
|
|
105
|
+
},
|
|
106
|
+
"expected": {
|
|
107
|
+
"yaml": "- name: Alice\n- name: Bob"
|
|
108
|
+
}
|
|
100
109
|
}
|
|
101
110
|
]
|
|
@@ -2,6 +2,23 @@
|
|
|
2
2
|
* Process function for the Message node.
|
|
3
3
|
* Outputs a message object, either from the input or from the settings.
|
|
4
4
|
*/
|
|
5
|
+
/**
|
|
6
|
+
* Render a variable value for injection into message text.
|
|
7
|
+
* Strings pass through untouched; everything else is JSON encoded so that
|
|
8
|
+
* nested objects and arrays read as data instead of "[object Object]".
|
|
9
|
+
*/
|
|
10
|
+
const renderVariable = (value) => {
|
|
11
|
+
if (value === null || value === undefined) return "";
|
|
12
|
+
if (typeof value === "string") return value;
|
|
13
|
+
try {
|
|
14
|
+
return JSON.stringify(value, null, 2);
|
|
15
|
+
} catch {
|
|
16
|
+
// Circular structures (or anything else JSON can't encode) fall back to
|
|
17
|
+
// the default coercion rather than failing the whole message.
|
|
18
|
+
return String(value);
|
|
19
|
+
}
|
|
20
|
+
};
|
|
21
|
+
|
|
5
22
|
export default async ({inputs, settings, config}) => {
|
|
6
23
|
|
|
7
24
|
// If an input value is provided, use it; otherwise use the value from settings
|
|
@@ -17,6 +34,12 @@ export default async ({inputs, settings, config}) => {
|
|
|
17
34
|
if(!inputs.variables) {
|
|
18
35
|
inputs.variables = [];
|
|
19
36
|
}
|
|
37
|
+
|
|
38
|
+
// A single connection can deliver an array of key-value objects, so flatten
|
|
39
|
+
// one level before looking keys up.
|
|
40
|
+
const variables = (Array.isArray(inputs.variables) ? inputs.variables : [inputs.variables])
|
|
41
|
+
.flatMap(entry => Array.isArray(entry) ? entry : [entry])
|
|
42
|
+
.filter(entry => entry !== null && typeof entry === "object");
|
|
20
43
|
|
|
21
44
|
// if we have variables and text content, we need to replace the text content with the variables
|
|
22
45
|
// do we have a text content item and what index is it
|
|
@@ -25,9 +48,9 @@ export default async ({inputs, settings, config}) => {
|
|
|
25
48
|
message.content[textContentIndex].text = message.content[textContentIndex].text.replace(/\{\{(.*?)\}\}/g, (match, p1) => {
|
|
26
49
|
|
|
27
50
|
// look for a variable with the key p1
|
|
28
|
-
let variable =
|
|
51
|
+
let variable = variables.find(variable => Object.keys(variable).find(key => key === p1));
|
|
29
52
|
if(variable) {
|
|
30
|
-
return variable[p1];
|
|
53
|
+
return renderVariable(variable[p1]);
|
|
31
54
|
}
|
|
32
55
|
return match;
|
|
33
56
|
});
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"description": "String variable is injected as-is",
|
|
4
|
+
"inputs": {
|
|
5
|
+
"variables": [
|
|
6
|
+
{
|
|
7
|
+
"name": "Ada"
|
|
8
|
+
}
|
|
9
|
+
]
|
|
10
|
+
},
|
|
11
|
+
"settings": {
|
|
12
|
+
"role": "user",
|
|
13
|
+
"content": "Hello {{name}}!"
|
|
14
|
+
},
|
|
15
|
+
"expected": {
|
|
16
|
+
"message": {
|
|
17
|
+
"role": "user",
|
|
18
|
+
"content": [
|
|
19
|
+
{
|
|
20
|
+
"type": "text",
|
|
21
|
+
"text": "Hello Ada!"
|
|
22
|
+
}
|
|
23
|
+
]
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
"description": "Nested object variable is JSON stringified, not [object Object]",
|
|
29
|
+
"inputs": {
|
|
30
|
+
"variables": [
|
|
31
|
+
{
|
|
32
|
+
"order": {
|
|
33
|
+
"id": 7,
|
|
34
|
+
"paid": true
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
},
|
|
39
|
+
"settings": {
|
|
40
|
+
"role": "user",
|
|
41
|
+
"content": "Order: {{order}}"
|
|
42
|
+
},
|
|
43
|
+
"expected": {
|
|
44
|
+
"message": {
|
|
45
|
+
"role": "user",
|
|
46
|
+
"content": [
|
|
47
|
+
{
|
|
48
|
+
"type": "text",
|
|
49
|
+
"text": "Order: {\n \"id\": 7,\n \"paid\": true\n}"
|
|
50
|
+
}
|
|
51
|
+
]
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
"description": "Array of objects delivered as a single connection is flattened",
|
|
57
|
+
"inputs": {
|
|
58
|
+
"variables": [
|
|
59
|
+
[
|
|
60
|
+
{
|
|
61
|
+
"first": "Ada"
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"last": "Lovelace"
|
|
65
|
+
}
|
|
66
|
+
]
|
|
67
|
+
]
|
|
68
|
+
},
|
|
69
|
+
"settings": {
|
|
70
|
+
"role": "user",
|
|
71
|
+
"content": "{{first}} {{last}}"
|
|
72
|
+
},
|
|
73
|
+
"expected": {
|
|
74
|
+
"message": {
|
|
75
|
+
"role": "user",
|
|
76
|
+
"content": [
|
|
77
|
+
{
|
|
78
|
+
"type": "text",
|
|
79
|
+
"text": "Ada Lovelace"
|
|
80
|
+
}
|
|
81
|
+
]
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
]
|
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
{
|
|
9
9
|
"name": "objects",
|
|
10
10
|
"display_name": "Objects",
|
|
11
|
-
"type": "object",
|
|
12
|
-
"description": "Objects to merge (later objects override earlier ones)",
|
|
11
|
+
"type": "object or array of objects",
|
|
12
|
+
"description": "Objects to merge, individually or as arrays of objects (later objects override earlier ones)",
|
|
13
13
|
"required": true,
|
|
14
14
|
"allow_multiple": true
|
|
15
15
|
},
|
|
@@ -2,8 +2,12 @@ export default async ({ inputs, settings, config }) => {
|
|
|
2
2
|
const objects = inputs.objects;
|
|
3
3
|
const arrayMode = inputs.array_mode ?? "replace";
|
|
4
4
|
|
|
5
|
-
// Normalize to array
|
|
6
|
-
|
|
5
|
+
// Normalize to a flat array. A single connection can deliver an array of
|
|
6
|
+
// objects, and a multi-connection input can deliver arrays alongside plain
|
|
7
|
+
// objects, so flatten one level before filtering.
|
|
8
|
+
const objectList = (Array.isArray(objects) ? objects : [objects]).flatMap(
|
|
9
|
+
(entry) => (Array.isArray(entry) ? entry : [entry])
|
|
10
|
+
);
|
|
7
11
|
|
|
8
12
|
// Filter out non-objects
|
|
9
13
|
const validObjects = objectList.filter(
|
|
@@ -142,5 +142,17 @@
|
|
|
142
142
|
"expected": {
|
|
143
143
|
"merged": {"level1": {"level2": {"level3": {"a": 1, "b": 2, "c": 3}}}}
|
|
144
144
|
}
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
"description": "Array of objects delivered as a single connection is flattened",
|
|
148
|
+
"inputs": {
|
|
149
|
+
"objects": [
|
|
150
|
+
[{"a": 1}, {"b": 2}],
|
|
151
|
+
{"c": 3}
|
|
152
|
+
]
|
|
153
|
+
},
|
|
154
|
+
"expected": {
|
|
155
|
+
"merged": {"a": 1, "b": 2, "c": 3}
|
|
156
|
+
}
|
|
145
157
|
}
|
|
146
158
|
]
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
{
|
|
2
|
+
"display_name": "Prompt Injection Guard",
|
|
3
|
+
"tagline": "Withhold prompt injections",
|
|
4
|
+
"description": "Scans user messages in a conversation for prompt-injection attempts — instruction overrides, jailbreak personas, system-prompt extraction, safety-bypass requests, template-marker spoofing, typoglycemia-scrambled variants, and base64-smuggled payloads. Flagged messages are replaced with a neutral withholding note; the model naturally tells the user the message couldn't be delivered and continues with the task. Detection is heuristic (patterns inspired by the OWASP LLM Prompt Injection Prevention Cheat Sheet) and runs entirely offline — no model call, no keys.",
|
|
5
|
+
"icon": "shield-exclamation",
|
|
6
|
+
"category": "messages",
|
|
7
|
+
"inputs": [
|
|
8
|
+
{
|
|
9
|
+
"name": "messages",
|
|
10
|
+
"display_name": "Messages",
|
|
11
|
+
"type": "conversation",
|
|
12
|
+
"description": "Array of message objects to scan. Only user-role messages are analyzed; assistant, tool, and system messages pass through untouched.",
|
|
13
|
+
"required": true
|
|
14
|
+
}
|
|
15
|
+
],
|
|
16
|
+
"outputs": [
|
|
17
|
+
{
|
|
18
|
+
"name": "messages",
|
|
19
|
+
"display_name": "Messages",
|
|
20
|
+
"type": "conversation",
|
|
21
|
+
"description": "The conversation with any flagged user messages replaced by the injection notice"
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"name": "flagged_count",
|
|
25
|
+
"display_name": "Flagged Count",
|
|
26
|
+
"type": "number",
|
|
27
|
+
"description": "Number of user messages that were flagged and replaced in this pass"
|
|
28
|
+
}
|
|
29
|
+
],
|
|
30
|
+
"settings": [
|
|
31
|
+
{
|
|
32
|
+
"name": "notice",
|
|
33
|
+
"display_name": "Notice",
|
|
34
|
+
"type": "string",
|
|
35
|
+
"description": "Custom replacement text for withheld messages. Leave unset to use the built-in neutral withholding note. Avoid instructional or authority-claiming text — models treat user-role commands from a claimed security layer as suspect.",
|
|
36
|
+
"required": false
|
|
37
|
+
}
|
|
38
|
+
]
|
|
39
|
+
}
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
// Heuristic prompt-injection detector. Patterns and layering follow the
|
|
2
|
+
// OWASP LLM Prompt Injection Prevention Cheat Sheet: normalization to
|
|
3
|
+
// defeat trivial obfuscation, direct-injection pattern families, fuzzy
|
|
4
|
+
// token matching for typoglycemia-scrambled variants, and decoding of
|
|
5
|
+
// base64 runs to catch encoding-smuggled payloads. Detection is
|
|
6
|
+
// deliberately conservative — patterns require an instruction aimed at
|
|
7
|
+
// the assistant, not mere mention of a keyword — because a false
|
|
8
|
+
// positive silently eats a legitimate user message.
|
|
9
|
+
|
|
10
|
+
// The replacement text is deliberately a neutral, third-person
|
|
11
|
+
// withholding note — no channel markers, no meta-authority framing,
|
|
12
|
+
// no instructions aimed at the model. A user-role message that
|
|
13
|
+
// *commands* the model while claiming to be a security layer reads
|
|
14
|
+
// exactly like an injection itself; live A/B against Claude models
|
|
15
|
+
// showed the instructional variant triggering "this looks like a
|
|
16
|
+
// simulated system prompt" skepticism, while this neutral note gets
|
|
17
|
+
// a clean "your message couldn't be delivered, please rephrase"
|
|
18
|
+
// response. The marker prefix doubles as the idempotency check.
|
|
19
|
+
const NOTICE_MARKER = "(Message withheld:";
|
|
20
|
+
|
|
21
|
+
const DEFAULT_NOTICE =
|
|
22
|
+
NOTICE_MARKER +
|
|
23
|
+
" this workspace's prompt-injection filter flagged the original" +
|
|
24
|
+
" content of this message, so it was not delivered. The original" +
|
|
25
|
+
" text is unavailable.)";
|
|
26
|
+
|
|
27
|
+
// ── Normalization ──────────────────────────────────────────────────
|
|
28
|
+
// NFKC folds fullwidth/compatibility characters, invisible characters
|
|
29
|
+
// are stripped (zero-width joiners are a documented smuggling channel),
|
|
30
|
+
// then casefold + whitespace collapse so patterns see one canonical
|
|
31
|
+
// form regardless of spacing or capitalization games.
|
|
32
|
+
|
|
33
|
+
const INVISIBLE_CHARS = /[\u00AD\u200B\u200C\u200D\u2060\uFEFF]/g;
|
|
34
|
+
|
|
35
|
+
function normalize(text) {
|
|
36
|
+
return text
|
|
37
|
+
.normalize("NFKC")
|
|
38
|
+
.replace(INVISIBLE_CHARS, "")
|
|
39
|
+
.toLowerCase()
|
|
40
|
+
.replace(/\s+/g, " ")
|
|
41
|
+
.trim();
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// ── Pattern families ───────────────────────────────────────────────
|
|
45
|
+
// All patterns run against normalized text (lowercase, single-spaced).
|
|
46
|
+
|
|
47
|
+
const PATTERNS = [
|
|
48
|
+
// Instruction override: "ignore all previous instructions", ...
|
|
49
|
+
/(ignore|disregard|forget|discard|overrule|override)\s((all|any|the|your|my|every|each)\s)*((previous|prior|earlier|above|preceding|original|initial|system|these|those)\s)+(instructions?|prompts?|rules?|directives?|guidelines?|context|messages?|programming|training|constraints?)/,
|
|
50
|
+
// "ignore everything above" / "forget your training"
|
|
51
|
+
/(ignore|disregard)\severything\s(above|before|else|prior)/,
|
|
52
|
+
/forget\s(all\s)?(your|the)\s(instructions?|rules?|training|programming|guidelines?)/,
|
|
53
|
+
// Replacement-instruction framing: "your new instructions are:"
|
|
54
|
+
/(new|updated|revised|real|actual|true)\s(system\s)?(instructions?|prompt|rules?|directives?)\s*(:|are\s|follow)/,
|
|
55
|
+
// Spoofed template / role markers: "[system]", "<sys>", "[INST]"
|
|
56
|
+
/(\[|<)\s*\/?\s*(system|sys|inst)\s*(\]|>)/,
|
|
57
|
+
// Jailbreak personas aimed at the assistant
|
|
58
|
+
/(you\sare\snow|you're\snow|youre\snow|from\snow\son\syou\sare)\s((in|into|the)\s)?(developer|dev|god|dan|jailbreak|jailbroken|unrestricted|unfiltered|admin|root)\s?mode/,
|
|
59
|
+
/\bdan\smode\b|\bdo\sanything\snow\b|\bact\sas\sdan\b|\bjailbreak\smode\b|\bjailbroken\b/,
|
|
60
|
+
/you\sare\s(now\s)?dan\b/,
|
|
61
|
+
// No-rules role hijack
|
|
62
|
+
/pretend\s(that\s)?(you|there)\s(are|have|were)\sno\s(rules?|restrictions?|guidelines?|limits?|filters?)/,
|
|
63
|
+
/you\shave\sno\s(rules?|restrictions?|guidelines?|limits?|filters?)/,
|
|
64
|
+
/(act|behave|respond|answer|reply)\s(as\sif\s)?([a-z']+\s){0,4}without\s(any\s)?(rules?|restrictions?|filters?|limitations?|guidelines?|censorship)/,
|
|
65
|
+
// System-prompt extraction (requires a system/hidden-style qualifier
|
|
66
|
+
// so "repeat the instructions for the recipe" stays clean)
|
|
67
|
+
/(reveal|show|print|display|output|repeat|recite|share|expose|leak|paste)\s(me\s)?(your|the)\s([a-z]+\s){0,2}system\s(prompt|instructions?|message)/,
|
|
68
|
+
/(reveal|show|print|display|output|repeat|recite|share|expose|leak|paste)\s(me\s)?(your|the)\s(hidden|initial|original|secret|internal|exact|full|complete|entire)\s(prompt|instructions?)/,
|
|
69
|
+
/(what\sis|what's|whats|tell\sme)\s(your|the)\s([a-z]+\s){0,2}system\s(prompt|instructions?|message)/,
|
|
70
|
+
/repeat\s(the\s)?(text|words?|everything|content|message)\s(above|before)/,
|
|
71
|
+
/starting\swith\s["']?(you\sare|i\sam)/,
|
|
72
|
+
// Safety-bypass requests
|
|
73
|
+
/(bypass|disable|remove|deactivate|circumvent|evade)\s((all|any|the|your)\s)*(safety|security|content|ethical|moral|alignment)\s?(measures?|filters?|checks?|guardrails?|guidelines?|restrictions?|protocols?|rules?|systems?)/,
|
|
74
|
+
/turn\soff\s((all|any|the|your)\s)*(safety|security|content)\s?(measures?|filters?|checks?|guardrails?|restrictions?)/,
|
|
75
|
+
];
|
|
76
|
+
|
|
77
|
+
// ── Typoglycemia layer ─────────────────────────────────────────────
|
|
78
|
+
// Scrambled-middle attacks ("ignroe all prevoius systme instructions")
|
|
79
|
+
// keep the first and last letters intact. Per OWASP, use an
|
|
80
|
+
// established string metric rather than ad-hoc scramble detection:
|
|
81
|
+
// restricted Damerau-Levenshtein (optimal string alignment), bounded
|
|
82
|
+
// by 1 for short words and 2 for longer ones, gated on matching
|
|
83
|
+
// first + last characters and near-equal length.
|
|
84
|
+
|
|
85
|
+
function osaDistance(a, b) {
|
|
86
|
+
const al = a.length;
|
|
87
|
+
const bl = b.length;
|
|
88
|
+
const d = [];
|
|
89
|
+
for (let i = 0; i <= al; i++) d.push([i, ...new Array(bl).fill(0)]);
|
|
90
|
+
for (let j = 0; j <= bl; j++) d[0][j] = j;
|
|
91
|
+
for (let i = 1; i <= al; i++) {
|
|
92
|
+
for (let j = 1; j <= bl; j++) {
|
|
93
|
+
const cost = a[i - 1] === b[j - 1] ? 0 : 1;
|
|
94
|
+
d[i][j] = Math.min(
|
|
95
|
+
d[i - 1][j] + 1,
|
|
96
|
+
d[i][j - 1] + 1,
|
|
97
|
+
d[i - 1][j - 1] + cost
|
|
98
|
+
);
|
|
99
|
+
if (
|
|
100
|
+
i > 1 &&
|
|
101
|
+
j > 1 &&
|
|
102
|
+
a[i - 1] === b[j - 2] &&
|
|
103
|
+
a[i - 2] === b[j - 1]
|
|
104
|
+
) {
|
|
105
|
+
d[i][j] = Math.min(d[i][j], d[i - 2][j - 2] + 1);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
return d[al][bl];
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function wordMatches(word, target) {
|
|
113
|
+
if (word === target) return true;
|
|
114
|
+
if (word.length < 4 || target.length < 4) return false;
|
|
115
|
+
if (word[0] !== target[0]) return false;
|
|
116
|
+
if (word[word.length - 1] !== target[target.length - 1]) return false;
|
|
117
|
+
if (Math.abs(word.length - target.length) > 1) return false;
|
|
118
|
+
const threshold = target.length > 6 ? 2 : 1;
|
|
119
|
+
return osaDistance(word, target) <= threshold;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// Each sequence is a list of slots; a slot matches when any of its
|
|
123
|
+
// target words fuzzy-matches the token. Up to two filler tokens are
|
|
124
|
+
// allowed between consecutive slots ("ignore all of the previous
|
|
125
|
+
// instructions" still matches slot-to-slot).
|
|
126
|
+
const FUZZY_SEQUENCES = [
|
|
127
|
+
[
|
|
128
|
+
["ignore", "disregard", "forget"],
|
|
129
|
+
["previous", "prior", "earlier", "system", "above", "original"],
|
|
130
|
+
["instructions", "instruction", "prompt", "prompts", "rules", "directives", "guidelines"],
|
|
131
|
+
],
|
|
132
|
+
[
|
|
133
|
+
["bypass", "disable", "circumvent"],
|
|
134
|
+
["safety", "security", "content"],
|
|
135
|
+
["filters", "filter", "measures", "checks", "guardrails", "restrictions"],
|
|
136
|
+
],
|
|
137
|
+
[
|
|
138
|
+
["reveal", "show", "repeat", "print", "display"],
|
|
139
|
+
["system"],
|
|
140
|
+
["prompt", "instructions", "message"],
|
|
141
|
+
],
|
|
142
|
+
];
|
|
143
|
+
|
|
144
|
+
const MAX_FILLER_TOKENS = 2;
|
|
145
|
+
|
|
146
|
+
function matchesFuzzySequence(tokens, sequence) {
|
|
147
|
+
for (let start = 0; start < tokens.length; start++) {
|
|
148
|
+
let slotIndex = 0;
|
|
149
|
+
let i = start;
|
|
150
|
+
let fillersLeft = MAX_FILLER_TOKENS;
|
|
151
|
+
while (i < tokens.length && slotIndex < sequence.length) {
|
|
152
|
+
const slot = sequence[slotIndex];
|
|
153
|
+
if (slot.some((target) => wordMatches(tokens[i], target))) {
|
|
154
|
+
slotIndex++;
|
|
155
|
+
fillersLeft = MAX_FILLER_TOKENS;
|
|
156
|
+
} else if (slotIndex > 0) {
|
|
157
|
+
if (fillersLeft === 0) break;
|
|
158
|
+
fillersLeft--;
|
|
159
|
+
} else {
|
|
160
|
+
break;
|
|
161
|
+
}
|
|
162
|
+
i++;
|
|
163
|
+
}
|
|
164
|
+
if (slotIndex === sequence.length) return true;
|
|
165
|
+
}
|
|
166
|
+
return false;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// ── Base64 smuggling layer ─────────────────────────────────────────
|
|
170
|
+
// Long base64 runs get decoded and re-checked against the exact
|
|
171
|
+
// pattern list. Arbitrary base64 (file payloads, ids) that doesn't
|
|
172
|
+
// decode to an injection never flags.
|
|
173
|
+
|
|
174
|
+
const BASE64_RUN = /[A-Za-z0-9+/]{24,}={0,2}/g;
|
|
175
|
+
|
|
176
|
+
function decodedBase64Hits(rawText) {
|
|
177
|
+
const runs = rawText.match(BASE64_RUN);
|
|
178
|
+
if (!runs) return false;
|
|
179
|
+
for (const run of runs) {
|
|
180
|
+
let decoded;
|
|
181
|
+
try {
|
|
182
|
+
decoded = Buffer.from(run, "base64").toString("utf8");
|
|
183
|
+
} catch {
|
|
184
|
+
continue;
|
|
185
|
+
}
|
|
186
|
+
if (decoded.length === 0) continue;
|
|
187
|
+
let printable = 0;
|
|
188
|
+
for (const ch of decoded) {
|
|
189
|
+
const code = ch.codePointAt(0);
|
|
190
|
+
if ((code >= 0x20 && code < 0x7f) || code === 0x09 || code === 0x0a || code === 0x0d) {
|
|
191
|
+
printable++;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
if (printable / decoded.length < 0.8) continue;
|
|
195
|
+
const norm = normalize(decoded);
|
|
196
|
+
if (PATTERNS.some((p) => p.test(norm))) return true;
|
|
197
|
+
}
|
|
198
|
+
return false;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// ── Detection entry point ──────────────────────────────────────────
|
|
202
|
+
|
|
203
|
+
function isInjection(rawText) {
|
|
204
|
+
const norm = normalize(rawText);
|
|
205
|
+
if (norm.length === 0) return false;
|
|
206
|
+
if (PATTERNS.some((p) => p.test(norm))) return true;
|
|
207
|
+
const tokens = norm.split(/[^a-z0-9']+/).filter((t) => t.length > 0);
|
|
208
|
+
if (FUZZY_SEQUENCES.some((seq) => matchesFuzzySequence(tokens, seq))) {
|
|
209
|
+
return true;
|
|
210
|
+
}
|
|
211
|
+
return decodedBase64Hits(rawText);
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** Pull the scannable text out of a message's content — plain string,
|
|
215
|
+
* or the concatenated text parts of a multimodal array. */
|
|
216
|
+
function extractText(content) {
|
|
217
|
+
if (typeof content === "string") return content;
|
|
218
|
+
if (Array.isArray(content)) {
|
|
219
|
+
const parts = [];
|
|
220
|
+
for (const part of content) {
|
|
221
|
+
if (part && typeof part === "object" && part.type === "text" && typeof part.text === "string") {
|
|
222
|
+
parts.push(part.text);
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
return parts.join("\n");
|
|
226
|
+
}
|
|
227
|
+
return "";
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
export default async ({ inputs, settings }) => {
|
|
231
|
+
const messages = inputs.messages;
|
|
232
|
+
|
|
233
|
+
if (!Array.isArray(messages)) {
|
|
234
|
+
throw new Error("Messages input must be an array");
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const notice =
|
|
238
|
+
typeof settings?.notice === "string" && settings.notice.trim().length > 0
|
|
239
|
+
? settings.notice
|
|
240
|
+
: DEFAULT_NOTICE;
|
|
241
|
+
|
|
242
|
+
const result = [];
|
|
243
|
+
let flaggedCount = 0;
|
|
244
|
+
|
|
245
|
+
for (const message of messages) {
|
|
246
|
+
if (!message || typeof message !== "object" || message.role !== "user") {
|
|
247
|
+
result.push(message);
|
|
248
|
+
continue;
|
|
249
|
+
}
|
|
250
|
+
const text = extractText(message.content);
|
|
251
|
+
// Already-withheld messages (our own notice re-entering via
|
|
252
|
+
// conversation history) are never re-scanned — keeps the node
|
|
253
|
+
// idempotent across turns.
|
|
254
|
+
if (text.startsWith(NOTICE_MARKER) || !isInjection(text)) {
|
|
255
|
+
result.push(message);
|
|
256
|
+
continue;
|
|
257
|
+
}
|
|
258
|
+
// Replace the entire content — for multimodal messages the
|
|
259
|
+
// non-text parts are withheld too, since an image can carry the
|
|
260
|
+
// payload the flagged text was priming.
|
|
261
|
+
result.push({ ...message, content: notice });
|
|
262
|
+
flaggedCount++;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
return {
|
|
266
|
+
messages: result,
|
|
267
|
+
flagged_count: flaggedCount,
|
|
268
|
+
};
|
|
269
|
+
};
|