@framers/agentos-ext-content-policy-rewriter 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +91 -0
- package/dist/ContentPolicyRewriterGuardrail.d.ts +30 -0
- package/dist/ContentPolicyRewriterGuardrail.js +104 -0
- package/dist/KeywordPreFilter.d.ts +19 -0
- package/dist/KeywordPreFilter.js +35 -0
- package/dist/LlmPolicyJudge.d.ts +11 -0
- package/dist/LlmPolicyJudge.js +36 -0
- package/dist/LlmRewriter.d.ts +11 -0
- package/dist/LlmRewriter.js +30 -0
- package/dist/categories.d.ts +22 -0
- package/dist/categories.js +67 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.js +77 -0
- package/dist/presets.d.ts +7 -0
- package/dist/presets.js +47 -0
- package/dist/types.d.ts +41 -0
- package/dist/types.js +8 -0
- package/package.json +29 -0
- package/src/ContentPolicyRewriterGuardrail.ts +126 -0
- package/src/KeywordPreFilter.ts +49 -0
- package/src/LlmPolicyJudge.ts +41 -0
- package/src/LlmRewriter.ts +32 -0
- package/src/categories.ts +77 -0
- package/src/index.ts +92 -0
- package/src/presets.ts +53 -0
- package/src/types.ts +60 -0
- package/test/ContentPolicyRewriter.spec.ts +79 -0
- package/test/KeywordPreFilter.spec.ts +46 -0
- package/test/LlmPolicyJudge.spec.ts +51 -0
- package/test/LlmRewriter.spec.ts +34 -0
- package/test/presets.spec.ts +30 -0
- package/tsconfig.json +14 -0
- package/vitest.config.ts +8 -0
package/README.md
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# @framers/agentos-ext-content-policy-rewriter
|
|
2
|
+
|
|
3
|
+
Opt-in content policy guardrail for AgentOS — detects violations in agent output and either blocks or rewrites them to compliant versions via LLM judge.
|
|
4
|
+
|
|
5
|
+
**Agents are uncensored by default.** This extension only activates when explicitly configured.
|
|
6
|
+
|
|
7
|
+
## Features
|
|
8
|
+
|
|
9
|
+
- **Two-layer hybrid pipeline:** keyword pre-filter on streaming chunks (zero-cost) + LLM judge/rewriter on final response
|
|
10
|
+
- **8 configurable categories:** illegal_harmful, adult, profanity, violence, self_harm, hate_speech, illegal_activity, custom
|
|
11
|
+
- **4 presets:** uncensored, uncensored-safe, family-friendly, enterprise
|
|
12
|
+
- **Fully configurable:** every category can be enabled/disabled, action set to block or sanitize
|
|
13
|
+
- **No hardcoded restrictions:** all policies are user-controlled
|
|
14
|
+
|
|
15
|
+
## Quick Start
|
|
16
|
+
|
|
17
|
+
```typescript
|
|
18
|
+
import { createContentPolicyRewriter } from '@framers/agentos-ext-content-policy-rewriter';
|
|
19
|
+
|
|
20
|
+
// Minimal — blocks illegal_harmful content only (default)
|
|
21
|
+
const pack = createContentPolicyRewriter({});
|
|
22
|
+
|
|
23
|
+
// Family-friendly preset
|
|
24
|
+
const pack = createContentPolicyRewriter('family-friendly');
|
|
25
|
+
|
|
26
|
+
// Custom configuration
|
|
27
|
+
const pack = createContentPolicyRewriter({
|
|
28
|
+
categories: {
|
|
29
|
+
adult: { enabled: true, action: 'sanitize' },
|
|
30
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
31
|
+
violence: { enabled: true, action: 'block' },
|
|
32
|
+
},
|
|
33
|
+
customRules: 'Never mention competitor products by name.',
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
// Truly uncensored — zero filtering
|
|
37
|
+
const pack = createContentPolicyRewriter('uncensored');
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## agent.config.json
|
|
41
|
+
|
|
42
|
+
```json
|
|
43
|
+
{
|
|
44
|
+
"guardrails": {
|
|
45
|
+
"contentPolicy": {
|
|
46
|
+
"enabled": true,
|
|
47
|
+
"categories": {
|
|
48
|
+
"illegal_harmful": { "enabled": true, "action": "block" },
|
|
49
|
+
"adult": { "enabled": true, "action": "sanitize" },
|
|
50
|
+
"profanity": { "enabled": true, "action": "sanitize" }
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Or shorthand:
|
|
58
|
+
|
|
59
|
+
```json
|
|
60
|
+
{
|
|
61
|
+
"guardrails": {
|
|
62
|
+
"contentPolicy": "uncensored-safe"
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Categories
|
|
68
|
+
|
|
69
|
+
| Category | Description | Default |
|
|
70
|
+
|---|---|---|
|
|
71
|
+
| `illegal_harmful` | CSAM, sexual assault, bestiality, exploitation | enabled, block |
|
|
72
|
+
| `adult` | Consensual sexually explicit content | disabled |
|
|
73
|
+
| `profanity` | Slurs, vulgar language | disabled |
|
|
74
|
+
| `violence` | Graphic violence, gore | disabled |
|
|
75
|
+
| `self_harm` | Self-harm, suicide instructions | disabled |
|
|
76
|
+
| `hate_speech` | Discriminatory, bigoted content | disabled |
|
|
77
|
+
| `illegal_activity` | Drug synthesis, weapons manufacturing | disabled |
|
|
78
|
+
| `custom` | User-defined policy rules | disabled |
|
|
79
|
+
|
|
80
|
+
## Presets
|
|
81
|
+
|
|
82
|
+
| Preset | Effect |
|
|
83
|
+
|---|---|
|
|
84
|
+
| `uncensored` | All categories disabled — zero filtering |
|
|
85
|
+
| `uncensored-safe` | Only `illegal_harmful` enabled |
|
|
86
|
+
| `family-friendly` | All categories enabled (sanitize where possible) |
|
|
87
|
+
| `enterprise` | All categories enabled + custom rules |
|
|
88
|
+
|
|
89
|
+
## License
|
|
90
|
+
|
|
91
|
+
MIT
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Main guardrail orchestrator for content policy rewriting.
|
|
3
|
+
*
|
|
4
|
+
* Implements IGuardrailService with a two-layer hybrid pipeline:
|
|
5
|
+
* - Layer 1: KeywordPreFilter on streaming TEXT_DELTA chunks (zero-cost)
|
|
6
|
+
* - Layer 2: LLM Policy Judge + LLM Rewriter on FINAL_RESPONSE
|
|
7
|
+
*
|
|
8
|
+
* Phase 1 sanitizer — sequential execution, can return SANITIZE with modifiedText.
|
|
9
|
+
*
|
|
10
|
+
* @module content-policy-rewriter/ContentPolicyRewriterGuardrail
|
|
11
|
+
*/
|
|
12
|
+
import type { ContentPolicyRewriterConfig, LlmInvoker } from './types.js';
|
|
13
|
+
export interface ContentPolicyRewriterOptions extends ContentPolicyRewriterConfig {
|
|
14
|
+
llmInvoker: LlmInvoker;
|
|
15
|
+
}
|
|
16
|
+
export declare class ContentPolicyRewriterGuardrail {
|
|
17
|
+
readonly config: {
|
|
18
|
+
canSanitize: boolean;
|
|
19
|
+
evaluateStreamingChunks: boolean;
|
|
20
|
+
streamingMode: "per-chunk";
|
|
21
|
+
};
|
|
22
|
+
private enabledCategories;
|
|
23
|
+
private keywordFilter;
|
|
24
|
+
private judge;
|
|
25
|
+
private rewriter;
|
|
26
|
+
private customRules?;
|
|
27
|
+
constructor(options: ContentPolicyRewriterOptions);
|
|
28
|
+
evaluateInput(_payload: any): Promise<any>;
|
|
29
|
+
evaluateOutput(payload: any): Promise<any>;
|
|
30
|
+
}
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Main guardrail orchestrator for content policy rewriting.
|
|
3
|
+
*
|
|
4
|
+
* Implements IGuardrailService with a two-layer hybrid pipeline:
|
|
5
|
+
* - Layer 1: KeywordPreFilter on streaming TEXT_DELTA chunks (zero-cost)
|
|
6
|
+
* - Layer 2: LLM Policy Judge + LLM Rewriter on FINAL_RESPONSE
|
|
7
|
+
*
|
|
8
|
+
* Phase 1 sanitizer — sequential execution, can return SANITIZE with modifiedText.
|
|
9
|
+
*
|
|
10
|
+
* @module content-policy-rewriter/ContentPolicyRewriterGuardrail
|
|
11
|
+
*/
|
|
12
|
+
import { resolveCategoryConfig } from './categories.js';
|
|
13
|
+
import { ALL_POLICY_CATEGORIES } from './types.js';
|
|
14
|
+
import { KeywordPreFilter } from './KeywordPreFilter.js';
|
|
15
|
+
import { LlmPolicyJudge } from './LlmPolicyJudge.js';
|
|
16
|
+
import { LlmRewriter } from './LlmRewriter.js';
|
|
17
|
+
/** GuardrailAction values (avoid import issues by inlining). */
|
|
18
|
+
const BLOCK = 'block';
|
|
19
|
+
const SANITIZE = 'sanitize';
|
|
20
|
+
export class ContentPolicyRewriterGuardrail {
|
|
21
|
+
config = {
|
|
22
|
+
canSanitize: true,
|
|
23
|
+
evaluateStreamingChunks: true,
|
|
24
|
+
streamingMode: 'per-chunk',
|
|
25
|
+
};
|
|
26
|
+
enabledCategories;
|
|
27
|
+
keywordFilter;
|
|
28
|
+
judge;
|
|
29
|
+
rewriter;
|
|
30
|
+
customRules;
|
|
31
|
+
constructor(options) {
|
|
32
|
+
this.customRules = options.customRules;
|
|
33
|
+
// Resolve enabled categories
|
|
34
|
+
this.enabledCategories = {};
|
|
35
|
+
for (const cat of ALL_POLICY_CATEGORIES) {
|
|
36
|
+
const resolved = resolveCategoryConfig(cat, options.categories);
|
|
37
|
+
if (resolved.enabled) {
|
|
38
|
+
this.enabledCategories[cat] = { enabled: true, action: resolved.action };
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
this.keywordFilter = new KeywordPreFilter(options.keywordLists);
|
|
42
|
+
this.judge = new LlmPolicyJudge(options.llmInvoker);
|
|
43
|
+
this.rewriter = new LlmRewriter(options.llmInvoker);
|
|
44
|
+
if (options.streamingPreFilter === false) {
|
|
45
|
+
this.config.evaluateStreamingChunks = false;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
async evaluateInput(_payload) {
|
|
49
|
+
return null; // Content policy rewriter only evaluates output
|
|
50
|
+
}
|
|
51
|
+
async evaluateOutput(payload) {
|
|
52
|
+
const chunk = payload?.chunk;
|
|
53
|
+
if (!chunk)
|
|
54
|
+
return null;
|
|
55
|
+
const hasEnabledCategories = Object.keys(this.enabledCategories).length > 0;
|
|
56
|
+
if (!hasEnabledCategories)
|
|
57
|
+
return null;
|
|
58
|
+
// Layer 1: Keyword pre-filter on streaming chunks
|
|
59
|
+
if (chunk.type === 'TEXT_DELTA' && chunk.text) {
|
|
60
|
+
const match = this.keywordFilter.scan(chunk.text, this.enabledCategories);
|
|
61
|
+
if (match) {
|
|
62
|
+
const action = this.enabledCategories[match.category]?.action ?? BLOCK;
|
|
63
|
+
return {
|
|
64
|
+
action,
|
|
65
|
+
reasonCode: `KEYWORD_${match.category.toUpperCase()}`,
|
|
66
|
+
...(action === SANITIZE ? { modifiedText: '[Content removed by policy]' } : {}),
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
return null;
|
|
70
|
+
}
|
|
71
|
+
// Layer 2: LLM judge + rewriter on final response
|
|
72
|
+
if (chunk.type === 'FINAL_RESPONSE' && chunk.finalResponseText) {
|
|
73
|
+
const text = chunk.finalResponseText;
|
|
74
|
+
// Quick keyword check — BLOCK immediately if keyword match + action=block
|
|
75
|
+
const kwMatch = this.keywordFilter.scan(text, this.enabledCategories);
|
|
76
|
+
if (kwMatch && this.enabledCategories[kwMatch.category]?.action === BLOCK) {
|
|
77
|
+
return {
|
|
78
|
+
action: BLOCK,
|
|
79
|
+
reasonCode: `KEYWORD_${kwMatch.category.toUpperCase()}`,
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
// LLM classification
|
|
83
|
+
const judgeResult = await this.judge.classify(text, this.enabledCategories, this.customRules);
|
|
84
|
+
if (judgeResult.violations.length === 0)
|
|
85
|
+
return null;
|
|
86
|
+
// Check if all violations map to BLOCK
|
|
87
|
+
const allBlock = judgeResult.violations.every(v => this.enabledCategories[v.category]?.action === BLOCK);
|
|
88
|
+
if (allBlock) {
|
|
89
|
+
return {
|
|
90
|
+
action: BLOCK,
|
|
91
|
+
reasonCode: judgeResult.violations.map(v => v.category).join(','),
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
// At least one SANITIZE — rewrite
|
|
95
|
+
const rewritten = await this.rewriter.rewrite(text, judgeResult.violations, this.customRules);
|
|
96
|
+
return {
|
|
97
|
+
action: SANITIZE,
|
|
98
|
+
modifiedText: rewritten,
|
|
99
|
+
reasonCode: judgeResult.violations.map(v => v.category).join(','),
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
return null;
|
|
103
|
+
}
|
|
104
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 1: Zero-cost keyword/regex pre-filter for streaming chunks.
|
|
3
|
+
* Scans text against per-category keyword lists. No LLM calls.
|
|
4
|
+
* @module content-policy-rewriter/KeywordPreFilter
|
|
5
|
+
*/
|
|
6
|
+
import type { PolicyCategory, CategoryConfig } from './types.js';
|
|
7
|
+
export interface KeywordMatch {
|
|
8
|
+
category: PolicyCategory;
|
|
9
|
+
keyword: string;
|
|
10
|
+
}
|
|
11
|
+
export declare class KeywordPreFilter {
|
|
12
|
+
private patterns;
|
|
13
|
+
constructor(customLists?: Partial<Record<PolicyCategory, string[]>>);
|
|
14
|
+
/**
|
|
15
|
+
* Scan text against enabled category keyword lists.
|
|
16
|
+
* @returns First match found, or null if clean.
|
|
17
|
+
*/
|
|
18
|
+
scan(text: string, enabledCategories: Partial<Record<PolicyCategory, CategoryConfig>>): KeywordMatch | null;
|
|
19
|
+
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 1: Zero-cost keyword/regex pre-filter for streaming chunks.
|
|
3
|
+
* Scans text against per-category keyword lists. No LLM calls.
|
|
4
|
+
* @module content-policy-rewriter/KeywordPreFilter
|
|
5
|
+
*/
|
|
6
|
+
import { DEFAULT_KEYWORD_LISTS } from './categories.js';
|
|
7
|
+
export class KeywordPreFilter {
|
|
8
|
+
patterns;
|
|
9
|
+
constructor(customLists) {
|
|
10
|
+
this.patterns = new Map();
|
|
11
|
+
const merged = { ...DEFAULT_KEYWORD_LISTS, ...customLists };
|
|
12
|
+
for (const [cat, keywords] of Object.entries(merged)) {
|
|
13
|
+
if (keywords?.length) {
|
|
14
|
+
this.patterns.set(cat, keywords.map(kw => new RegExp(`\\b${kw.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i')));
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Scan text against enabled category keyword lists.
|
|
20
|
+
* @returns First match found, or null if clean.
|
|
21
|
+
*/
|
|
22
|
+
scan(text, enabledCategories) {
|
|
23
|
+
for (const [cat, patterns] of this.patterns) {
|
|
24
|
+
const cfg = enabledCategories[cat];
|
|
25
|
+
if (!cfg?.enabled)
|
|
26
|
+
continue;
|
|
27
|
+
for (const re of patterns) {
|
|
28
|
+
const match = text.match(re);
|
|
29
|
+
if (match)
|
|
30
|
+
return { category: cat, keyword: match[0] };
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
return null;
|
|
34
|
+
}
|
|
35
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 2a: LLM-based policy violation classifier.
|
|
3
|
+
* Sends the full response text to an LLM with a structured classification prompt.
|
|
4
|
+
* @module content-policy-rewriter/LlmPolicyJudge
|
|
5
|
+
*/
|
|
6
|
+
import type { LlmInvoker, PolicyCategory, CategoryConfig, JudgeResult } from './types.js';
|
|
7
|
+
export declare class LlmPolicyJudge {
|
|
8
|
+
private llmInvoker;
|
|
9
|
+
constructor(llmInvoker: LlmInvoker);
|
|
10
|
+
classify(text: string, enabledCategories: Partial<Record<PolicyCategory, CategoryConfig>>, customRules?: string): Promise<JudgeResult>;
|
|
11
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 2a: LLM-based policy violation classifier.
|
|
3
|
+
* Sends the full response text to an LLM with a structured classification prompt.
|
|
4
|
+
* @module content-policy-rewriter/LlmPolicyJudge
|
|
5
|
+
*/
|
|
6
|
+
import { CATEGORY_DESCRIPTIONS } from './categories.js';
|
|
7
|
+
const SYSTEM_PROMPT = `You are a content policy classifier. Analyze the provided text and identify any violations of the enabled policy categories. Respond with JSON only. If no violations, respond: { "violations": [] }`;
|
|
8
|
+
export class LlmPolicyJudge {
|
|
9
|
+
llmInvoker;
|
|
10
|
+
constructor(llmInvoker) {
|
|
11
|
+
this.llmInvoker = llmInvoker;
|
|
12
|
+
}
|
|
13
|
+
async classify(text, enabledCategories, customRules) {
|
|
14
|
+
const categoryList = Object.entries(enabledCategories)
|
|
15
|
+
.filter(([, cfg]) => cfg?.enabled)
|
|
16
|
+
.map(([cat]) => `- ${cat}: ${CATEGORY_DESCRIPTIONS[cat]}`)
|
|
17
|
+
.join('\n');
|
|
18
|
+
if (!categoryList)
|
|
19
|
+
return { violations: [] };
|
|
20
|
+
let userPrompt = `Enabled categories:\n${categoryList}\n`;
|
|
21
|
+
if (customRules)
|
|
22
|
+
userPrompt += `\nCustom rules: ${customRules}\n`;
|
|
23
|
+
userPrompt += `\nText to analyze:\n"""\n${text}\n"""\n\nRespond with JSON: { "violations": [{ "category": "...", "severity": "low|medium|high", "spans": ["offending phrase"] }] }`;
|
|
24
|
+
try {
|
|
25
|
+
const raw = await this.llmInvoker(SYSTEM_PROMPT, userPrompt);
|
|
26
|
+
const cleaned = raw.replace(/```json\n?|```\n?/g, '').trim();
|
|
27
|
+
const parsed = JSON.parse(cleaned);
|
|
28
|
+
if (!Array.isArray(parsed.violations))
|
|
29
|
+
return { violations: [] };
|
|
30
|
+
return parsed;
|
|
31
|
+
}
|
|
32
|
+
catch {
|
|
33
|
+
return { violations: [] };
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 2b: LLM-based content rewriter.
|
|
3
|
+
* Takes text with identified violations and rewrites it to remove them.
|
|
4
|
+
* @module content-policy-rewriter/LlmRewriter
|
|
5
|
+
*/
|
|
6
|
+
import type { LlmInvoker, PolicyViolation } from './types.js';
|
|
7
|
+
export declare class LlmRewriter {
|
|
8
|
+
private llmInvoker;
|
|
9
|
+
constructor(llmInvoker: LlmInvoker);
|
|
10
|
+
rewrite(text: string, violations: PolicyViolation[], customRules?: string): Promise<string>;
|
|
11
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 2b: LLM-based content rewriter.
|
|
3
|
+
* Takes text with identified violations and rewrites it to remove them.
|
|
4
|
+
* @module content-policy-rewriter/LlmRewriter
|
|
5
|
+
*/
|
|
6
|
+
const SYSTEM_PROMPT = `Rewrite the following text to remove policy violations while preserving the original meaning, tone, and information content. Do not add disclaimers or commentary — just output the clean version.`;
|
|
7
|
+
export class LlmRewriter {
|
|
8
|
+
llmInvoker;
|
|
9
|
+
constructor(llmInvoker) {
|
|
10
|
+
this.llmInvoker = llmInvoker;
|
|
11
|
+
}
|
|
12
|
+
async rewrite(text, violations, customRules) {
|
|
13
|
+
if (violations.length === 0)
|
|
14
|
+
return text;
|
|
15
|
+
const violationList = violations
|
|
16
|
+
.map(v => `- ${v.category}: ${v.spans.map(s => `"${s}"`).join(', ')}`)
|
|
17
|
+
.join('\n');
|
|
18
|
+
let userPrompt = `Violations to remove:\n${violationList}\n`;
|
|
19
|
+
if (customRules)
|
|
20
|
+
userPrompt += `\nAdditional rules: ${customRules}\n`;
|
|
21
|
+
userPrompt += `\nOriginal text:\n"""\n${text}\n"""\n\nRewritten text:`;
|
|
22
|
+
try {
|
|
23
|
+
const raw = await this.llmInvoker(SYSTEM_PROMPT, userPrompt);
|
|
24
|
+
return raw.replace(/^```\w*\s*\n?/gm, '').replace(/\n?```\s*$/gm, '').trim() || text;
|
|
25
|
+
}
|
|
26
|
+
catch {
|
|
27
|
+
return text;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Built-in keyword lists per policy category.
|
|
3
|
+
* @module content-policy-rewriter/categories
|
|
4
|
+
*/
|
|
5
|
+
import type { PolicyCategory, CategoryConfig } from './types.js';
|
|
6
|
+
/** Default category descriptions used in LLM judge prompts. */
|
|
7
|
+
export declare const CATEGORY_DESCRIPTIONS: Record<PolicyCategory, string>;
|
|
8
|
+
/** Default actions per category. */
|
|
9
|
+
export declare const DEFAULT_CATEGORY_ACTIONS: Record<PolicyCategory, 'block' | 'sanitize'>;
|
|
10
|
+
/** Default enabled state — only illegal_harmful is on by default. */
|
|
11
|
+
export declare const DEFAULT_CATEGORY_ENABLED: Record<PolicyCategory, boolean>;
|
|
12
|
+
/**
|
|
13
|
+
* Built-in keyword lists for the streaming pre-filter.
|
|
14
|
+
* Intentionally broad — false positives acceptable because the LLM judge
|
|
15
|
+
* runs on the final response for accuracy.
|
|
16
|
+
*/
|
|
17
|
+
export declare const DEFAULT_KEYWORD_LISTS: Partial<Record<PolicyCategory, string[]>>;
|
|
18
|
+
/** Resolve effective config for a category. */
|
|
19
|
+
export declare function resolveCategoryConfig(category: PolicyCategory, userConfig?: Partial<Record<PolicyCategory, CategoryConfig>>): {
|
|
20
|
+
enabled: boolean;
|
|
21
|
+
action: 'block' | 'sanitize';
|
|
22
|
+
};
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Built-in keyword lists per policy category.
|
|
3
|
+
* @module content-policy-rewriter/categories
|
|
4
|
+
*/
|
|
5
|
+
/** Default category descriptions used in LLM judge prompts. */
|
|
6
|
+
export const CATEGORY_DESCRIPTIONS = {
|
|
7
|
+
illegal_harmful: 'Child sexual abuse material (CSAM), sexual assault, bestiality, non-consensual acts, exploitation of minors, graphic torture',
|
|
8
|
+
adult: 'Consensual sexually explicit content, pornographic descriptions',
|
|
9
|
+
profanity: 'Slurs, vulgar language, obscenity, crude insults',
|
|
10
|
+
violence: 'Graphic violence, gore, detailed injury descriptions',
|
|
11
|
+
self_harm: 'Self-harm instructions, suicide methods, pro-anorexia content',
|
|
12
|
+
hate_speech: 'Discriminatory language, bigotry, slurs targeting protected groups',
|
|
13
|
+
illegal_activity: 'Drug synthesis instructions, weapons manufacturing, hacking tutorials',
|
|
14
|
+
custom: 'User-defined policy rules',
|
|
15
|
+
};
|
|
16
|
+
/** Default actions per category. */
|
|
17
|
+
export const DEFAULT_CATEGORY_ACTIONS = {
|
|
18
|
+
illegal_harmful: 'block',
|
|
19
|
+
adult: 'sanitize',
|
|
20
|
+
profanity: 'sanitize',
|
|
21
|
+
violence: 'sanitize',
|
|
22
|
+
self_harm: 'block',
|
|
23
|
+
hate_speech: 'sanitize',
|
|
24
|
+
illegal_activity: 'block',
|
|
25
|
+
custom: 'sanitize',
|
|
26
|
+
};
|
|
27
|
+
/** Default enabled state — only illegal_harmful is on by default. */
|
|
28
|
+
export const DEFAULT_CATEGORY_ENABLED = {
|
|
29
|
+
illegal_harmful: true,
|
|
30
|
+
adult: false,
|
|
31
|
+
profanity: false,
|
|
32
|
+
violence: false,
|
|
33
|
+
self_harm: false,
|
|
34
|
+
hate_speech: false,
|
|
35
|
+
illegal_activity: false,
|
|
36
|
+
custom: false,
|
|
37
|
+
};
|
|
38
|
+
/**
|
|
39
|
+
* Built-in keyword lists for the streaming pre-filter.
|
|
40
|
+
* Intentionally broad — false positives acceptable because the LLM judge
|
|
41
|
+
* runs on the final response for accuracy.
|
|
42
|
+
*/
|
|
43
|
+
export const DEFAULT_KEYWORD_LISTS = {
|
|
44
|
+
profanity: [
|
|
45
|
+
'fuck', 'shit', 'bitch', 'asshole', 'cunt', 'dick', 'bastard',
|
|
46
|
+
'damn', 'nigger', 'faggot', 'retard', 'slut', 'whore',
|
|
47
|
+
],
|
|
48
|
+
illegal_harmful: [
|
|
49
|
+
'child porn', 'cp links', 'lolicon', 'underage sex', 'rape fantasy',
|
|
50
|
+
'snuff film', 'crush video',
|
|
51
|
+
],
|
|
52
|
+
self_harm: [
|
|
53
|
+
'how to kill yourself', 'suicide method', 'cut yourself', 'end it all',
|
|
54
|
+
],
|
|
55
|
+
hate_speech: [
|
|
56
|
+
'kill all', 'gas the', 'ethnic cleansing', 'race war', 'white power',
|
|
57
|
+
'heil hitler', 'death to',
|
|
58
|
+
],
|
|
59
|
+
};
|
|
60
|
+
/** Resolve effective config for a category. */
|
|
61
|
+
export function resolveCategoryConfig(category, userConfig) {
|
|
62
|
+
const cfg = userConfig?.[category];
|
|
63
|
+
return {
|
|
64
|
+
enabled: cfg?.enabled ?? DEFAULT_CATEGORY_ENABLED[category],
|
|
65
|
+
action: cfg?.action ?? DEFAULT_CATEGORY_ACTIONS[category],
|
|
66
|
+
};
|
|
67
|
+
}
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Content Policy Rewriter — extension pack factory.
|
|
3
|
+
*
|
|
4
|
+
* Opt-in guardrail that detects content policy violations in agent output
|
|
5
|
+
* and either blocks or rewrites them via LLM. Agents are uncensored by
|
|
6
|
+
* default — this extension only activates when explicitly configured.
|
|
7
|
+
*
|
|
8
|
+
* @module content-policy-rewriter
|
|
9
|
+
*/
|
|
10
|
+
import type { ContentPolicyRewriterConfig, LlmInvoker } from './types.js';
|
|
11
|
+
import { ContentPolicyRewriterGuardrail } from './ContentPolicyRewriterGuardrail.js';
|
|
12
|
+
export * from './types.js';
|
|
13
|
+
export { ContentPolicyRewriterGuardrail } from './ContentPolicyRewriterGuardrail.js';
|
|
14
|
+
export { KeywordPreFilter } from './KeywordPreFilter.js';
|
|
15
|
+
export { LlmPolicyJudge } from './LlmPolicyJudge.js';
|
|
16
|
+
export { LlmRewriter } from './LlmRewriter.js';
|
|
17
|
+
export { resolvePreset } from './presets.js';
|
|
18
|
+
/**
|
|
19
|
+
* Create the content policy rewriter extension pack.
|
|
20
|
+
*
|
|
21
|
+
* @param config - Policy configuration or preset name.
|
|
22
|
+
* @param llmInvoker - Callback to invoke an LLM for classification and rewriting.
|
|
23
|
+
* @returns A configured ExtensionPack.
|
|
24
|
+
*/
|
|
25
|
+
export declare function createContentPolicyRewriter(config?: ContentPolicyRewriterConfig | string, llmInvoker?: LlmInvoker): {
|
|
26
|
+
name: string;
|
|
27
|
+
version: string;
|
|
28
|
+
descriptors: {
|
|
29
|
+
id: string;
|
|
30
|
+
kind: "guardrail";
|
|
31
|
+
priority: number;
|
|
32
|
+
payload: ContentPolicyRewriterGuardrail;
|
|
33
|
+
}[];
|
|
34
|
+
};
|
|
35
|
+
/** Manifest factory bridge for extension loader. */
|
|
36
|
+
export declare function createExtensionPack(context: any): {
|
|
37
|
+
name: string;
|
|
38
|
+
version: string;
|
|
39
|
+
descriptors: {
|
|
40
|
+
id: string;
|
|
41
|
+
kind: "guardrail";
|
|
42
|
+
priority: number;
|
|
43
|
+
payload: ContentPolicyRewriterGuardrail;
|
|
44
|
+
}[];
|
|
45
|
+
};
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Content Policy Rewriter — extension pack factory.
|
|
3
|
+
*
|
|
4
|
+
* Opt-in guardrail that detects content policy violations in agent output
|
|
5
|
+
* and either blocks or rewrites them via LLM. Agents are uncensored by
|
|
6
|
+
* default — this extension only activates when explicitly configured.
|
|
7
|
+
*
|
|
8
|
+
* @module content-policy-rewriter
|
|
9
|
+
*/
|
|
10
|
+
import { ContentPolicyRewriterGuardrail } from './ContentPolicyRewriterGuardrail.js';
|
|
11
|
+
import { resolvePreset } from './presets.js';
|
|
12
|
+
export * from './types.js';
|
|
13
|
+
export { ContentPolicyRewriterGuardrail } from './ContentPolicyRewriterGuardrail.js';
|
|
14
|
+
export { KeywordPreFilter } from './KeywordPreFilter.js';
|
|
15
|
+
export { LlmPolicyJudge } from './LlmPolicyJudge.js';
|
|
16
|
+
export { LlmRewriter } from './LlmRewriter.js';
|
|
17
|
+
export { resolvePreset } from './presets.js';
|
|
18
|
+
/**
|
|
19
|
+
* Create the content policy rewriter extension pack.
|
|
20
|
+
*
|
|
21
|
+
* @param config - Policy configuration or preset name.
|
|
22
|
+
* @param llmInvoker - Callback to invoke an LLM for classification and rewriting.
|
|
23
|
+
* @returns A configured ExtensionPack.
|
|
24
|
+
*/
|
|
25
|
+
export function createContentPolicyRewriter(config = {}, llmInvoker) {
|
|
26
|
+
const resolved = typeof config === 'string' ? resolvePreset(config) : config;
|
|
27
|
+
const invoker = llmInvoker ?? resolved.llmInvoker ?? createDefaultInvoker(resolved);
|
|
28
|
+
const guardrail = new ContentPolicyRewriterGuardrail({ ...resolved, llmInvoker: invoker });
|
|
29
|
+
return {
|
|
30
|
+
name: 'content-policy-rewriter',
|
|
31
|
+
version: '0.1.0',
|
|
32
|
+
descriptors: [
|
|
33
|
+
{
|
|
34
|
+
id: 'content-policy-rewriter-guardrail',
|
|
35
|
+
kind: 'guardrail',
|
|
36
|
+
priority: 3,
|
|
37
|
+
payload: guardrail,
|
|
38
|
+
},
|
|
39
|
+
],
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
/** Manifest factory bridge for extension loader. */
|
|
43
|
+
export function createExtensionPack(context) {
|
|
44
|
+
return createContentPolicyRewriter(context.options, context.llmInvoker);
|
|
45
|
+
}
|
|
46
|
+
/** Default LLM invoker using fetch to OpenAI-compatible API. */
|
|
47
|
+
function createDefaultInvoker(config) {
|
|
48
|
+
return async (systemPrompt, userPrompt) => {
|
|
49
|
+
const apiKey = process.env.OPENAI_API_KEY || process.env.OPENROUTER_API_KEY;
|
|
50
|
+
if (!apiKey)
|
|
51
|
+
throw new Error('No LLM API key for content policy judge');
|
|
52
|
+
const isOpenRouter = !process.env.OPENAI_API_KEY && !!process.env.OPENROUTER_API_KEY;
|
|
53
|
+
const baseUrl = isOpenRouter ? 'https://openrouter.ai/api/v1' : 'https://api.openai.com/v1';
|
|
54
|
+
const model = config.llm?.model ?? (isOpenRouter ? 'anthropic/claude-haiku-4-5-20251001' : 'gpt-4o-mini');
|
|
55
|
+
const res = await fetch(`${baseUrl}/chat/completions`, {
|
|
56
|
+
method: 'POST',
|
|
57
|
+
headers: {
|
|
58
|
+
'Content-Type': 'application/json',
|
|
59
|
+
Authorization: `Bearer ${apiKey}`,
|
|
60
|
+
...(isOpenRouter && { 'HTTP-Referer': 'https://agentos.sh' }),
|
|
61
|
+
},
|
|
62
|
+
body: JSON.stringify({
|
|
63
|
+
model,
|
|
64
|
+
messages: [
|
|
65
|
+
{ role: 'system', content: systemPrompt },
|
|
66
|
+
{ role: 'user', content: userPrompt },
|
|
67
|
+
],
|
|
68
|
+
max_tokens: 2048,
|
|
69
|
+
temperature: 0.1,
|
|
70
|
+
}),
|
|
71
|
+
});
|
|
72
|
+
if (!res.ok)
|
|
73
|
+
throw new Error(`LLM API returned ${res.status}`);
|
|
74
|
+
const data = (await res.json());
|
|
75
|
+
return data.choices?.[0]?.message?.content ?? '';
|
|
76
|
+
};
|
|
77
|
+
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Preset configurations for common content policy scenarios.
|
|
3
|
+
* @module content-policy-rewriter/presets
|
|
4
|
+
*/
|
|
5
|
+
import type { ContentPolicyRewriterConfig, ContentPolicyPreset } from './types.js';
|
|
6
|
+
/** Resolve a preset string or config object into a full config. */
|
|
7
|
+
export declare function resolvePreset(input: ContentPolicyPreset | ContentPolicyRewriterConfig): ContentPolicyRewriterConfig;
|
package/dist/presets.js
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Preset configurations for common content policy scenarios.
|
|
3
|
+
* @module content-policy-rewriter/presets
|
|
4
|
+
*/
|
|
5
|
+
const ALL_ENABLED_SANITIZE = {
|
|
6
|
+
illegal_harmful: { enabled: true, action: 'block' },
|
|
7
|
+
adult: { enabled: true, action: 'sanitize' },
|
|
8
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
9
|
+
violence: { enabled: true, action: 'sanitize' },
|
|
10
|
+
self_harm: { enabled: true, action: 'block' },
|
|
11
|
+
hate_speech: { enabled: true, action: 'sanitize' },
|
|
12
|
+
illegal_activity: { enabled: true, action: 'block' },
|
|
13
|
+
};
|
|
14
|
+
const PRESET_CONFIGS = {
|
|
15
|
+
uncensored: {
|
|
16
|
+
categories: {
|
|
17
|
+
illegal_harmful: { enabled: false },
|
|
18
|
+
adult: { enabled: false },
|
|
19
|
+
profanity: { enabled: false },
|
|
20
|
+
violence: { enabled: false },
|
|
21
|
+
self_harm: { enabled: false },
|
|
22
|
+
hate_speech: { enabled: false },
|
|
23
|
+
illegal_activity: { enabled: false },
|
|
24
|
+
},
|
|
25
|
+
},
|
|
26
|
+
'uncensored-safe': {
|
|
27
|
+
categories: {
|
|
28
|
+
illegal_harmful: { enabled: true, action: 'block' },
|
|
29
|
+
},
|
|
30
|
+
},
|
|
31
|
+
'family-friendly': {
|
|
32
|
+
categories: ALL_ENABLED_SANITIZE,
|
|
33
|
+
},
|
|
34
|
+
enterprise: {
|
|
35
|
+
categories: ALL_ENABLED_SANITIZE,
|
|
36
|
+
},
|
|
37
|
+
};
|
|
38
|
+
/** Resolve a preset string or config object into a full config. */
|
|
39
|
+
export function resolvePreset(input) {
|
|
40
|
+
if (typeof input === 'string') {
|
|
41
|
+
const preset = PRESET_CONFIGS[input];
|
|
42
|
+
if (!preset)
|
|
43
|
+
return {};
|
|
44
|
+
return { ...preset };
|
|
45
|
+
}
|
|
46
|
+
return input;
|
|
47
|
+
}
|