@framers/agentos-ext-content-policy-rewriter 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +91 -0
- package/dist/ContentPolicyRewriterGuardrail.d.ts +30 -0
- package/dist/ContentPolicyRewriterGuardrail.js +104 -0
- package/dist/KeywordPreFilter.d.ts +19 -0
- package/dist/KeywordPreFilter.js +35 -0
- package/dist/LlmPolicyJudge.d.ts +11 -0
- package/dist/LlmPolicyJudge.js +36 -0
- package/dist/LlmRewriter.d.ts +11 -0
- package/dist/LlmRewriter.js +30 -0
- package/dist/categories.d.ts +22 -0
- package/dist/categories.js +67 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.js +77 -0
- package/dist/presets.d.ts +7 -0
- package/dist/presets.js +47 -0
- package/dist/types.d.ts +41 -0
- package/dist/types.js +8 -0
- package/package.json +29 -0
- package/src/ContentPolicyRewriterGuardrail.ts +126 -0
- package/src/KeywordPreFilter.ts +49 -0
- package/src/LlmPolicyJudge.ts +41 -0
- package/src/LlmRewriter.ts +32 -0
- package/src/categories.ts +77 -0
- package/src/index.ts +92 -0
- package/src/presets.ts +53 -0
- package/src/types.ts +60 -0
- package/test/ContentPolicyRewriter.spec.ts +79 -0
- package/test/KeywordPreFilter.spec.ts +46 -0
- package/test/LlmPolicyJudge.spec.ts +51 -0
- package/test/LlmRewriter.spec.ts +34 -0
- package/test/presets.spec.ts +30 -0
- package/tsconfig.json +14 -0
- package/vitest.config.ts +8 -0
package/dist/types.d.ts
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Types for the content policy rewriter guardrail.
|
|
3
|
+
* @module content-policy-rewriter/types
|
|
4
|
+
*/
|
|
5
|
+
/** Policy categories that can be individually enabled/disabled. */
|
|
6
|
+
export type PolicyCategory = 'illegal_harmful' | 'adult' | 'profanity' | 'violence' | 'self_harm' | 'hate_speech' | 'illegal_activity' | 'custom';
|
|
7
|
+
export declare const ALL_POLICY_CATEGORIES: PolicyCategory[];
|
|
8
|
+
/** Per-category configuration. */
|
|
9
|
+
export interface CategoryConfig {
|
|
10
|
+
enabled?: boolean;
|
|
11
|
+
action?: 'block' | 'sanitize';
|
|
12
|
+
}
|
|
13
|
+
/** A detected policy violation. */
|
|
14
|
+
export interface PolicyViolation {
|
|
15
|
+
category: PolicyCategory;
|
|
16
|
+
severity: 'low' | 'medium' | 'high';
|
|
17
|
+
spans: string[];
|
|
18
|
+
}
|
|
19
|
+
/** Result from the LLM policy judge. */
|
|
20
|
+
export interface JudgeResult {
|
|
21
|
+
violations: PolicyViolation[];
|
|
22
|
+
}
|
|
23
|
+
/** LLM invoker callback — matches the pattern used by ml-classifiers. */
|
|
24
|
+
export type LlmInvoker = (systemPrompt: string, userPrompt: string) => Promise<string>;
|
|
25
|
+
/** Full configuration for the content policy rewriter. */
|
|
26
|
+
export interface ContentPolicyRewriterConfig {
|
|
27
|
+
categories?: Partial<Record<PolicyCategory, CategoryConfig>>;
|
|
28
|
+
customRules?: string;
|
|
29
|
+
llm?: {
|
|
30
|
+
provider?: string;
|
|
31
|
+
model?: string;
|
|
32
|
+
};
|
|
33
|
+
/** Override built-in keyword lists per category. */
|
|
34
|
+
keywordLists?: Partial<Record<PolicyCategory, string[]>>;
|
|
35
|
+
/** Enable keyword pre-filter on streaming chunks. Default: true. */
|
|
36
|
+
streamingPreFilter?: boolean;
|
|
37
|
+
/** LLM invoker callback. If not provided, uses a default fetch-based invoker. */
|
|
38
|
+
llmInvoker?: LlmInvoker;
|
|
39
|
+
}
|
|
40
|
+
/** Preset names for shorthand configuration. */
|
|
41
|
+
export type ContentPolicyPreset = 'uncensored' | 'uncensored-safe' | 'family-friendly' | 'enterprise';
|
package/dist/types.js
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Types for the content policy rewriter guardrail.
|
|
3
|
+
* @module content-policy-rewriter/types
|
|
4
|
+
*/
|
|
5
|
+
export const ALL_POLICY_CATEGORIES = [
|
|
6
|
+
'illegal_harmful', 'adult', 'profanity', 'violence',
|
|
7
|
+
'self_harm', 'hate_speech', 'illegal_activity', 'custom',
|
|
8
|
+
];
|
package/package.json
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@framers/agentos-ext-content-policy-rewriter",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Opt-in content policy guardrail for AgentOS — detects violations and rewrites or blocks output via LLM judge",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "./dist/index.js",
|
|
7
|
+
"types": "./dist/index.d.ts",
|
|
8
|
+
"exports": {
|
|
9
|
+
".": {
|
|
10
|
+
"import": "./dist/index.js",
|
|
11
|
+
"types": "./dist/index.d.ts"
|
|
12
|
+
}
|
|
13
|
+
},
|
|
14
|
+
"scripts": {
|
|
15
|
+
"build": "tsc",
|
|
16
|
+
"clean": "rm -rf dist",
|
|
17
|
+
"typecheck": "tsc --noEmit",
|
|
18
|
+
"test": "vitest run"
|
|
19
|
+
},
|
|
20
|
+
"peerDependencies": {
|
|
21
|
+
"@framers/agentos": ">=0.1.0"
|
|
22
|
+
},
|
|
23
|
+
"devDependencies": {
|
|
24
|
+
"@framers/agentos": "workspace:*",
|
|
25
|
+
"typescript": "^5.7.0",
|
|
26
|
+
"vitest": "^3.0.0"
|
|
27
|
+
},
|
|
28
|
+
"license": "MIT"
|
|
29
|
+
}
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Main guardrail orchestrator for content policy rewriting.
|
|
3
|
+
*
|
|
4
|
+
* Implements IGuardrailService with a two-layer hybrid pipeline:
|
|
5
|
+
* - Layer 1: KeywordPreFilter on streaming TEXT_DELTA chunks (zero-cost)
|
|
6
|
+
* - Layer 2: LLM Policy Judge + LLM Rewriter on FINAL_RESPONSE
|
|
7
|
+
*
|
|
8
|
+
* Phase 1 sanitizer — sequential execution, can return SANITIZE with modifiedText.
|
|
9
|
+
*
|
|
10
|
+
* @module content-policy-rewriter/ContentPolicyRewriterGuardrail
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import type { ContentPolicyRewriterConfig, PolicyCategory, CategoryConfig, LlmInvoker } from './types.js';
|
|
14
|
+
import { resolveCategoryConfig } from './categories.js';
|
|
15
|
+
import { ALL_POLICY_CATEGORIES } from './types.js';
|
|
16
|
+
import { KeywordPreFilter } from './KeywordPreFilter.js';
|
|
17
|
+
import { LlmPolicyJudge } from './LlmPolicyJudge.js';
|
|
18
|
+
import { LlmRewriter } from './LlmRewriter.js';
|
|
19
|
+
|
|
20
|
+
/** GuardrailAction values (avoid import issues by inlining). */
|
|
21
|
+
const BLOCK = 'block';
|
|
22
|
+
const SANITIZE = 'sanitize';
|
|
23
|
+
|
|
24
|
+
export interface ContentPolicyRewriterOptions extends ContentPolicyRewriterConfig {
|
|
25
|
+
llmInvoker: LlmInvoker;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export class ContentPolicyRewriterGuardrail {
|
|
29
|
+
readonly config = {
|
|
30
|
+
canSanitize: true,
|
|
31
|
+
evaluateStreamingChunks: true,
|
|
32
|
+
streamingMode: 'per-chunk' as const,
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
private enabledCategories: Partial<Record<PolicyCategory, CategoryConfig>>;
|
|
36
|
+
private keywordFilter: KeywordPreFilter;
|
|
37
|
+
private judge: LlmPolicyJudge;
|
|
38
|
+
private rewriter: LlmRewriter;
|
|
39
|
+
private customRules?: string;
|
|
40
|
+
|
|
41
|
+
constructor(options: ContentPolicyRewriterOptions) {
|
|
42
|
+
this.customRules = options.customRules;
|
|
43
|
+
|
|
44
|
+
// Resolve enabled categories
|
|
45
|
+
this.enabledCategories = {};
|
|
46
|
+
for (const cat of ALL_POLICY_CATEGORIES) {
|
|
47
|
+
const resolved = resolveCategoryConfig(cat, options.categories);
|
|
48
|
+
if (resolved.enabled) {
|
|
49
|
+
this.enabledCategories[cat] = { enabled: true, action: resolved.action };
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
this.keywordFilter = new KeywordPreFilter(options.keywordLists);
|
|
54
|
+
this.judge = new LlmPolicyJudge(options.llmInvoker);
|
|
55
|
+
this.rewriter = new LlmRewriter(options.llmInvoker);
|
|
56
|
+
|
|
57
|
+
if (options.streamingPreFilter === false) {
|
|
58
|
+
this.config.evaluateStreamingChunks = false;
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
async evaluateInput(_payload: any): Promise<any> {
|
|
63
|
+
return null; // Content policy rewriter only evaluates output
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
async evaluateOutput(payload: any): Promise<any> {
|
|
67
|
+
const chunk = payload?.chunk;
|
|
68
|
+
if (!chunk) return null;
|
|
69
|
+
|
|
70
|
+
const hasEnabledCategories = Object.keys(this.enabledCategories).length > 0;
|
|
71
|
+
if (!hasEnabledCategories) return null;
|
|
72
|
+
|
|
73
|
+
// Layer 1: Keyword pre-filter on streaming chunks
|
|
74
|
+
if (chunk.type === 'TEXT_DELTA' && chunk.text) {
|
|
75
|
+
const match = this.keywordFilter.scan(chunk.text, this.enabledCategories);
|
|
76
|
+
if (match) {
|
|
77
|
+
const action = this.enabledCategories[match.category]?.action ?? BLOCK;
|
|
78
|
+
return {
|
|
79
|
+
action,
|
|
80
|
+
reasonCode: `KEYWORD_${match.category.toUpperCase()}`,
|
|
81
|
+
...(action === SANITIZE ? { modifiedText: '[Content removed by policy]' } : {}),
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
return null;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// Layer 2: LLM judge + rewriter on final response
|
|
88
|
+
if (chunk.type === 'FINAL_RESPONSE' && chunk.finalResponseText) {
|
|
89
|
+
const text = chunk.finalResponseText;
|
|
90
|
+
|
|
91
|
+
// Quick keyword check — BLOCK immediately if keyword match + action=block
|
|
92
|
+
const kwMatch = this.keywordFilter.scan(text, this.enabledCategories);
|
|
93
|
+
if (kwMatch && this.enabledCategories[kwMatch.category]?.action === BLOCK) {
|
|
94
|
+
return {
|
|
95
|
+
action: BLOCK,
|
|
96
|
+
reasonCode: `KEYWORD_${kwMatch.category.toUpperCase()}`,
|
|
97
|
+
};
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// LLM classification
|
|
101
|
+
const judgeResult = await this.judge.classify(text, this.enabledCategories, this.customRules);
|
|
102
|
+
if (judgeResult.violations.length === 0) return null;
|
|
103
|
+
|
|
104
|
+
// Check if all violations map to BLOCK
|
|
105
|
+
const allBlock = judgeResult.violations.every(
|
|
106
|
+
v => this.enabledCategories[v.category]?.action === BLOCK,
|
|
107
|
+
);
|
|
108
|
+
if (allBlock) {
|
|
109
|
+
return {
|
|
110
|
+
action: BLOCK,
|
|
111
|
+
reasonCode: judgeResult.violations.map(v => v.category).join(','),
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// At least one SANITIZE — rewrite
|
|
116
|
+
const rewritten = await this.rewriter.rewrite(text, judgeResult.violations, this.customRules);
|
|
117
|
+
return {
|
|
118
|
+
action: SANITIZE,
|
|
119
|
+
modifiedText: rewritten,
|
|
120
|
+
reasonCode: judgeResult.violations.map(v => v.category).join(','),
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
return null;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 1: Zero-cost keyword/regex pre-filter for streaming chunks.
|
|
3
|
+
* Scans text against per-category keyword lists. No LLM calls.
|
|
4
|
+
* @module content-policy-rewriter/KeywordPreFilter
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import type { PolicyCategory, CategoryConfig } from './types.js';
|
|
8
|
+
import { DEFAULT_KEYWORD_LISTS } from './categories.js';
|
|
9
|
+
|
|
10
|
+
export interface KeywordMatch {
|
|
11
|
+
category: PolicyCategory;
|
|
12
|
+
keyword: string;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export class KeywordPreFilter {
|
|
16
|
+
private patterns: Map<PolicyCategory, RegExp[]>;
|
|
17
|
+
|
|
18
|
+
constructor(customLists?: Partial<Record<PolicyCategory, string[]>>) {
|
|
19
|
+
this.patterns = new Map();
|
|
20
|
+
const merged = { ...DEFAULT_KEYWORD_LISTS, ...customLists };
|
|
21
|
+
for (const [cat, keywords] of Object.entries(merged)) {
|
|
22
|
+
if (keywords?.length) {
|
|
23
|
+
this.patterns.set(
|
|
24
|
+
cat as PolicyCategory,
|
|
25
|
+
keywords.map(kw => new RegExp(`\\b${kw.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i')),
|
|
26
|
+
);
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Scan text against enabled category keyword lists.
|
|
33
|
+
* @returns First match found, or null if clean.
|
|
34
|
+
*/
|
|
35
|
+
scan(
|
|
36
|
+
text: string,
|
|
37
|
+
enabledCategories: Partial<Record<PolicyCategory, CategoryConfig>>,
|
|
38
|
+
): KeywordMatch | null {
|
|
39
|
+
for (const [cat, patterns] of this.patterns) {
|
|
40
|
+
const cfg = enabledCategories[cat];
|
|
41
|
+
if (!cfg?.enabled) continue;
|
|
42
|
+
for (const re of patterns) {
|
|
43
|
+
const match = text.match(re);
|
|
44
|
+
if (match) return { category: cat, keyword: match[0] };
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
return null;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 2a: LLM-based policy violation classifier.
|
|
3
|
+
* Sends the full response text to an LLM with a structured classification prompt.
|
|
4
|
+
* @module content-policy-rewriter/LlmPolicyJudge
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import type { LlmInvoker, PolicyCategory, CategoryConfig, JudgeResult } from './types.js';
|
|
8
|
+
import { CATEGORY_DESCRIPTIONS } from './categories.js';
|
|
9
|
+
|
|
10
|
+
const SYSTEM_PROMPT = `You are a content policy classifier. Analyze the provided text and identify any violations of the enabled policy categories. Respond with JSON only. If no violations, respond: { "violations": [] }`;
|
|
11
|
+
|
|
12
|
+
export class LlmPolicyJudge {
|
|
13
|
+
constructor(private llmInvoker: LlmInvoker) {}
|
|
14
|
+
|
|
15
|
+
async classify(
|
|
16
|
+
text: string,
|
|
17
|
+
enabledCategories: Partial<Record<PolicyCategory, CategoryConfig>>,
|
|
18
|
+
customRules?: string,
|
|
19
|
+
): Promise<JudgeResult> {
|
|
20
|
+
const categoryList = Object.entries(enabledCategories)
|
|
21
|
+
.filter(([, cfg]) => cfg?.enabled)
|
|
22
|
+
.map(([cat]) => `- ${cat}: ${CATEGORY_DESCRIPTIONS[cat as PolicyCategory]}`)
|
|
23
|
+
.join('\n');
|
|
24
|
+
|
|
25
|
+
if (!categoryList) return { violations: [] };
|
|
26
|
+
|
|
27
|
+
let userPrompt = `Enabled categories:\n${categoryList}\n`;
|
|
28
|
+
if (customRules) userPrompt += `\nCustom rules: ${customRules}\n`;
|
|
29
|
+
userPrompt += `\nText to analyze:\n"""\n${text}\n"""\n\nRespond with JSON: { "violations": [{ "category": "...", "severity": "low|medium|high", "spans": ["offending phrase"] }] }`;
|
|
30
|
+
|
|
31
|
+
try {
|
|
32
|
+
const raw = await this.llmInvoker(SYSTEM_PROMPT, userPrompt);
|
|
33
|
+
const cleaned = raw.replace(/```json\n?|```\n?/g, '').trim();
|
|
34
|
+
const parsed = JSON.parse(cleaned) as JudgeResult;
|
|
35
|
+
if (!Array.isArray(parsed.violations)) return { violations: [] };
|
|
36
|
+
return parsed;
|
|
37
|
+
} catch {
|
|
38
|
+
return { violations: [] };
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Layer 2b: LLM-based content rewriter.
|
|
3
|
+
* Takes text with identified violations and rewrites it to remove them.
|
|
4
|
+
* @module content-policy-rewriter/LlmRewriter
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import type { LlmInvoker, PolicyViolation } from './types.js';
|
|
8
|
+
|
|
9
|
+
const SYSTEM_PROMPT = `Rewrite the following text to remove policy violations while preserving the original meaning, tone, and information content. Do not add disclaimers or commentary — just output the clean version.`;
|
|
10
|
+
|
|
11
|
+
export class LlmRewriter {
|
|
12
|
+
constructor(private llmInvoker: LlmInvoker) {}
|
|
13
|
+
|
|
14
|
+
async rewrite(text: string, violations: PolicyViolation[], customRules?: string): Promise<string> {
|
|
15
|
+
if (violations.length === 0) return text;
|
|
16
|
+
|
|
17
|
+
const violationList = violations
|
|
18
|
+
.map(v => `- ${v.category}: ${v.spans.map(s => `"${s}"`).join(', ')}`)
|
|
19
|
+
.join('\n');
|
|
20
|
+
|
|
21
|
+
let userPrompt = `Violations to remove:\n${violationList}\n`;
|
|
22
|
+
if (customRules) userPrompt += `\nAdditional rules: ${customRules}\n`;
|
|
23
|
+
userPrompt += `\nOriginal text:\n"""\n${text}\n"""\n\nRewritten text:`;
|
|
24
|
+
|
|
25
|
+
try {
|
|
26
|
+
const raw = await this.llmInvoker(SYSTEM_PROMPT, userPrompt);
|
|
27
|
+
return raw.replace(/^```\w*\s*\n?/gm, '').replace(/\n?```\s*$/gm, '').trim() || text;
|
|
28
|
+
} catch {
|
|
29
|
+
return text;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Built-in keyword lists per policy category.
|
|
3
|
+
* @module content-policy-rewriter/categories
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import type { PolicyCategory, CategoryConfig } from './types.js';
|
|
7
|
+
|
|
8
|
+
/** Default category descriptions used in LLM judge prompts. */
|
|
9
|
+
export const CATEGORY_DESCRIPTIONS: Record<PolicyCategory, string> = {
|
|
10
|
+
illegal_harmful: 'Child sexual abuse material (CSAM), sexual assault, bestiality, non-consensual acts, exploitation of minors, graphic torture',
|
|
11
|
+
adult: 'Consensual sexually explicit content, pornographic descriptions',
|
|
12
|
+
profanity: 'Slurs, vulgar language, obscenity, crude insults',
|
|
13
|
+
violence: 'Graphic violence, gore, detailed injury descriptions',
|
|
14
|
+
self_harm: 'Self-harm instructions, suicide methods, pro-anorexia content',
|
|
15
|
+
hate_speech: 'Discriminatory language, bigotry, slurs targeting protected groups',
|
|
16
|
+
illegal_activity: 'Drug synthesis instructions, weapons manufacturing, hacking tutorials',
|
|
17
|
+
custom: 'User-defined policy rules',
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
/** Default actions per category. */
|
|
21
|
+
export const DEFAULT_CATEGORY_ACTIONS: Record<PolicyCategory, 'block' | 'sanitize'> = {
|
|
22
|
+
illegal_harmful: 'block',
|
|
23
|
+
adult: 'sanitize',
|
|
24
|
+
profanity: 'sanitize',
|
|
25
|
+
violence: 'sanitize',
|
|
26
|
+
self_harm: 'block',
|
|
27
|
+
hate_speech: 'sanitize',
|
|
28
|
+
illegal_activity: 'block',
|
|
29
|
+
custom: 'sanitize',
|
|
30
|
+
};
|
|
31
|
+
|
|
32
|
+
/** Default enabled state — only illegal_harmful is on by default. */
|
|
33
|
+
export const DEFAULT_CATEGORY_ENABLED: Record<PolicyCategory, boolean> = {
|
|
34
|
+
illegal_harmful: true,
|
|
35
|
+
adult: false,
|
|
36
|
+
profanity: false,
|
|
37
|
+
violence: false,
|
|
38
|
+
self_harm: false,
|
|
39
|
+
hate_speech: false,
|
|
40
|
+
illegal_activity: false,
|
|
41
|
+
custom: false,
|
|
42
|
+
};
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Built-in keyword lists for the streaming pre-filter.
|
|
46
|
+
* Intentionally broad — false positives acceptable because the LLM judge
|
|
47
|
+
* runs on the final response for accuracy.
|
|
48
|
+
*/
|
|
49
|
+
export const DEFAULT_KEYWORD_LISTS: Partial<Record<PolicyCategory, string[]>> = {
|
|
50
|
+
profanity: [
|
|
51
|
+
'fuck', 'shit', 'bitch', 'asshole', 'cunt', 'dick', 'bastard',
|
|
52
|
+
'damn', 'nigger', 'faggot', 'retard', 'slut', 'whore',
|
|
53
|
+
],
|
|
54
|
+
illegal_harmful: [
|
|
55
|
+
'child porn', 'cp links', 'lolicon', 'underage sex', 'rape fantasy',
|
|
56
|
+
'snuff film', 'crush video',
|
|
57
|
+
],
|
|
58
|
+
self_harm: [
|
|
59
|
+
'how to kill yourself', 'suicide method', 'cut yourself', 'end it all',
|
|
60
|
+
],
|
|
61
|
+
hate_speech: [
|
|
62
|
+
'kill all', 'gas the', 'ethnic cleansing', 'race war', 'white power',
|
|
63
|
+
'heil hitler', 'death to',
|
|
64
|
+
],
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
/** Resolve effective config for a category. */
|
|
68
|
+
export function resolveCategoryConfig(
|
|
69
|
+
category: PolicyCategory,
|
|
70
|
+
userConfig?: Partial<Record<PolicyCategory, CategoryConfig>>,
|
|
71
|
+
): { enabled: boolean; action: 'block' | 'sanitize' } {
|
|
72
|
+
const cfg = userConfig?.[category];
|
|
73
|
+
return {
|
|
74
|
+
enabled: cfg?.enabled ?? DEFAULT_CATEGORY_ENABLED[category],
|
|
75
|
+
action: cfg?.action ?? DEFAULT_CATEGORY_ACTIONS[category],
|
|
76
|
+
};
|
|
77
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Content Policy Rewriter — extension pack factory.
|
|
3
|
+
*
|
|
4
|
+
* Opt-in guardrail that detects content policy violations in agent output
|
|
5
|
+
* and either blocks or rewrites them via LLM. Agents are uncensored by
|
|
6
|
+
* default — this extension only activates when explicitly configured.
|
|
7
|
+
*
|
|
8
|
+
* @module content-policy-rewriter
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import type { ContentPolicyRewriterConfig, LlmInvoker } from './types.js';
|
|
12
|
+
import { ContentPolicyRewriterGuardrail } from './ContentPolicyRewriterGuardrail.js';
|
|
13
|
+
import { resolvePreset } from './presets.js';
|
|
14
|
+
|
|
15
|
+
export * from './types.js';
|
|
16
|
+
export { ContentPolicyRewriterGuardrail } from './ContentPolicyRewriterGuardrail.js';
|
|
17
|
+
export { KeywordPreFilter } from './KeywordPreFilter.js';
|
|
18
|
+
export { LlmPolicyJudge } from './LlmPolicyJudge.js';
|
|
19
|
+
export { LlmRewriter } from './LlmRewriter.js';
|
|
20
|
+
export { resolvePreset } from './presets.js';
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Create the content policy rewriter extension pack.
|
|
24
|
+
*
|
|
25
|
+
* @param config - Policy configuration or preset name.
|
|
26
|
+
* @param llmInvoker - Callback to invoke an LLM for classification and rewriting.
|
|
27
|
+
* @returns A configured ExtensionPack.
|
|
28
|
+
*/
|
|
29
|
+
export function createContentPolicyRewriter(
|
|
30
|
+
config: ContentPolicyRewriterConfig | string = {},
|
|
31
|
+
llmInvoker?: LlmInvoker,
|
|
32
|
+
) {
|
|
33
|
+
const resolved = typeof config === 'string' ? resolvePreset(config as any) : config;
|
|
34
|
+
|
|
35
|
+
const invoker: LlmInvoker = llmInvoker ?? resolved.llmInvoker ?? createDefaultInvoker(resolved);
|
|
36
|
+
const guardrail = new ContentPolicyRewriterGuardrail({ ...resolved, llmInvoker: invoker });
|
|
37
|
+
|
|
38
|
+
return {
|
|
39
|
+
name: 'content-policy-rewriter',
|
|
40
|
+
version: '0.1.0',
|
|
41
|
+
descriptors: [
|
|
42
|
+
{
|
|
43
|
+
id: 'content-policy-rewriter-guardrail',
|
|
44
|
+
kind: 'guardrail' as const,
|
|
45
|
+
priority: 3,
|
|
46
|
+
payload: guardrail,
|
|
47
|
+
},
|
|
48
|
+
],
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Manifest factory bridge for extension loader. */
|
|
53
|
+
export function createExtensionPack(context: any) {
|
|
54
|
+
return createContentPolicyRewriter(
|
|
55
|
+
context.options as ContentPolicyRewriterConfig,
|
|
56
|
+
context.llmInvoker,
|
|
57
|
+
);
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** Default LLM invoker using fetch to OpenAI-compatible API. */
|
|
61
|
+
function createDefaultInvoker(config: ContentPolicyRewriterConfig): LlmInvoker {
|
|
62
|
+
return async (systemPrompt: string, userPrompt: string): Promise<string> => {
|
|
63
|
+
const apiKey = process.env.OPENAI_API_KEY || process.env.OPENROUTER_API_KEY;
|
|
64
|
+
if (!apiKey) throw new Error('No LLM API key for content policy judge');
|
|
65
|
+
|
|
66
|
+
const isOpenRouter = !process.env.OPENAI_API_KEY && !!process.env.OPENROUTER_API_KEY;
|
|
67
|
+
const baseUrl = isOpenRouter ? 'https://openrouter.ai/api/v1' : 'https://api.openai.com/v1';
|
|
68
|
+
const model = config.llm?.model ?? (isOpenRouter ? 'anthropic/claude-haiku-4-5-20251001' : 'gpt-4o-mini');
|
|
69
|
+
|
|
70
|
+
const res = await fetch(`${baseUrl}/chat/completions`, {
|
|
71
|
+
method: 'POST',
|
|
72
|
+
headers: {
|
|
73
|
+
'Content-Type': 'application/json',
|
|
74
|
+
Authorization: `Bearer ${apiKey}`,
|
|
75
|
+
...(isOpenRouter && { 'HTTP-Referer': 'https://agentos.sh' }),
|
|
76
|
+
},
|
|
77
|
+
body: JSON.stringify({
|
|
78
|
+
model,
|
|
79
|
+
messages: [
|
|
80
|
+
{ role: 'system', content: systemPrompt },
|
|
81
|
+
{ role: 'user', content: userPrompt },
|
|
82
|
+
],
|
|
83
|
+
max_tokens: 2048,
|
|
84
|
+
temperature: 0.1,
|
|
85
|
+
}),
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
if (!res.ok) throw new Error(`LLM API returned ${res.status}`);
|
|
89
|
+
const data = (await res.json()) as any;
|
|
90
|
+
return data.choices?.[0]?.message?.content ?? '';
|
|
91
|
+
};
|
|
92
|
+
}
|
package/src/presets.ts
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Preset configurations for common content policy scenarios.
|
|
3
|
+
* @module content-policy-rewriter/presets
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import type { ContentPolicyRewriterConfig, ContentPolicyPreset, PolicyCategory, CategoryConfig } from './types.js';
|
|
7
|
+
|
|
8
|
+
const ALL_ENABLED_SANITIZE: Partial<Record<PolicyCategory, CategoryConfig>> = {
|
|
9
|
+
illegal_harmful: { enabled: true, action: 'block' },
|
|
10
|
+
adult: { enabled: true, action: 'sanitize' },
|
|
11
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
12
|
+
violence: { enabled: true, action: 'sanitize' },
|
|
13
|
+
self_harm: { enabled: true, action: 'block' },
|
|
14
|
+
hate_speech: { enabled: true, action: 'sanitize' },
|
|
15
|
+
illegal_activity: { enabled: true, action: 'block' },
|
|
16
|
+
};
|
|
17
|
+
|
|
18
|
+
const PRESET_CONFIGS: Record<ContentPolicyPreset, Partial<ContentPolicyRewriterConfig>> = {
|
|
19
|
+
uncensored: {
|
|
20
|
+
categories: {
|
|
21
|
+
illegal_harmful: { enabled: false },
|
|
22
|
+
adult: { enabled: false },
|
|
23
|
+
profanity: { enabled: false },
|
|
24
|
+
violence: { enabled: false },
|
|
25
|
+
self_harm: { enabled: false },
|
|
26
|
+
hate_speech: { enabled: false },
|
|
27
|
+
illegal_activity: { enabled: false },
|
|
28
|
+
},
|
|
29
|
+
},
|
|
30
|
+
'uncensored-safe': {
|
|
31
|
+
categories: {
|
|
32
|
+
illegal_harmful: { enabled: true, action: 'block' },
|
|
33
|
+
},
|
|
34
|
+
},
|
|
35
|
+
'family-friendly': {
|
|
36
|
+
categories: ALL_ENABLED_SANITIZE,
|
|
37
|
+
},
|
|
38
|
+
enterprise: {
|
|
39
|
+
categories: ALL_ENABLED_SANITIZE,
|
|
40
|
+
},
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
/** Resolve a preset string or config object into a full config. */
|
|
44
|
+
export function resolvePreset(
|
|
45
|
+
input: ContentPolicyPreset | ContentPolicyRewriterConfig,
|
|
46
|
+
): ContentPolicyRewriterConfig {
|
|
47
|
+
if (typeof input === 'string') {
|
|
48
|
+
const preset = PRESET_CONFIGS[input];
|
|
49
|
+
if (!preset) return {};
|
|
50
|
+
return { ...preset };
|
|
51
|
+
}
|
|
52
|
+
return input;
|
|
53
|
+
}
|
package/src/types.ts
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Types for the content policy rewriter guardrail.
|
|
3
|
+
* @module content-policy-rewriter/types
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/** Policy categories that can be individually enabled/disabled. */
|
|
7
|
+
export type PolicyCategory =
|
|
8
|
+
| 'illegal_harmful'
|
|
9
|
+
| 'adult'
|
|
10
|
+
| 'profanity'
|
|
11
|
+
| 'violence'
|
|
12
|
+
| 'self_harm'
|
|
13
|
+
| 'hate_speech'
|
|
14
|
+
| 'illegal_activity'
|
|
15
|
+
| 'custom';
|
|
16
|
+
|
|
17
|
+
export const ALL_POLICY_CATEGORIES: PolicyCategory[] = [
|
|
18
|
+
'illegal_harmful', 'adult', 'profanity', 'violence',
|
|
19
|
+
'self_harm', 'hate_speech', 'illegal_activity', 'custom',
|
|
20
|
+
];
|
|
21
|
+
|
|
22
|
+
/** Per-category configuration. */
|
|
23
|
+
export interface CategoryConfig {
|
|
24
|
+
enabled?: boolean;
|
|
25
|
+
action?: 'block' | 'sanitize';
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** A detected policy violation. */
|
|
29
|
+
export interface PolicyViolation {
|
|
30
|
+
category: PolicyCategory;
|
|
31
|
+
severity: 'low' | 'medium' | 'high';
|
|
32
|
+
spans: string[];
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Result from the LLM policy judge. */
|
|
36
|
+
export interface JudgeResult {
|
|
37
|
+
violations: PolicyViolation[];
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** LLM invoker callback — matches the pattern used by ml-classifiers. */
|
|
41
|
+
export type LlmInvoker = (systemPrompt: string, userPrompt: string) => Promise<string>;
|
|
42
|
+
|
|
43
|
+
/** Full configuration for the content policy rewriter. */
|
|
44
|
+
export interface ContentPolicyRewriterConfig {
|
|
45
|
+
categories?: Partial<Record<PolicyCategory, CategoryConfig>>;
|
|
46
|
+
customRules?: string;
|
|
47
|
+
llm?: {
|
|
48
|
+
provider?: string;
|
|
49
|
+
model?: string;
|
|
50
|
+
};
|
|
51
|
+
/** Override built-in keyword lists per category. */
|
|
52
|
+
keywordLists?: Partial<Record<PolicyCategory, string[]>>;
|
|
53
|
+
/** Enable keyword pre-filter on streaming chunks. Default: true. */
|
|
54
|
+
streamingPreFilter?: boolean;
|
|
55
|
+
/** LLM invoker callback. If not provided, uses a default fetch-based invoker. */
|
|
56
|
+
llmInvoker?: LlmInvoker;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** Preset names for shorthand configuration. */
|
|
60
|
+
export type ContentPolicyPreset = 'uncensored' | 'uncensored-safe' | 'family-friendly' | 'enterprise';
|