@framers/agentos-ext-content-policy-rewriter 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +91 -0
- package/dist/ContentPolicyRewriterGuardrail.d.ts +30 -0
- package/dist/ContentPolicyRewriterGuardrail.js +104 -0
- package/dist/KeywordPreFilter.d.ts +19 -0
- package/dist/KeywordPreFilter.js +35 -0
- package/dist/LlmPolicyJudge.d.ts +11 -0
- package/dist/LlmPolicyJudge.js +36 -0
- package/dist/LlmRewriter.d.ts +11 -0
- package/dist/LlmRewriter.js +30 -0
- package/dist/categories.d.ts +22 -0
- package/dist/categories.js +67 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.js +77 -0
- package/dist/presets.d.ts +7 -0
- package/dist/presets.js +47 -0
- package/dist/types.d.ts +41 -0
- package/dist/types.js +8 -0
- package/package.json +29 -0
- package/src/ContentPolicyRewriterGuardrail.ts +126 -0
- package/src/KeywordPreFilter.ts +49 -0
- package/src/LlmPolicyJudge.ts +41 -0
- package/src/LlmRewriter.ts +32 -0
- package/src/categories.ts +77 -0
- package/src/index.ts +92 -0
- package/src/presets.ts +53 -0
- package/src/types.ts +60 -0
- package/test/ContentPolicyRewriter.spec.ts +79 -0
- package/test/KeywordPreFilter.spec.ts +46 -0
- package/test/LlmPolicyJudge.spec.ts +51 -0
- package/test/LlmRewriter.spec.ts +34 -0
- package/test/presets.spec.ts +30 -0
- package/tsconfig.json +14 -0
- package/vitest.config.ts +8 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import { describe, it, expect, vi } from 'vitest';
|
|
2
|
+
import { ContentPolicyRewriterGuardrail } from '../src/ContentPolicyRewriterGuardrail.js';
|
|
3
|
+
|
|
4
|
+
describe('ContentPolicyRewriterGuardrail', () => {
|
|
5
|
+
it('has correct guardrail config', () => {
|
|
6
|
+
const g = new ContentPolicyRewriterGuardrail({
|
|
7
|
+
llmInvoker: vi.fn(),
|
|
8
|
+
categories: { profanity: { enabled: true } },
|
|
9
|
+
});
|
|
10
|
+
expect(g.config.canSanitize).toBe(true);
|
|
11
|
+
expect(g.config.evaluateStreamingChunks).toBe(true);
|
|
12
|
+
});
|
|
13
|
+
|
|
14
|
+
it('allows clean streaming chunk', async () => {
|
|
15
|
+
const g = new ContentPolicyRewriterGuardrail({
|
|
16
|
+
llmInvoker: vi.fn(),
|
|
17
|
+
categories: { profanity: { enabled: true } },
|
|
18
|
+
});
|
|
19
|
+
const result = await g.evaluateOutput({
|
|
20
|
+
chunk: { type: 'TEXT_DELTA', text: 'Hello world' },
|
|
21
|
+
});
|
|
22
|
+
expect(result).toBeNull();
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
it('blocks streaming chunk with keyword match', async () => {
|
|
26
|
+
const g = new ContentPolicyRewriterGuardrail({
|
|
27
|
+
llmInvoker: vi.fn(),
|
|
28
|
+
categories: { profanity: { enabled: true, action: 'block' } },
|
|
29
|
+
});
|
|
30
|
+
const result = await g.evaluateOutput({
|
|
31
|
+
chunk: { type: 'TEXT_DELTA', text: 'What the fuck' },
|
|
32
|
+
});
|
|
33
|
+
expect(result).not.toBeNull();
|
|
34
|
+
expect(result!.action).toBe('block');
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
it('sanitizes final response via LLM', async () => {
|
|
38
|
+
const mockLlm = vi.fn();
|
|
39
|
+
mockLlm.mockResolvedValueOnce('{ "violations": [{ "category": "profanity", "severity": "medium", "spans": ["damn"] }] }');
|
|
40
|
+
mockLlm.mockResolvedValueOnce('The person was very upset.');
|
|
41
|
+
|
|
42
|
+
const g = new ContentPolicyRewriterGuardrail({
|
|
43
|
+
llmInvoker: mockLlm,
|
|
44
|
+
categories: { profanity: { enabled: true, action: 'sanitize' } },
|
|
45
|
+
});
|
|
46
|
+
const result = await g.evaluateOutput({
|
|
47
|
+
chunk: { type: 'FINAL_RESPONSE', finalResponseText: 'The person was damn upset.' },
|
|
48
|
+
});
|
|
49
|
+
expect(result).not.toBeNull();
|
|
50
|
+
expect(result!.action).toBe('sanitize');
|
|
51
|
+
expect(result!.modifiedText).toBe('The person was very upset.');
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
it('allows clean final response (no LLM rewrite call)', async () => {
|
|
55
|
+
const mockLlm = vi.fn();
|
|
56
|
+
mockLlm.mockResolvedValueOnce('{ "violations": [] }');
|
|
57
|
+
|
|
58
|
+
const g = new ContentPolicyRewriterGuardrail({
|
|
59
|
+
llmInvoker: mockLlm,
|
|
60
|
+
categories: { profanity: { enabled: true } },
|
|
61
|
+
});
|
|
62
|
+
const result = await g.evaluateOutput({
|
|
63
|
+
chunk: { type: 'FINAL_RESPONSE', finalResponseText: 'Everything is fine.' },
|
|
64
|
+
});
|
|
65
|
+
expect(result).toBeNull();
|
|
66
|
+
expect(mockLlm).toHaveBeenCalledTimes(1);
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
it('returns null when no categories enabled', async () => {
|
|
70
|
+
const g = new ContentPolicyRewriterGuardrail({
|
|
71
|
+
llmInvoker: vi.fn(),
|
|
72
|
+
categories: {},
|
|
73
|
+
});
|
|
74
|
+
const result = await g.evaluateOutput({
|
|
75
|
+
chunk: { type: 'FINAL_RESPONSE', finalResponseText: 'Anything goes.' },
|
|
76
|
+
});
|
|
77
|
+
expect(result).toBeNull();
|
|
78
|
+
});
|
|
79
|
+
});
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { describe, it, expect } from 'vitest';
|
|
2
|
+
import { KeywordPreFilter } from '../src/KeywordPreFilter.js';
|
|
3
|
+
|
|
4
|
+
describe('KeywordPreFilter', () => {
|
|
5
|
+
const filter = new KeywordPreFilter();
|
|
6
|
+
|
|
7
|
+
it('returns null for clean text', () => {
|
|
8
|
+
const result = filter.scan('Hello, how are you?', {
|
|
9
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
10
|
+
});
|
|
11
|
+
expect(result).toBeNull();
|
|
12
|
+
});
|
|
13
|
+
|
|
14
|
+
it('detects profanity keyword', () => {
|
|
15
|
+
const result = filter.scan('What the fuck is this?', {
|
|
16
|
+
profanity: { enabled: true, action: 'block' },
|
|
17
|
+
});
|
|
18
|
+
expect(result).not.toBeNull();
|
|
19
|
+
expect(result!.category).toBe('profanity');
|
|
20
|
+
});
|
|
21
|
+
|
|
22
|
+
it('skips disabled categories', () => {
|
|
23
|
+
const result = filter.scan('What the fuck is this?', {
|
|
24
|
+
profanity: { enabled: false, action: 'block' },
|
|
25
|
+
});
|
|
26
|
+
expect(result).toBeNull();
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
it('uses custom keyword list when provided', () => {
|
|
30
|
+
const customFilter = new KeywordPreFilter({
|
|
31
|
+
profanity: ['dingus', 'bozo'],
|
|
32
|
+
});
|
|
33
|
+
const result = customFilter.scan('You absolute bozo', {
|
|
34
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
35
|
+
});
|
|
36
|
+
expect(result).not.toBeNull();
|
|
37
|
+
expect(result!.category).toBe('profanity');
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
it('is case-insensitive', () => {
|
|
41
|
+
const result = filter.scan('What the FUCK', {
|
|
42
|
+
profanity: { enabled: true, action: 'block' },
|
|
43
|
+
});
|
|
44
|
+
expect(result).not.toBeNull();
|
|
45
|
+
});
|
|
46
|
+
});
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
|
2
|
+
import { LlmPolicyJudge } from '../src/LlmPolicyJudge.js';
|
|
3
|
+
import type { LlmInvoker } from '../src/types.js';
|
|
4
|
+
|
|
5
|
+
describe('LlmPolicyJudge', () => {
|
|
6
|
+
let mockLlm: LlmInvoker;
|
|
7
|
+
|
|
8
|
+
beforeEach(() => {
|
|
9
|
+
mockLlm = vi.fn();
|
|
10
|
+
});
|
|
11
|
+
|
|
12
|
+
it('returns empty violations for clean content', async () => {
|
|
13
|
+
(mockLlm as any).mockResolvedValueOnce('{ "violations": [] }');
|
|
14
|
+
const judge = new LlmPolicyJudge(mockLlm);
|
|
15
|
+
const result = await judge.classify('Hello world', {
|
|
16
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
17
|
+
});
|
|
18
|
+
expect(result.violations).toEqual([]);
|
|
19
|
+
});
|
|
20
|
+
|
|
21
|
+
it('parses violation response correctly', async () => {
|
|
22
|
+
(mockLlm as any).mockResolvedValueOnce(JSON.stringify({
|
|
23
|
+
violations: [{ category: 'profanity', severity: 'medium', spans: ['bad word'] }],
|
|
24
|
+
}));
|
|
25
|
+
const judge = new LlmPolicyJudge(mockLlm);
|
|
26
|
+
const result = await judge.classify('text with bad word', {
|
|
27
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
28
|
+
});
|
|
29
|
+
expect(result.violations).toHaveLength(1);
|
|
30
|
+
expect(result.violations[0].category).toBe('profanity');
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('handles malformed LLM response gracefully', async () => {
|
|
34
|
+
(mockLlm as any).mockResolvedValueOnce('not json at all');
|
|
35
|
+
const judge = new LlmPolicyJudge(mockLlm);
|
|
36
|
+
const result = await judge.classify('some text', {
|
|
37
|
+
profanity: { enabled: true, action: 'sanitize' },
|
|
38
|
+
});
|
|
39
|
+
expect(result.violations).toEqual([]);
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
it('includes custom rules in prompt', async () => {
|
|
43
|
+
(mockLlm as any).mockResolvedValueOnce('{ "violations": [] }');
|
|
44
|
+
const judge = new LlmPolicyJudge(mockLlm);
|
|
45
|
+
await judge.classify('text', {
|
|
46
|
+
custom: { enabled: true, action: 'sanitize' },
|
|
47
|
+
}, 'Never mention competitor X');
|
|
48
|
+
const call = (mockLlm as any).mock.calls[0];
|
|
49
|
+
expect(call[1]).toContain('Never mention competitor X');
|
|
50
|
+
});
|
|
51
|
+
});
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import { describe, it, expect, vi } from 'vitest';
|
|
2
|
+
import { LlmRewriter } from '../src/LlmRewriter.js';
|
|
3
|
+
import type { LlmInvoker, PolicyViolation } from '../src/types.js';
|
|
4
|
+
|
|
5
|
+
describe('LlmRewriter', () => {
|
|
6
|
+
const mockLlm: LlmInvoker = vi.fn();
|
|
7
|
+
|
|
8
|
+
it('returns rewritten text from LLM', async () => {
|
|
9
|
+
(mockLlm as any).mockResolvedValueOnce('The person expressed frustration.');
|
|
10
|
+
const rewriter = new LlmRewriter(mockLlm);
|
|
11
|
+
const violations: PolicyViolation[] = [
|
|
12
|
+
{ category: 'profanity', severity: 'medium', spans: ['bad word'] },
|
|
13
|
+
];
|
|
14
|
+
const result = await rewriter.rewrite('The person said bad word in anger.', violations);
|
|
15
|
+
expect(result).toBe('The person expressed frustration.');
|
|
16
|
+
});
|
|
17
|
+
|
|
18
|
+
it('returns original text on LLM failure', async () => {
|
|
19
|
+
(mockLlm as any).mockRejectedValueOnce(new Error('LLM unavailable'));
|
|
20
|
+
const rewriter = new LlmRewriter(mockLlm);
|
|
21
|
+
const violations = [{ category: 'profanity' as const, severity: 'low' as const, spans: ['x'] }];
|
|
22
|
+
const result = await rewriter.rewrite('original text', violations);
|
|
23
|
+
expect(result).toBe('original text');
|
|
24
|
+
});
|
|
25
|
+
|
|
26
|
+
it('strips markdown fences from LLM response', async () => {
|
|
27
|
+
(mockLlm as any).mockResolvedValueOnce('```\nClean version of the text.\n```');
|
|
28
|
+
const rewriter = new LlmRewriter(mockLlm);
|
|
29
|
+
const result = await rewriter.rewrite('dirty text', [
|
|
30
|
+
{ category: 'profanity', severity: 'low', spans: ['dirty'] },
|
|
31
|
+
]);
|
|
32
|
+
expect(result).toBe('Clean version of the text.');
|
|
33
|
+
});
|
|
34
|
+
});
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { describe, it, expect } from 'vitest';
|
|
2
|
+
import { resolvePreset } from '../src/presets.js';
|
|
3
|
+
|
|
4
|
+
describe('presets', () => {
|
|
5
|
+
it('uncensored disables all categories', () => {
|
|
6
|
+
const cfg = resolvePreset('uncensored');
|
|
7
|
+
for (const cat of Object.values(cfg.categories ?? {})) {
|
|
8
|
+
expect(cat?.enabled).toBe(false);
|
|
9
|
+
}
|
|
10
|
+
});
|
|
11
|
+
|
|
12
|
+
it('uncensored-safe enables only illegal_harmful', () => {
|
|
13
|
+
const cfg = resolvePreset('uncensored-safe');
|
|
14
|
+
expect(cfg.categories?.illegal_harmful?.enabled).toBe(true);
|
|
15
|
+
expect(cfg.categories?.adult?.enabled).toBeUndefined();
|
|
16
|
+
});
|
|
17
|
+
|
|
18
|
+
it('family-friendly enables all categories', () => {
|
|
19
|
+
const cfg = resolvePreset('family-friendly');
|
|
20
|
+
expect(cfg.categories?.adult?.enabled).toBe(true);
|
|
21
|
+
expect(cfg.categories?.profanity?.enabled).toBe(true);
|
|
22
|
+
expect(cfg.categories?.violence?.enabled).toBe(true);
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
it('passes through config objects unchanged', () => {
|
|
26
|
+
const input = { categories: { adult: { enabled: true } } };
|
|
27
|
+
const cfg = resolvePreset(input);
|
|
28
|
+
expect(cfg).toEqual(input);
|
|
29
|
+
});
|
|
30
|
+
});
|
package/tsconfig.json
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"compilerOptions": {
|
|
3
|
+
"target": "ES2022",
|
|
4
|
+
"module": "ESNext",
|
|
5
|
+
"moduleResolution": "bundler",
|
|
6
|
+
"declaration": true,
|
|
7
|
+
"outDir": "dist",
|
|
8
|
+
"rootDir": "src",
|
|
9
|
+
"strict": true,
|
|
10
|
+
"esModuleInterop": true,
|
|
11
|
+
"skipLibCheck": true
|
|
12
|
+
},
|
|
13
|
+
"include": ["src"]
|
|
14
|
+
}
|