@royalcat/opencode-dcp-rc 4.0.1 → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,954 @@
1
+ /**
2
+ * Removes DCP-injected metadata from model output before it reaches the stored
3
+ * conversation or the user interface.
4
+ *
5
+ * The injected reminders, message-id tags and compact id lines are addressed to
6
+ * the model, but models occasionally copy them into their visible reply or
7
+ * reasoning. The request-side `stripHallucinations` pass keeps those copies out
8
+ * of later requests; this module keeps them out of the stored history in the
9
+ * first place by scrubbing the provider response stream. Two surfaces are
10
+ * covered: raw HTTP responses (`http.response` hook, native provider packages)
11
+ * and standardized AI SDK stream parts (`aisdk.language` hook).
12
+ *
13
+ * Every failure path falls back to the untouched response: scrubbing must never
14
+ * break or delay model traffic.
15
+ */
16
+
17
+ import type { Plugin } from "@opencode/plugin"
18
+ import type { Logger } from "../logger"
19
+
20
+ export interface OutputScrubConfig {
21
+ /** Strip `<dcp...>` reminder/tag blocks from model output. */
22
+ modelOutput: boolean
23
+ /** Strip line-standalone compact id tags (`@N@`, `@bN@`, `@blocked@`). */
24
+ messageIds: boolean
25
+ }
26
+
27
+ export function outputScrubEnabled(config: OutputScrubConfig): boolean {
28
+ return config.modelOutput || config.messageIds
29
+ }
30
+
31
+ // Local copies of the request-side patterns in lib/messages/utils.ts. They are
32
+ // intentionally duplicated: the request-side pass must stay untouched, and
33
+ // keeping the output-side patterns here makes the scrub module self-contained.
34
+ const DCP_PAIRED_TAG_REGEX = /<dcp[^>]*>[\s\S]*?<\/dcp[^>]*>/gi
35
+ const DCP_UNPAIRED_TAG_REGEX = /<\/?dcp[^>]*>/gi
36
+ const ID_LINE_REGEX =
37
+ /^[ \t]*@(?:[1-9]\d*|b[1-9]\d*|blocked)@(?:[ \t]+\[(?:low|medium|high)\])?[ \t]*$/
38
+
39
+ function isIdLine(line: string): boolean {
40
+ const value = line.endsWith("\r") ? line.slice(0, -1) : line
41
+ return ID_LINE_REGEX.test(value)
42
+ }
43
+
44
+ /** True while `text` could still grow into a complete compact id line. */
45
+ function isIdLinePrefix(text: string): boolean {
46
+ let index = 0
47
+ while (index < text.length && (text[index] === " " || text[index] === "\t")) {
48
+ index += 1
49
+ }
50
+ if (index === text.length) {
51
+ return true
52
+ }
53
+ if (text[index] !== "@") {
54
+ return false
55
+ }
56
+ index += 1
57
+ if (index === text.length) {
58
+ return true
59
+ }
60
+ if (text.startsWith("blocked", index)) {
61
+ index += 7
62
+ } else if (text[index] === "b") {
63
+ if (index + 1 === text.length) {
64
+ return true
65
+ }
66
+ if (!/\d/.test(text[index + 1]!)) {
67
+ return false
68
+ }
69
+ index += 2
70
+ while (index < text.length && /\d/.test(text[index]!)) {
71
+ index += 1
72
+ }
73
+ } else if (/[1-9]/.test(text[index]!)) {
74
+ index += 1
75
+ while (index < text.length && /\d/.test(text[index]!)) {
76
+ index += 1
77
+ }
78
+ } else {
79
+ return false
80
+ }
81
+ if (index === text.length) {
82
+ return true
83
+ }
84
+ if (text[index] !== "@") {
85
+ return false
86
+ }
87
+ return isIdPriorityPrefix(text.slice(index + 1))
88
+ }
89
+
90
+ /** After the closing `@`: optional whitespace and an optional priority label. */
91
+ function isIdPriorityPrefix(text: string): boolean {
92
+ if (text.length === 0 || /^[ \t]+$/.test(text)) {
93
+ return true
94
+ }
95
+ return /^[ \t]*\[(?:l(?:o(?:w)?)?|m(?:e(?:d(?:i(?:u(?:m)?)?)?)?)?|h(?:i(?:g(?:h)?)?)?)?\]?[ \t]*$/.test(
96
+ text,
97
+ )
98
+ }
99
+
100
+ export function stripDcpArtifacts(text: string, config: OutputScrubConfig): string {
101
+ let out = text
102
+ if (config.modelOutput) {
103
+ out = out.replace(DCP_PAIRED_TAG_REGEX, "").replace(DCP_UNPAIRED_TAG_REGEX, "")
104
+ }
105
+ if (config.messageIds && out.includes("\n")) {
106
+ out = out
107
+ .split("\n")
108
+ .filter((line) => !isIdLine(line))
109
+ .join("\n")
110
+ } else if (config.messageIds && isIdLine(out)) {
111
+ out = ""
112
+ }
113
+ return out
114
+ }
115
+
116
+ const TAG_OPEN = "<dcp"
117
+ const TAG_CLOSE = "</dcp"
118
+ const MAX_TAG_LENGTH = 512
119
+ const MAX_BLOCK_LENGTH = 8192
120
+
121
+ /** Longest suffix of `text` that could still be the start of a tag marker. */
122
+ function heldTagSuffix(text: string): string {
123
+ const max = Math.min(4, text.length)
124
+ for (let length = max; length >= 1; length -= 1) {
125
+ const suffix = text.slice(text.length - length).toLowerCase()
126
+ if (TAG_OPEN.startsWith(suffix) || TAG_CLOSE.startsWith(suffix)) {
127
+ return text.slice(text.length - length)
128
+ }
129
+ }
130
+ return ""
131
+ }
132
+
133
+ type TagMode = "text" | "open" | "close" | "block"
134
+
135
+ /**
136
+ * Stateful `<dcp...>` remover. Tags may be split across arbitrary chunk
137
+ * boundaries, so the stripper holds back only what could still complete into a
138
+ * tag. Unpaired opening tags release their body on flush (matching the
139
+ * request-side behavior of leaving non-tag content intact).
140
+ */
141
+ class DcpTagStripper {
142
+ private mode: TagMode = "text"
143
+ private carry = ""
144
+
145
+ push(chunk: string): string {
146
+ this.carry += chunk
147
+ let out = ""
148
+ for (;;) {
149
+ if (this.mode === "text") {
150
+ const lower = this.carry.toLowerCase()
151
+ const open = lower.indexOf(TAG_OPEN)
152
+ const close = lower.indexOf(TAG_CLOSE)
153
+ let index = -1
154
+ let next: TagMode = "open"
155
+ if (open === -1 && close === -1) {
156
+ index = -1
157
+ } else if (open === -1) {
158
+ index = close
159
+ next = "close"
160
+ } else if (close === -1) {
161
+ index = open
162
+ } else if (open <= close) {
163
+ index = open
164
+ } else {
165
+ index = close
166
+ next = "close"
167
+ }
168
+ if (index === -1) {
169
+ const held = heldTagSuffix(this.carry)
170
+ out += held ? this.carry.slice(0, -held.length) : this.carry
171
+ this.carry = held
172
+ break
173
+ }
174
+ out += this.carry.slice(0, index)
175
+ this.carry = this.carry.slice(index)
176
+ this.mode = next
177
+ continue
178
+ }
179
+ if (this.mode === "open" || this.mode === "close") {
180
+ const end = this.carry.indexOf(">")
181
+ if (end === -1) {
182
+ if (this.carry.length > MAX_TAG_LENGTH) {
183
+ // Not a real tag after all; fail open.
184
+ out += this.carry
185
+ this.carry = ""
186
+ this.mode = "text"
187
+ }
188
+ break
189
+ }
190
+ const wasOpen = this.mode === "open"
191
+ this.carry = this.carry.slice(end + 1)
192
+ this.mode = wasOpen ? "block" : "text"
193
+ continue
194
+ }
195
+ // Inside a paired block: drop everything through the closing tag.
196
+ const close = this.carry.toLowerCase().indexOf(TAG_CLOSE)
197
+ if (close !== -1) {
198
+ const end = this.carry.indexOf(">", close)
199
+ if (end !== -1) {
200
+ this.carry = this.carry.slice(end + 1)
201
+ this.mode = "text"
202
+ continue
203
+ }
204
+ }
205
+ if (this.carry.length > MAX_BLOCK_LENGTH) {
206
+ // Never seen a closing tag; fail open with the buffered body.
207
+ out += this.carry
208
+ this.carry = ""
209
+ this.mode = "text"
210
+ }
211
+ break
212
+ }
213
+ return out
214
+ }
215
+
216
+ flush(): string {
217
+ // A body without a closing tag is released (unpaired-tag semantics); a
218
+ // partial tag or closing marker has no useful content and is dropped.
219
+ const out = this.mode === "block" ? this.carry : ""
220
+ this.carry = ""
221
+ this.mode = "text"
222
+ return out
223
+ }
224
+
225
+ reset(): void {
226
+ this.carry = ""
227
+ this.mode = "text"
228
+ }
229
+ }
230
+
231
+ /**
232
+ * Stateful compact-id line remover. Only lines that consist of nothing but a
233
+ * compact id tag are removed; inline mentions elsewhere in the text are kept.
234
+ */
235
+ class IdLineStripper {
236
+ private carry = ""
237
+
238
+ push(chunk: string): string {
239
+ this.carry += chunk
240
+ let out = ""
241
+ for (;;) {
242
+ const newline = this.carry.indexOf("\n")
243
+ if (newline === -1) {
244
+ if (isIdLinePrefix(this.carry)) {
245
+ break
246
+ }
247
+ out += this.carry
248
+ this.carry = ""
249
+ break
250
+ }
251
+ const line = this.carry.slice(0, newline)
252
+ this.carry = this.carry.slice(newline + 1)
253
+ if (!isIdLine(line)) {
254
+ out += line + "\n"
255
+ }
256
+ }
257
+ return out
258
+ }
259
+
260
+ flush(): string {
261
+ const held = this.carry
262
+ this.carry = ""
263
+ return isIdLine(held) ? "" : held
264
+ }
265
+
266
+ reset(): void {
267
+ this.carry = ""
268
+ }
269
+ }
270
+
271
+ /** Streaming text scrubber: tag blocks first, then compact-id lines. */
272
+ export class DcpStreamScrubber {
273
+ private readonly tags = new DcpTagStripper()
274
+ private readonly ids = new IdLineStripper()
275
+
276
+ constructor(private readonly config: OutputScrubConfig) {}
277
+
278
+ push(chunk: string): string {
279
+ let out = chunk
280
+ if (this.config.modelOutput) {
281
+ out = this.tags.push(out)
282
+ }
283
+ if (this.config.messageIds) {
284
+ out = this.ids.push(out)
285
+ }
286
+ return out
287
+ }
288
+
289
+ flush(): string {
290
+ let out = this.config.modelOutput ? this.tags.flush() : ""
291
+ if (out && this.config.messageIds) {
292
+ out = this.ids.push(out)
293
+ }
294
+ if (this.config.messageIds) {
295
+ out += this.ids.flush()
296
+ }
297
+ return out
298
+ }
299
+
300
+ reset(): void {
301
+ this.tags.reset()
302
+ this.ids.reset()
303
+ }
304
+ }
305
+
306
+ function assignText(target: Record<string, any>, key: string, value: string): boolean {
307
+ if (target[key] === value) {
308
+ return false
309
+ }
310
+ target[key] = value
311
+ return true
312
+ }
313
+
314
+ const TEXT_PART_TYPES = new Set(["text", "output_text", "reasoning_text", "summary_text"])
315
+
316
+ /**
317
+ * Scrubs provider payloads in place. Stateful for streamed deltas, complete
318
+ * (stateless) for done/full payloads. Unknown shapes are left untouched.
319
+ */
320
+ export class ProviderPayloadScrubber {
321
+ private readonly streams = new Map<string, DcpStreamScrubber>()
322
+
323
+ constructor(private readonly config: OutputScrubConfig) {}
324
+
325
+ /** Scrub one streamed provider event. Returns true when it was modified. */
326
+ scrub(payload: unknown): boolean {
327
+ try {
328
+ if (!payload || typeof payload !== "object") {
329
+ return false
330
+ }
331
+ const record = payload as Record<string, any>
332
+ const type = typeof record.type === "string" ? record.type : ""
333
+ if (Array.isArray(record.choices)) {
334
+ return this.scrubChat(record, false)
335
+ }
336
+ if (Array.isArray(record.candidates)) {
337
+ return this.scrubGoogle(record, false)
338
+ }
339
+ if (type.startsWith("response.")) {
340
+ return this.scrubResponses(record)
341
+ }
342
+ if (
343
+ type.startsWith("content_block") ||
344
+ type === "message" ||
345
+ type === "message_delta"
346
+ ) {
347
+ return this.scrubAnthropic(record)
348
+ }
349
+ return false
350
+ } catch {
351
+ return false
352
+ }
353
+ }
354
+
355
+ /** Scrub a complete (non-streamed) JSON payload. */
356
+ scrubComplete(payload: unknown): boolean {
357
+ try {
358
+ if (Array.isArray(payload)) {
359
+ let changed = false
360
+ for (const item of payload) {
361
+ changed = this.scrubComplete(item) || changed
362
+ }
363
+ return changed
364
+ }
365
+ if (!payload || typeof payload !== "object") {
366
+ return false
367
+ }
368
+ const record = payload as Record<string, any>
369
+ if (Array.isArray(record.choices)) {
370
+ return this.scrubChat(record, true)
371
+ }
372
+ if (Array.isArray(record.candidates)) {
373
+ return this.scrubGoogle(record, true)
374
+ }
375
+ if (record.type === "message" && Array.isArray(record.content)) {
376
+ return this.scrubAnthropic(record)
377
+ }
378
+ if (Array.isArray(record.output)) {
379
+ return this.scrubItems(record.output)
380
+ }
381
+ if (typeof record.output_text === "string") {
382
+ return assignText(record, "output_text", this.complete(record.output_text))
383
+ }
384
+ return false
385
+ } catch {
386
+ return false
387
+ }
388
+ }
389
+
390
+ private complete(text: string): string {
391
+ return stripDcpArtifacts(text, this.config)
392
+ }
393
+
394
+ private stream(key: string): DcpStreamScrubber {
395
+ let scrubber = this.streams.get(key)
396
+ if (!scrubber) {
397
+ scrubber = new DcpStreamScrubber(this.config)
398
+ this.streams.set(key, scrubber)
399
+ }
400
+ return scrubber
401
+ }
402
+
403
+ private textFor(value: string, key: string | undefined, complete: boolean): string {
404
+ if (complete || !key) {
405
+ return this.complete(value)
406
+ }
407
+ return this.stream(key).push(value)
408
+ }
409
+
410
+ private scrubField(target: any, key: string, streamKey?: string, complete = false): boolean {
411
+ if (!target || typeof target !== "object" || typeof target[key] !== "string") {
412
+ return false
413
+ }
414
+ return assignText(target, key, this.textFor(target[key], streamKey, complete))
415
+ }
416
+
417
+ private scrubParts(parts: unknown, key?: string, complete = false): boolean {
418
+ if (!Array.isArray(parts)) {
419
+ return false
420
+ }
421
+ let changed = false
422
+ parts.forEach((part: any, index: number) => {
423
+ if (!part || typeof part !== "object" || typeof part.text !== "string") {
424
+ return
425
+ }
426
+ if (part.type && !TEXT_PART_TYPES.has(part.type)) {
427
+ return
428
+ }
429
+ changed =
430
+ assignText(
431
+ part,
432
+ "text",
433
+ this.textFor(part.text, key ? `${key}:${index}` : undefined, complete),
434
+ ) || changed
435
+ })
436
+ return changed
437
+ }
438
+
439
+ private scrubChat(record: any, complete: boolean): boolean {
440
+ let changed = false
441
+ const choices = Array.isArray(record.choices) ? record.choices : []
442
+ choices.forEach((choice: any, index: number) => {
443
+ const delta = choice?.delta
444
+ if (delta && typeof delta === "object") {
445
+ changed = this.scrubField(delta, "content", `chat:content:${index}`) || changed
446
+ changed =
447
+ this.scrubField(delta, "reasoning_content", `chat:reasoning:${index}`) ||
448
+ changed
449
+ changed = this.scrubField(delta, "reasoning", `chat:reasoning:${index}`) || changed
450
+ changed = this.scrubParts(delta.content, `chat:parts:${index}`) || changed
451
+ if (choice?.finish_reason) {
452
+ this.streams.delete(`chat:content:${index}`)
453
+ this.streams.delete(`chat:reasoning:${index}`)
454
+ this.streams.delete(`chat:parts:${index}`)
455
+ }
456
+ }
457
+ const message = choice?.message
458
+ if (message && typeof message === "object") {
459
+ changed = this.scrubField(message, "content", undefined, true) || changed
460
+ changed = this.scrubField(message, "reasoning_content", undefined, true) || changed
461
+ changed = this.scrubField(message, "reasoning", undefined, true) || changed
462
+ changed = this.scrubParts(message.content, undefined, true) || changed
463
+ }
464
+ })
465
+ return changed
466
+ }
467
+
468
+ private scrubResponses(record: any): boolean {
469
+ const type = record.type as string
470
+ const itemId = record.item_id ?? ""
471
+ const outputIndex = record.output_index ?? ""
472
+ const contentIndex = record.content_index ?? ""
473
+ const key = (family: string) => `resp:${family}:${itemId}:${outputIndex}:${contentIndex}`
474
+ switch (type) {
475
+ case "response.output_text.delta":
476
+ return this.scrubField(record, "delta", key("text"))
477
+ case "response.reasoning_text.delta":
478
+ return this.scrubField(record, "delta", key("reasoning"))
479
+ case "response.reasoning_summary_text.delta":
480
+ return this.scrubField(record, "delta", key("summary"))
481
+ case "response.output_text.done":
482
+ case "response.reasoning_text.done":
483
+ case "response.reasoning_summary_text.done":
484
+ return this.scrubField(record, "text", undefined, true)
485
+ case "response.content_part.added":
486
+ case "response.content_part.done":
487
+ case "response.reasoning_summary_part.added":
488
+ case "response.reasoning_summary_part.done":
489
+ return this.scrubParts(record.part ? [record.part] : undefined, undefined, true)
490
+ case "response.output_item.added":
491
+ case "response.output_item.done":
492
+ return this.scrubItems(record.item ? [record.item] : undefined)
493
+ case "response.completed": {
494
+ const changed = this.scrubResponseObject(record.response)
495
+ this.streams.clear()
496
+ return changed
497
+ }
498
+ default:
499
+ return false
500
+ }
501
+ }
502
+
503
+ private scrubResponseObject(response: any): boolean {
504
+ if (!response || typeof response !== "object") {
505
+ return false
506
+ }
507
+ let changed = this.scrubItems(response.output)
508
+ if (typeof response.output_text === "string") {
509
+ changed =
510
+ assignText(response, "output_text", this.complete(response.output_text)) || changed
511
+ }
512
+ return changed
513
+ }
514
+
515
+ private scrubItems(items: unknown): boolean {
516
+ if (!Array.isArray(items)) {
517
+ return false
518
+ }
519
+ let changed = false
520
+ items.forEach((entry: any) => {
521
+ if (!entry || typeof entry !== "object") {
522
+ return
523
+ }
524
+ changed = this.scrubParts(entry.content, undefined, true) || changed
525
+ changed = this.scrubParts(entry.summary, undefined, true) || changed
526
+ })
527
+ return changed
528
+ }
529
+
530
+ private scrubAnthropic(record: any): boolean {
531
+ const type = record.type
532
+ if (type === "content_block_start") {
533
+ const block = record.content_block
534
+ if (!block || typeof block !== "object") {
535
+ return false
536
+ }
537
+ if (block.type === "text") {
538
+ return this.scrubField(block, "text", `anthropic:text:${record.index}`)
539
+ }
540
+ if (block.type === "thinking") {
541
+ return this.scrubField(block, "thinking", `anthropic:thinking:${record.index}`)
542
+ }
543
+ return false
544
+ }
545
+ if (type === "content_block_delta") {
546
+ const delta = record.delta
547
+ if (!delta || typeof delta !== "object") {
548
+ return false
549
+ }
550
+ if (delta.type === "text_delta") {
551
+ return this.scrubField(delta, "text", `anthropic:text:${record.index}`)
552
+ }
553
+ if (delta.type === "thinking_delta") {
554
+ return this.scrubField(delta, "thinking", `anthropic:thinking:${record.index}`)
555
+ }
556
+ return false
557
+ }
558
+ if (type === "content_block_stop") {
559
+ this.streams.delete(`anthropic:text:${record.index}`)
560
+ this.streams.delete(`anthropic:thinking:${record.index}`)
561
+ return false
562
+ }
563
+ if (type === "message" && Array.isArray(record.content)) {
564
+ let changed = false
565
+ record.content.forEach((block: any) => {
566
+ if (block?.type === "text") {
567
+ changed = this.scrubField(block, "text", undefined, true) || changed
568
+ } else if (block?.type === "thinking") {
569
+ changed = this.scrubField(block, "thinking", undefined, true) || changed
570
+ }
571
+ })
572
+ return changed
573
+ }
574
+ return false
575
+ }
576
+
577
+ private scrubGoogle(record: any, complete: boolean): boolean {
578
+ const candidates = Array.isArray(record.candidates) ? record.candidates : []
579
+ let changed = false
580
+ candidates.forEach((candidate: any, candidateIndex: number) => {
581
+ const parts = candidate?.content?.parts
582
+ if (Array.isArray(parts)) {
583
+ parts.forEach((part: any, partIndex: number) => {
584
+ if (!part || typeof part !== "object" || typeof part.text !== "string") {
585
+ return
586
+ }
587
+ changed =
588
+ assignText(
589
+ part,
590
+ "text",
591
+ this.textFor(
592
+ part.text,
593
+ complete ? undefined : `google:${candidateIndex}:${partIndex}`,
594
+ complete,
595
+ ),
596
+ ) || changed
597
+ })
598
+ }
599
+ if (candidate?.finishReason) {
600
+ this.streams.clear()
601
+ }
602
+ })
603
+ return changed
604
+ }
605
+ }
606
+
607
+ function findBoundary(buffer: string): { start: number; end: number; separator: string } | null {
608
+ const lf = buffer.indexOf("\n\n")
609
+ const crlf = buffer.indexOf("\r\n\r\n")
610
+ if (lf === -1 && crlf === -1) {
611
+ return null
612
+ }
613
+ if (crlf !== -1 && (lf === -1 || crlf <= lf)) {
614
+ return { start: crlf, end: crlf + 4, separator: "\r\n\r\n" }
615
+ }
616
+ return { start: lf, end: lf + 2, separator: "\n\n" }
617
+ }
618
+
619
+ /** Scrub a complete JSON body (non-streamed responses). */
620
+ export function scrubJsonBody(
621
+ body: string,
622
+ config: OutputScrubConfig,
623
+ scrubber?: ProviderPayloadScrubber,
624
+ ): string {
625
+ const trimmed = body.trim()
626
+ if (!trimmed.startsWith("{") && !trimmed.startsWith("[")) {
627
+ return body
628
+ }
629
+ try {
630
+ const payload = JSON.parse(trimmed)
631
+ const worker = scrubber ?? new ProviderPayloadScrubber(config)
632
+ if (worker.scrubComplete(payload)) {
633
+ return JSON.stringify(payload)
634
+ }
635
+ } catch {
636
+ // Malformed or unsupported bodies pass through untouched.
637
+ }
638
+ return body
639
+ }
640
+
641
+ /** SSE transformer that scrubs known provider events and passes the rest through. */
642
+ export function createSseScrubTransform(config: OutputScrubConfig): any {
643
+ const TransformStreamImpl = (globalThis as any).TransformStream
644
+ const decoder = new TextDecoder()
645
+ const encoder = new TextEncoder()
646
+ const scrubber = new ProviderPayloadScrubber(config)
647
+ let buffer = ""
648
+
649
+ const processBlock = (block: string): string => {
650
+ try {
651
+ const lines = block.split("\n")
652
+ const dataLines: Array<{ index: number; space: string; value: string }> = []
653
+ lines.forEach((raw, index) => {
654
+ const line = raw.endsWith("\r") ? raw.slice(0, -1) : raw
655
+ const match = /^data:( ?)(.*)$/.exec(line)
656
+ if (match) {
657
+ dataLines.push({ index, space: match[1] ?? "", value: match[2] ?? "" })
658
+ }
659
+ })
660
+ if (dataLines.length === 0) {
661
+ return block
662
+ }
663
+ const data = dataLines.map((entry) => entry.value).join("\n")
664
+ const payloadText = data.trim()
665
+ if (!payloadText || payloadText === "[DONE]") {
666
+ return block
667
+ }
668
+ let payload: unknown
669
+ try {
670
+ payload = JSON.parse(data)
671
+ } catch {
672
+ return block
673
+ }
674
+ if (!scrubber.scrub(payload)) {
675
+ return block
676
+ }
677
+ const first = dataLines[0]!
678
+ const drop = new Set(dataLines.slice(1).map((entry) => entry.index))
679
+ const rebuilt: string[] = []
680
+ lines.forEach((raw, index) => {
681
+ if (drop.has(index)) {
682
+ return
683
+ }
684
+ if (index === first.index) {
685
+ rebuilt.push(`data:${first.space}${JSON.stringify(payload)}`)
686
+ return
687
+ }
688
+ rebuilt.push(raw)
689
+ })
690
+ return rebuilt.join("\n")
691
+ } catch {
692
+ return block
693
+ }
694
+ }
695
+
696
+ return new TransformStreamImpl({
697
+ transform(chunk: Uint8Array, controller: any) {
698
+ let enqueued = false
699
+ try {
700
+ buffer += decoder.decode(chunk, { stream: true })
701
+ let boundary = findBoundary(buffer)
702
+ while (boundary) {
703
+ const block = buffer.slice(0, boundary.start)
704
+ buffer = buffer.slice(boundary.end)
705
+ controller.enqueue(encoder.encode(processBlock(block) + boundary.separator))
706
+ enqueued = true
707
+ boundary = findBoundary(buffer)
708
+ }
709
+ } catch {
710
+ if (!enqueued) {
711
+ try {
712
+ controller.enqueue(chunk)
713
+ } catch {
714
+ // The stream is already closed; nothing to do.
715
+ }
716
+ }
717
+ }
718
+ },
719
+ flush(controller: any) {
720
+ try {
721
+ buffer += decoder.decode()
722
+ if (buffer) {
723
+ controller.enqueue(encoder.encode(processBlock(buffer)))
724
+ }
725
+ buffer = ""
726
+ } catch {
727
+ // Never fail the response because of scrubbing.
728
+ }
729
+ },
730
+ })
731
+ }
732
+
733
+ /** Buffers a JSON body, scrubs it at stream end. Used for non-SSE responses. */
734
+ export function createJsonScrubTransform(config: OutputScrubConfig): any {
735
+ const TransformStreamImpl = (globalThis as any).TransformStream
736
+ const decoder = new TextDecoder()
737
+ const encoder = new TextEncoder()
738
+ const scrubber = new ProviderPayloadScrubber(config)
739
+ let buffer = ""
740
+
741
+ return new TransformStreamImpl({
742
+ transform(chunk: Uint8Array, _controller: any) {
743
+ try {
744
+ buffer += decoder.decode(chunk, { stream: true })
745
+ } catch {
746
+ // Ignore malformed byte sequences; the flush pass passes through.
747
+ }
748
+ },
749
+ flush(controller: any) {
750
+ try {
751
+ buffer += decoder.decode()
752
+ controller.enqueue(encoder.encode(scrubJsonBody(buffer, config, scrubber)))
753
+ buffer = ""
754
+ } catch {
755
+ try {
756
+ controller.enqueue(encoder.encode(buffer))
757
+ buffer = ""
758
+ } catch {
759
+ // The stream is already closed; nothing to do.
760
+ }
761
+ }
762
+ },
763
+ })
764
+ }
765
+
766
+ /**
767
+ * Wrap a provider response so its body is scrubbed while streaming. Returns the
768
+ * original response when nothing can or should be done.
769
+ */
770
+ export function wrapScrubResponse(response: any, config: OutputScrubConfig): any {
771
+ try {
772
+ if (!response || typeof response !== "object" || !response.body) {
773
+ return response
774
+ }
775
+ const headers = new Headers(response.headers)
776
+ const contentType = headers.get("content-type") ?? ""
777
+ let body: any
778
+ if (/text\/event-stream/i.test(contentType)) {
779
+ body = response.body.pipeThrough(createSseScrubTransform(config))
780
+ } else if (/json/i.test(contentType)) {
781
+ body = response.body.pipeThrough(createJsonScrubTransform(config))
782
+ } else {
783
+ return response
784
+ }
785
+ headers.delete("content-length")
786
+ return new Response(body, {
787
+ status: response.status,
788
+ statusText: response.statusText,
789
+ headers,
790
+ })
791
+ } catch {
792
+ return response
793
+ }
794
+ }
795
+
796
+ /**
797
+ * Best-effort registration of the response scrub hook. Applies to every session
798
+ * request kind (primary, compaction, title, generate) through the shared HTTP
799
+ * transport used by native provider packages. AI SDK providers are covered by
800
+ * the language-model wrapper instead.
801
+ */
802
+ export async function installResponseScrubHook(
803
+ ctx: Plugin.Context,
804
+ config: OutputScrubConfig,
805
+ logger: Logger,
806
+ ): Promise<void> {
807
+ if (!outputScrubEnabled(config)) {
808
+ return
809
+ }
810
+ const session = (ctx as any)?.session
811
+ if (!session || typeof session.hook !== "function") {
812
+ return
813
+ }
814
+ try {
815
+ await session.hook("http.response", async (event: any) => {
816
+ try {
817
+ const response = event?.response
818
+ if (!response || typeof response !== "object") {
819
+ return
820
+ }
821
+ const wrapped = wrapScrubResponse(response, config)
822
+ if (wrapped !== response) {
823
+ event.response = wrapped
824
+ }
825
+ } catch (error: any) {
826
+ logger.debug("DCP output scrub failed", { error: error?.message })
827
+ }
828
+ })
829
+ logger.debug("DCP output scrub hook installed")
830
+ } catch (error: any) {
831
+ logger.warn("DCP output scrub hook registration failed", { error: error?.message })
832
+ }
833
+ }
834
+
835
+ function isTextDelta(part: any): { family: string; field: string } | undefined {
836
+ if (part?.type === "text-delta") {
837
+ return { family: "text-delta", field: deltaField(part) }
838
+ }
839
+ if (part?.type === "reasoning-delta" || part?.type === "reasoning") {
840
+ return { family: "reasoning-delta", field: deltaField(part) }
841
+ }
842
+ return undefined
843
+ }
844
+
845
+ function deltaField(part: any): string {
846
+ if (typeof part.text === "string") {
847
+ return "text"
848
+ }
849
+ if (typeof part.delta === "string") {
850
+ return "delta"
851
+ }
852
+ return "textDelta"
853
+ }
854
+
855
+ /**
856
+ * Standardized AI SDK stream-part scrubber. Deltas pass through a stateful
857
+ * scrubber, end/finish parts are scrubbed statelessly, and any text the
858
+ * scrubber held back is re-emitted as an extra delta before the end part.
859
+ */
860
+ export class AisdkPartScrubber {
861
+ private readonly streams = new Map<string, DcpStreamScrubber>()
862
+
863
+ constructor(private readonly config: OutputScrubConfig) {}
864
+
865
+ process(part: any): any[] {
866
+ try {
867
+ if (!part || typeof part !== "object") {
868
+ return [part]
869
+ }
870
+ const delta = isTextDelta(part)
871
+ if (delta) {
872
+ const key = `${delta.family}:${part.id ?? ""}`
873
+ return [{ ...part, [delta.field]: this.stream(key).push(part[delta.field]) }]
874
+ }
875
+ if (part.type === "text-start" || part.type === "reasoning-start") {
876
+ const family = part.type === "text-start" ? "text-delta" : "reasoning-delta"
877
+ this.streams.delete(`${family}:${part.id ?? ""}`)
878
+ return [part]
879
+ }
880
+ if (part.type === "text-end" || part.type === "reasoning-end") {
881
+ const family = part.type === "text-end" ? "text-delta" : "reasoning-delta"
882
+ const out: any[] = []
883
+ const leftover = this.take(`${family}:${part.id ?? ""}`)
884
+ if (leftover) {
885
+ out.push({ type: family, id: part.id, text: leftover })
886
+ }
887
+ const field = deltaField(part)
888
+ if (typeof part[field] === "string") {
889
+ out.push({ ...part, [field]: stripDcpArtifacts(part[field], this.config) })
890
+ } else {
891
+ out.push(part)
892
+ }
893
+ return out
894
+ }
895
+ if (part.type === "finish") {
896
+ return [...this.flushStreams(), part]
897
+ }
898
+ return [part]
899
+ } catch {
900
+ return [part]
901
+ }
902
+ }
903
+
904
+ private stream(key: string): DcpStreamScrubber {
905
+ let scrubber = this.streams.get(key)
906
+ if (!scrubber) {
907
+ scrubber = new DcpStreamScrubber(this.config)
908
+ this.streams.set(key, scrubber)
909
+ }
910
+ return scrubber
911
+ }
912
+
913
+ private take(key: string): string {
914
+ const scrubber = this.streams.get(key)
915
+ if (!scrubber) {
916
+ return ""
917
+ }
918
+ this.streams.delete(key)
919
+ return scrubber.flush()
920
+ }
921
+
922
+ private flushStreams(): any[] {
923
+ const out: any[] = []
924
+ for (const [key, scrubber] of this.streams) {
925
+ const leftover = scrubber.flush()
926
+ if (!leftover) {
927
+ continue
928
+ }
929
+ const separator = key.indexOf(":")
930
+ out.push({
931
+ type: separator === -1 ? "text-delta" : key.slice(0, separator),
932
+ id: separator === -1 ? "" : key.slice(separator + 1),
933
+ text: leftover,
934
+ })
935
+ }
936
+ this.streams.clear()
937
+ return out
938
+ }
939
+ }
940
+
941
+ /** Stateless scrub of a non-streamed AI SDK result (`doGenerate`). */
942
+ export function stripAisdkResultContent(result: any, config: OutputScrubConfig): void {
943
+ if (!result || !Array.isArray(result.content)) {
944
+ return
945
+ }
946
+ for (const part of result.content) {
947
+ if (!part || typeof part !== "object") {
948
+ continue
949
+ }
950
+ if ((part.type === "text" || part.type === "reasoning") && typeof part.text === "string") {
951
+ part.text = stripDcpArtifacts(part.text, config)
952
+ }
953
+ }
954
+ }