ds4-context-engine 0.4.6 → 0.4.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/docs/ADR/068-downgrade-rendering-equivalent-spans.md +75 -0
- package/docs/ADR/README.md +1 -0
- package/docs/COMPACTION.md +2 -2
- package/docs/releases/0.4.6.md +21 -6
- package/docs/releases/0.4.7.md +135 -0
- package/package.json +2 -2
- package/src/pi-adapter/summary-generator.ts +27 -8
- package/src/pi-adapter/version.ts +1 -1
package/README.md
CHANGED
|
@@ -16,7 +16,7 @@ bounded active context with provenance
|
|
|
16
16
|
Pi provider
|
|
17
17
|
```
|
|
18
18
|
|
|
19
|
-
> **Project status:** The coordinated `0.4.0` release adds `storage.scope: "agent" | "project"`, now defaulting to per-project SQLite projections with shared token calibration in the agent database (opt out with `storage.scope: "agent"`). The opt-in BPE estimation and bounded auto-tuning from `0.3.10` keep `chars-v1` and disabled auto-tuning as their defaults. The bounded compaction controls from `0.3.9` remain in place; canonical history, SQLite schema 16 and runtime contracts are unchanged. Pi remains pinned to `0.84.3`. Patch `0.4.1` states the contiguous-span rule for compaction summaries in the summarizer prompt while leaving validation strictness, the eight-bullet repair bound and every provider-facing default unchanged. Patch `0.4.2` adds class-only diagnostics for rejected exact-value spans: the existing `custom_fallback` warning carries `unsupportedSpanClasses` as class names and counters, never span text, with validation, repair bounds and provider-facing defaults unchanged. Patch `0.4.3` makes those diagnostics interpretable: every distinct span receives the cheap transformed-form lookups before the bounded near-miss analysis starts, each near-miss candidate may spend only an equal share of what remains, and spans that were not analysed report `not-classified-length`, `not-classified-partial` or `not-classified-budget` instead of `no-near-miss`. Patch `0.4.4` accepts a backticked exact value whose canonical JSON-escaped rendering is literally present in the evidence — the decoded rendering a summarizer produces when it quotes serialized JSON — while absent values, the reverse escape direction, span composition and single-character variants stay rejected; the diagnostics probe budget rises from 4000 to 24000 evidence-source scans. Patch `0.4.5` grades the summarizer quoting fallback: the prompt requires one contiguous excerpt per backticked span, tells the model to backtick only the fragments that are themselves contiguous or to keep the fact as ordinary text without backticks instead of composing a span, and reserves bullet omission for facts with no support in the evidence; diagnostics split composition into `composed-adjacent-present` (both parts next to each other in one source, so the span may be a re-rendering of a contiguous region) and `composed-two-present-parts` (parts found at unrelated positions), with validation strictness, the eight-bullet and 25% repair bounds and every provider-facing default unchanged. Patch `0.4.6` makes an engine/core artifact mismatch visible instead of a missing function: the core exports `CORE_VERSION` and the engine verifies at session start and at every compaction attempt that the loaded core matches its own version and exposes the entry points it calls, so a core rebuilt under a running Pi — which a `/reload` does not pick up — reports one actionable line, keeps the DS4 compaction layer inert instead of failing mid-generation, and never blocks Pi; validation strictness, repair bounds and every provider-facing default stay unchanged. See the [0.4.6 release record](docs/releases/0.4.6.md), the [0.4.5 release record](docs/releases/0.4.5.md), the [0.4.4 release record](docs/releases/0.4.4.md), the [0.4.3 release record](docs/releases/0.4.3.md), the [0.4.2 release record](docs/releases/0.4.2.md), the [0.4.1 release record](docs/releases/0.4.1.md) and the [0.4.0 release record](docs/releases/0.4.0.md), [ADR 067](docs/ADR/067-core-version-guard.md), [ADR 066](docs/ADR/066-quoting-downgrade-and-span-adjacency.md), [ADR 065](docs/ADR/065-exact-value-escaped-equivalence.md), [ADR 064](docs/ADR/064-per-project-databases-with-shared-calibration.md) and [model-awareness validation](docs/MODEL_AWARENESS.md).
|
|
19
|
+
> **Project status:** The coordinated `0.4.0` release adds `storage.scope: "agent" | "project"`, now defaulting to per-project SQLite projections with shared token calibration in the agent database (opt out with `storage.scope: "agent"`). The opt-in BPE estimation and bounded auto-tuning from `0.3.10` keep `chars-v1` and disabled auto-tuning as their defaults. The bounded compaction controls from `0.3.9` remain in place; canonical history, SQLite schema 16 and runtime contracts are unchanged. Pi remains pinned to `0.84.3`. Patch `0.4.1` states the contiguous-span rule for compaction summaries in the summarizer prompt while leaving validation strictness, the eight-bullet repair bound and every provider-facing default unchanged. Patch `0.4.2` adds class-only diagnostics for rejected exact-value spans: the existing `custom_fallback` warning carries `unsupportedSpanClasses` as class names and counters, never span text, with validation, repair bounds and provider-facing defaults unchanged. Patch `0.4.3` makes those diagnostics interpretable: every distinct span receives the cheap transformed-form lookups before the bounded near-miss analysis starts, each near-miss candidate may spend only an equal share of what remains, and spans that were not analysed report `not-classified-length`, `not-classified-partial` or `not-classified-budget` instead of `no-near-miss`. Patch `0.4.4` accepts a backticked exact value whose canonical JSON-escaped rendering is literally present in the evidence — the decoded rendering a summarizer produces when it quotes serialized JSON — while absent values, the reverse escape direction, span composition and single-character variants stay rejected; the diagnostics probe budget rises from 4000 to 24000 evidence-source scans. Patch `0.4.5` grades the summarizer quoting fallback: the prompt requires one contiguous excerpt per backticked span, tells the model to backtick only the fragments that are themselves contiguous or to keep the fact as ordinary text without backticks instead of composing a span, and reserves bullet omission for facts with no support in the evidence; diagnostics split composition into `composed-adjacent-present` (both parts next to each other in one source, so the span may be a re-rendering of a contiguous region) and `composed-two-present-parts` (parts found at unrelated positions), with validation strictness, the eight-bullet and 25% repair bounds and every provider-facing default unchanged. Patch `0.4.6` makes an engine/core artifact mismatch visible instead of a missing function: the core exports `CORE_VERSION` and the engine verifies at session start and at every compaction attempt that the loaded core matches its own version and exposes the entry points it calls, so a core rebuilt under a running Pi — which a `/reload` does not pick up — reports one actionable line, keeps the DS4 compaction layer inert instead of failing mid-generation, and never blocks Pi; validation strictness, repair bounds and every provider-facing default stay unchanged. Patch `0.4.7` stops the recurrent fail-closed fallback at its source: when the summarizer re-renders a value the evidence holds in another form — JSON-escaped or unescaped, collapsed whitespace, stripped typographic characters, or two parts adjacent in one source — the repair retracts only the backticks and keeps the fact, so those spans no longer consume the eight-bullet bound; invented values, unrelated compositions, one-character deviations, case changes and unanalysed spans keep the existing fail-closed bullet removal, validation strictness is unchanged, and both effects are recorded as counters (`unsupported-exact-bullets-pruned`, `unsupported-exact-spans-unquoted`). See the [0.4.7 release record](docs/releases/0.4.7.md), the [0.4.6 release record](docs/releases/0.4.6.md), the [0.4.5 release record](docs/releases/0.4.5.md), the [0.4.4 release record](docs/releases/0.4.4.md), the [0.4.3 release record](docs/releases/0.4.3.md), the [0.4.2 release record](docs/releases/0.4.2.md), the [0.4.1 release record](docs/releases/0.4.1.md) and the [0.4.0 release record](docs/releases/0.4.0.md), [ADR 068](docs/ADR/068-downgrade-rendering-equivalent-spans.md), [ADR 067](docs/ADR/067-core-version-guard.md), [ADR 066](docs/ADR/066-quoting-downgrade-and-span-adjacency.md), [ADR 065](docs/ADR/065-exact-value-escaped-equivalence.md), [ADR 064](docs/ADR/064-per-project-databases-with-shared-calibration.md) and [model-awareness validation](docs/MODEL_AWARENESS.md).
|
|
20
20
|
|
|
21
21
|
**Current compaction defaults:** `compaction.directUpdate=true`, `compaction.inputBudget="context"`, `compaction.segmentTargetTokens=30000`, `compaction.maxRequestInputTokens=64000`, `compaction.maxOperationInputTokens=2000000`, `compaction.maxConcurrentSegments=2`. Every DS4 provider attempt is bounded by the effective request limit, and the operation limit includes retries; `inputBudget="summary"` remains an explicit throughput-oriented opt-in. Existing compaction/master switches still apply. See [latency controls and compatibility](docs/COMPACTION.md#latency-controls). No real-provider speedup is claimed from mock tests. The five optional editing/reading/artifact/job features introduced in `0.3.4` remain default-off.
|
|
22
22
|
|
|
@@ -515,6 +515,7 @@ scripts package and release-readiness checks
|
|
|
515
515
|
- [Roadmap 0.2.0](docs/ROADMAP_0.2.0.md)
|
|
516
516
|
- [Release process](docs/RELEASING.md)
|
|
517
517
|
- [0.2.0 release readiness](docs/RELEASE_READINESS_0.2.0.md)
|
|
518
|
+
- [0.4.7 release notes](docs/releases/0.4.7.md)
|
|
518
519
|
- [0.4.6 release notes](docs/releases/0.4.6.md)
|
|
519
520
|
- [0.4.5 release notes](docs/releases/0.4.5.md)
|
|
520
521
|
- [0.4.4 release notes](docs/releases/0.4.4.md)
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# 068 — Downgrade rendering-equivalent spans instead of spending the bullet budget
|
|
2
|
+
|
|
3
|
+
**Date:** 2026-09-27
|
|
4
|
+
**Status:** Accepted
|
|
5
|
+
**Related:** [065](065-exact-value-escaped-equivalence.md), [066](066-quoting-downgrade-and-span-adjacency.md), [067](067-core-version-guard.md)
|
|
6
|
+
|
|
7
|
+
## Context
|
|
8
|
+
|
|
9
|
+
Five coordinated releases reduced, but never removed, the recurrent
|
|
10
|
+
`compaction.custom_fallback` with `unsupported-exact-value;
|
|
11
|
+
repair=too-many-bullets`. The affected-bullet series from production was 30/20 →
|
|
12
|
+
23/19 → 19/15 → 17/11 → 16/12, and the class-only reports attributed the majority
|
|
13
|
+
of the rejected spans to rendering or composition, never to invention:
|
|
14
|
+
`escaped-form-present` 11 of 16 in the first interpretable run, then composition
|
|
15
|
+
9 of 14 with 2 one-character deviations, 2 absences and 1 over-length span in the
|
|
16
|
+
next.
|
|
17
|
+
|
|
18
|
+
Each patch removed one class — escaped equivalence in 0.4.4, prompt grading in
|
|
19
|
+
0.4.5 — while the bound stayed exhausted by the remaining ones. The repair was
|
|
20
|
+
treating a *re-rendered* value exactly like an *invented* one: it deleted whole
|
|
21
|
+
bullets, and beyond eight bullets it refused to repair at all and handed the
|
|
22
|
+
session to Pi. The information in those spans is present in the evidence; only
|
|
23
|
+
the verbatim surface form is not. Discarding a supported fact, or failing, is the
|
|
24
|
+
wrong response to a rendering difference.
|
|
25
|
+
|
|
26
|
+
## Decision
|
|
27
|
+
|
|
28
|
+
- Validation is unchanged. A backticked value is still accepted only when it
|
|
29
|
+
occurs literally in the evidence or as its canonical JSON-escaped rendering.
|
|
30
|
+
Nothing that was rejected before becomes an accepted exact value.
|
|
31
|
+
- The bounded repair splits the rejected spans into two groups.
|
|
32
|
+
- **Rendering-equivalent** — `escaped-form-present`, `unescaped-form-present`,
|
|
33
|
+
`whitespace-collapsed-present`, `typographic-variant-present`,
|
|
34
|
+
`composed-adjacent-present`: DS4 removes only the two backticks. The text
|
|
35
|
+
stays, the fact stays, the exact-value claim is retracted, and the result is
|
|
36
|
+
re-validated. This applies deterministically the ladder the 0.4.5 prompt asks
|
|
37
|
+
the model to follow, instead of depending on the model complying.
|
|
38
|
+
- **Everything else** — `single-deletion-present`,
|
|
39
|
+
`composed-two-present-parts`, `case-variant-present`, `no-near-miss` and
|
|
40
|
+
every `not-classified-*`: the existing fail-closed treatment, whole-bullet
|
|
41
|
+
removal within the eight-bullet and 25% bounds, otherwise fallback to Pi.
|
|
42
|
+
- The excluded relations are excluded because prose would hide a wrong claim
|
|
43
|
+
rather than a re-rendered one: a case change can alter an identifier, a
|
|
44
|
+
one-character deviation is indistinguishable from a wrong value, and
|
|
45
|
+
`composed-two-present-parts` is precisely the association the evidence does not
|
|
46
|
+
contain ([ADR 066](066-quoting-downgrade-and-span-adjacency.md)).
|
|
47
|
+
`not-classified-*` means *not analysed*, which is not evidence of support, so
|
|
48
|
+
those spans stay fail-closed as well.
|
|
49
|
+
- The eight-bullet and 25% bounds are not raised. They now count only the bullets
|
|
50
|
+
that still need removal, because a downgraded span costs no bullet.
|
|
51
|
+
- Both effects are recorded as counters, never as span text:
|
|
52
|
+
`unsupported-exact-bullets-pruned` for removed bullets and
|
|
53
|
+
`unsupported-exact-spans-unquoted` for downgraded spans.
|
|
54
|
+
|
|
55
|
+
## Consequences
|
|
56
|
+
|
|
57
|
+
- The dominant production failure mode no longer consumes the bullet bound. In
|
|
58
|
+
the core fixtures, twelve affected bullets now downgrade to plain text and
|
|
59
|
+
validate, where the same input previously answered `too-many-bullets`; the
|
|
60
|
+
integration fixture with eleven affected bullets completes with
|
|
61
|
+
`validationStatus: "warning"` instead of Pi compacting the session.
|
|
62
|
+
- Fail-closed survives where it matters. An invented value keeps the
|
|
63
|
+
bullet-removal path, and the same eleven-bullet fixture still answers
|
|
64
|
+
`repair=too-many-bullets` when the evidence holds the two parts at unrelated
|
|
65
|
+
positions rather than next to each other.
|
|
66
|
+
- The repair now consumes the class analysis, which is bounded and deterministic
|
|
67
|
+
but budget-limited (default 24000 evidence-source scans). A span left
|
|
68
|
+
unanalysed is never downgraded.
|
|
69
|
+
- A downgraded value is still *stated* as prose: DS4 stops asserting exactness it
|
|
70
|
+
cannot verify, but it does not delete the fact. That is the deliberate trade —
|
|
71
|
+
a fact with a possibly imperfect surface form beats losing the fact or losing
|
|
72
|
+
the whole compaction — and it is visible in the recorded counters.
|
|
73
|
+
- The `UnsupportedSpanClassReport` documentation no longer claims the relations
|
|
74
|
+
are observation-only: they now select the repair treatment. Validation still
|
|
75
|
+
never consults them.
|
package/docs/ADR/README.md
CHANGED
|
@@ -71,5 +71,6 @@ The initial decisions from the development plan are accepted:
|
|
|
71
71
|
| [065](065-exact-value-escaped-equivalence.md) | Accept canonical JSON-escaped exact values in compaction summary validation | Accepted |
|
|
72
72
|
| [066](066-quoting-downgrade-and-span-adjacency.md) | Grade the summarizer quoting fallback and separate adjacent from unrelated span composition | Accepted |
|
|
73
73
|
| [067](067-core-version-guard.md) | Detect an engine/core artifact mismatch before compaction runs | Accepted |
|
|
74
|
+
| [068](068-downgrade-rendering-equivalent-spans.md) | Downgrade rendering-equivalent spans instead of spending the bullet budget | Accepted |
|
|
74
75
|
|
|
75
76
|
Each decision will receive a dedicated record when implementation pressure introduces alternatives or consequences not already covered by the development plan.
|
package/docs/COMPACTION.md
CHANGED
|
@@ -32,11 +32,11 @@ Fan-out and fan-in are bounded to 32 segment requests, 64 aggregate requests, an
|
|
|
32
32
|
## Critical Exact Values
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
-
Every section must occur once, in order, and contain content or `- None`. DS4 replaces each unique `Files Read` and `Files Modified` section with one exact path per bullet from Pi's sanitized file-operation inventory before validation; missing or duplicate sections still fail. Backticked exact values must occur literally in the serialized segment source, ordered child-summary content, or those known file-operation paths, either as written or in the canonical JSON-escaped rendering of the value. A decoded value whose encoded text is present in the evidence is accepted ([ADR 065](ADR/065-exact-value-escaped-equivalence.md)); the reverse direction — an escaped span whose raw form is present — remains unsupported and is reported as `unescaped-form-present`. A bounded unsupported exact-value bullet is removed as a whole and recorded as a validation warning
|
|
35
|
+
Every section must occur once, in order, and contain content or `- None`. DS4 replaces each unique `Files Read` and `Files Modified` section with one exact path per bullet from Pi's sanitized file-operation inventory before validation; missing or duplicate sections still fail. Backticked exact values must occur literally in the serialized segment source, ordered child-summary content, or those known file-operation paths, either as written or in the canonical JSON-escaped rendering of the value. A decoded value whose encoded text is present in the evidence is accepted ([ADR 065](ADR/065-exact-value-escaped-equivalence.md)); the reverse direction — an escaped span whose raw form is present — remains unsupported and is reported as `unescaped-form-present`. A bounded unsupported exact-value bullet is removed as a whole and recorded as a validation warning, unless every rejected span in it is *rendering-equivalent*: when the evidence holds the same information in a different rendering — JSON-escaped or unescaped, collapsed whitespace, stripped typographic characters, or two parts adjacent in one source — the repair retracts only the two backticks and keeps the text, so the fact survives and the exact-value claim does not ([ADR 068](ADR/068-downgrade-rendering-equivalent-spans.md)). Unsupported prose, more than eight bullets that still need removal, or removal above 25% still fails closed to Pi. The prompt requires every backticked span to be one contiguous excerpt copied as-is and forbids assembling one span from separately supported values (a setting name with its value, a path with a line range, a command with its flags), and it gives the model a cheaper alternative than composing: backtick only the fragments that are themselves contiguous with the joining text outside the backticks, or write the value as ordinary text without backticks, keeping the bullet. A bullet is omitted only when the fact itself has no support in the evidence ([ADR 066](ADR/066-quoting-downgrade-and-span-adjacency.md)). Every segment is validated independently against only its own sanitized source and deterministic file inventory. Every aggregate is validated independently against only the sanitized content of its ordered children and cumulative deterministic file inventory. A direct update is validated against both the sanitized previous summary and new source, plus cumulative sanitized file evidence; previous-summary-only exact values remain valid evidence. Custom focus instructions are not factual evidence. Any unrepaired failure prevents the whole graph batch from being installed.
|
|
36
36
|
|
|
37
37
|
An unrepaired exact-value failure reports only the stage, issue code, categorical repair status, unsupported-span count, and affected-bullet count. Repair statuses distinguish an unsupported location, more than eight bullets, removal above 25%, and an unexpected invalid second validation. The disputed text is intentionally absent from logs, UI notifications, and diagnostics because it may contain sensitive source material.
|
|
38
38
|
|
|
39
|
-
The same failure carries a class-only span report on the fallback warning (`unsupportedSpanClasses`): rejected-span count, affected bullets, length buckets, character shapes (spaces, backslashes, escape sequences, separators, quotes, typographic characters, JSON punctuation), and how each distinct span relates to the evidence — escaped or unescaped rendering, collapsed whitespace, case or typographic variant, one-character deletion, two separately present values joined into one span — adjacent in one source with nothing but joining punctuation and spaces between them (`composed-adjacent-present`) or found at unrelated positions (`composed-two-present-parts`) — or no near-miss (every applicable lookup ran and found nothing). The adjacency question separates a join that re-renders a contiguous evidence region from an association the evidence never contains: a joiner gap holds no letters, digits or line breaks, so two parts on separate lines are not adjacent. The report contains class names and counters only, never span text, and the classifier
|
|
39
|
+
The same failure carries a class-only span report on the fallback warning (`unsupportedSpanClasses`): rejected-span count, affected bullets, length buckets, character shapes (spaces, backslashes, escape sequences, separators, quotes, typographic characters, JSON punctuation), and how each distinct span relates to the evidence — escaped or unescaped rendering, collapsed whitespace, case or typographic variant, one-character deletion, two separately present values joined into one span — adjacent in one source with nothing but joining punctuation and spaces between them (`composed-adjacent-present`) or found at unrelated positions (`composed-two-present-parts`) — or no near-miss (every applicable lookup ran and found nothing). The adjacency question separates a join that re-renders a contiguous evidence region from an association the evidence never contains: a joiner gap holds no letters, digits or line breaks, so two parts on separate lines are not adjacent. The report contains class names and counters only, never span text, and validation never consults the classifier; the repair uses the same relations to decide which spans are downgraded instead of spending a bullet on them, which is why the eight-bullet bound counts only removals. It exists to separate a summarizer that invents values from one whose rendering or composition rules differ from the validation domain, which require opposite fixes. The first production report produced by the 0.4.4 diagnostics attributed 9 of 14 rejected spans to composition, 2 to a one-character deviation, 2 to absence and 1 to the length limit, with no rendering class left at all.
|
|
40
40
|
|
|
41
41
|
Three relations mark spans that were *not* analysed, and they must never be read as invention. `not-classified-length` means the span exceeds the near-miss length limit (96 characters); `not-classified-partial` means its share of the budget ran out; `not-classified-budget` means the shared budget was already exhausted. To keep the histogram interpretable, transformed-form lookups cover every distinct span before any near-miss analysis starts, and each near-miss candidate may spend only an equal share of what remains (at least 32 lookups, never more than the budget left). The report also carries `spansClassifiedCheap` (span occurrences attributed by a transformed-form relation), `corpusSources`, `probeBudget` and `probesUsed`, so a run can be read without guessing how much of the budget was consumed. The shared budget defaults to 24000 evidence-source scans, where one lookup scans every corpus source once. `escaped-form-present` is not expected in reports from 0.4.4 on: validation accepts the canonical escaped rendering and the classifier consumes the same predicate, so such spans no longer reach it; the class stays in the taxonomy for compatibility. `classificationComplete` is false when the shared budget or a per-span share stopped an analysis; a `not-classified-length` span does not clear it, because that limit is deterministic rather than a resource shortfall.
|
|
42
42
|
|
package/docs/releases/0.4.6.md
CHANGED
|
@@ -129,9 +129,24 @@ repair bounds or the fail-closed decision changes
|
|
|
129
129
|
|
|
130
130
|
## Publication and registry verification
|
|
131
131
|
|
|
132
|
-
Published manually in dependency order
|
|
133
|
-
`ds4-context-reference-adapter@0.4.6`, then `ds4-context-engine@0.4.6`, all
|
|
134
|
-
the default `latest` tag, after `npm run pack:check` on the same commit
|
|
135
|
-
|
|
136
|
-
`
|
|
137
|
-
|
|
132
|
+
Published manually in dependency order from `92912a1`: `ds4-context-core@0.4.6`,
|
|
133
|
+
then `ds4-context-reference-adapter@0.4.6`, then `ds4-context-engine@0.4.6`, all
|
|
134
|
+
with the default `latest` tag, after `npm run pack:check` on the same commit.
|
|
135
|
+
|
|
136
|
+
The first two `npm run registry:check -- 0.4.6` attempts, each with a fresh npm
|
|
137
|
+
cache, failed with `ETARGET No matching version found for
|
|
138
|
+
ds4-context-core@0.4.6` — the known registry propagation delay after a publish,
|
|
139
|
+
not a packaging fault. Once both `npm view ds4-context-core@0.4.6 version` and
|
|
140
|
+
`npm view ds4-context-engine@0.4.6 version` resolved, the check reported
|
|
141
|
+
"Verified all DS4 registry packages at exact version 0.4.6": all three packages
|
|
142
|
+
install in a clean consumer, the exact adapter/core dependencies resolve, the
|
|
143
|
+
public core and KV exports import, the compiled reference conformance and
|
|
144
|
+
packaged quality corpus run, the `ds4-context-storage` CLI shim responds, and the
|
|
145
|
+
published Pi extension starts against isolated offline RPC state.
|
|
146
|
+
`npm view <package> dist-tags.latest` resolves to 0.4.6 for all three packages.
|
|
147
|
+
|
|
148
|
+
Annotated tag `v0.4.6` and the GitHub Release were created from `92912a1`:
|
|
149
|
+
<https://github.com/Alucard24/ds4-context-engine/releases/tag/v0.4.6>. No session
|
|
150
|
+
data, credentials, provider payloads or prompt text were read or written while
|
|
151
|
+
preparing this release, and no provider call is involved in the release
|
|
152
|
+
procedure.
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
# Release 0.4.7 — Downgrade rendering-equivalent spans instead of spending the bullet budget
|
|
2
|
+
|
|
3
|
+
**Coordinated packages:** `ds4-context-core`, `ds4-context-reference-adapter`, and `ds4-context-engine` 0.4.7.
|
|
4
|
+
**Implementation commit:** `1483bce`.
|
|
5
|
+
|
|
6
|
+
## Summary
|
|
7
|
+
|
|
8
|
+
Behaviour patch that removes the source of the recurrent fail-closed fallback.
|
|
9
|
+
Until 0.4.6 the repair treated a re-rendered exact value exactly like an invented
|
|
10
|
+
one: it deleted whole bullets, and beyond eight affected bullets it refused to
|
|
11
|
+
repair at all and handed the session to Pi. The class reports had shown for four
|
|
12
|
+
releases that this was the common case — rendering and composition, not
|
|
13
|
+
invention. The repair now retracts only the backticks of a span whose information
|
|
14
|
+
the evidence holds in another rendering, keeps the text and the fact, and
|
|
15
|
+
re-validates. Invented values, unrelated compositions, one-character deviations,
|
|
16
|
+
case changes and unanalysed spans keep the existing fail-closed bullet removal.
|
|
17
|
+
Validation strictness, the eight-bullet and 25% bounds, the fail-closed fallback,
|
|
18
|
+
SQLite schema 16, canonical JSONL history and every provider-facing default are
|
|
19
|
+
unchanged.
|
|
20
|
+
|
|
21
|
+
## Changes
|
|
22
|
+
|
|
23
|
+
- `packages/core/src/compaction/summary-contract.ts`:
|
|
24
|
+
- `RENDERING_EQUIVALENT_RELATIONS` names the relations that mean "the evidence
|
|
25
|
+
holds this information in a different rendering": `escaped-form-present`,
|
|
26
|
+
`unescaped-form-present`, `whitespace-collapsed-present`,
|
|
27
|
+
`typographic-variant-present` and `composed-adjacent-present`.
|
|
28
|
+
- Deliberately excluded, because prose would hide a wrong claim rather than a
|
|
29
|
+
re-rendered one: `case-variant-present` (a case change can alter an
|
|
30
|
+
identifier), `single-deletion-present` (a one-character deviation is
|
|
31
|
+
indistinguishable from a wrong value), `composed-two-present-parts` (the
|
|
32
|
+
evidence never contains that association) and every `not-classified-*`
|
|
33
|
+
relation (*not analysed* is not evidence of support).
|
|
34
|
+
- `downgradeRenderingEquivalentSpans` removes the two backticks of each
|
|
35
|
+
rendering-equivalent span, byte-range based and applied after the bullet
|
|
36
|
+
removals so every span is matched at its final position.
|
|
37
|
+
- `analyzeUnsupportedExactValueBullets` partitions the rejected spans, applies
|
|
38
|
+
the unchanged bounds to the ones that still need removal, and reports the
|
|
39
|
+
new counters. `ExactValuePruneResult` and `ExactValuePruneAttempt` gained
|
|
40
|
+
`downgradedSpans`, and the attempt status gained `downgraded` and
|
|
41
|
+
`unrepairable`.
|
|
42
|
+
- The span classifier is no longer described as observation-only: it is
|
|
43
|
+
`analyzeSpanClasses` internally, and its relations now also select the repair
|
|
44
|
+
treatment. Validation still never consults them, and the public
|
|
45
|
+
`classifyUnsupportedExactValueSpans` report is unchanged.
|
|
46
|
+
- `src/pi-adapter/summary-generator.ts`: the repair outcome is recorded as counts
|
|
47
|
+
only — `unsupported-exact-bullets-pruned` for removed bullets and the new
|
|
48
|
+
`unsupported-exact-spans-unquoted` for retracted quotes.
|
|
49
|
+
- `tests/unit/summary-contract.test.ts`: five cases — a composed span with parts
|
|
50
|
+
adjacent in one source is downgraded and the summary still validates; a span
|
|
51
|
+
whose whitespace-collapsed and JSON-unescaped renderings are present is
|
|
52
|
+
downgraded too; unrelated compositions and one-character deviations still cost
|
|
53
|
+
their bullet; twelve affected bullets downgrade where the same summary without
|
|
54
|
+
the adjacent rendering still answers `too-many-bullets`; a mixed summary
|
|
55
|
+
removes only the one bullet that needs it and downgrades the other nine.
|
|
56
|
+
- `tests/integration/compaction.test.ts`: eleven affected bullets complete the
|
|
57
|
+
compaction with `validationStatus: "warning"` and
|
|
58
|
+
`validationIssueCodes: ["unsupported-exact-spans-unquoted"]` instead of falling
|
|
59
|
+
back, and the same fixture with the two parts at unrelated positions still
|
|
60
|
+
returns `repair=too-many-bullets` with Pi compacting the session.
|
|
61
|
+
- `docs/COMPACTION.md`, `docs/ADR/068-downgrade-rendering-equivalent-spans.md`
|
|
62
|
+
and the ADR index record the decision.
|
|
63
|
+
|
|
64
|
+
## Why
|
|
65
|
+
|
|
66
|
+
The affected-bullet series from production was 30/20 → 23/19 → 19/15 → 17/11 →
|
|
67
|
+
16/12, every run ending in `compaction.custom_fallback` with
|
|
68
|
+
`unsupported-exact-value; repair=too-many-bullets`. The class-only reports
|
|
69
|
+
attributed the rejected spans to rendering or composition in every interpretable
|
|
70
|
+
run: `escaped-form-present` 11 of 16 in the first, then composition 9 of 14 with
|
|
71
|
+
2 one-character deviations, 2 absences and 1 over-length span in the second.
|
|
72
|
+
`no-near-miss` — the class that means invention — was the smallest in both.
|
|
73
|
+
|
|
74
|
+
Each earlier patch removed one class (escaped equivalence in 0.4.4, prompt
|
|
75
|
+
grading in 0.4.5) while the bound stayed exhausted by the rest, so the fallback
|
|
76
|
+
persisted. The information in those spans was present all along; only the
|
|
77
|
+
verbatim surface form was not, and discarding a supported fact — or losing the
|
|
78
|
+
compaction — is the wrong response to a rendering difference. The repair now
|
|
79
|
+
applies deterministically the ladder the 0.4.5 prompt asks the model to follow,
|
|
80
|
+
instead of depending on the model complying
|
|
81
|
+
([ADR 068](ADR/068-downgrade-rendering-equivalent-spans.md)).
|
|
82
|
+
|
|
83
|
+
## Privacy and safety
|
|
84
|
+
|
|
85
|
+
- Only the backticks are removed. Span text is never logged, notified or
|
|
86
|
+
persisted in diagnostics: the recorded effects are two counters, and the
|
|
87
|
+
failure path keeps the class-only report.
|
|
88
|
+
- No provider call is added: the downgrade happens after generation, on text the
|
|
89
|
+
model already produced.
|
|
90
|
+
- The downgraded text is sanitized output that already passed privacy handling;
|
|
91
|
+
removing backticks does not change its classification.
|
|
92
|
+
|
|
93
|
+
## Compatibility and persistence
|
|
94
|
+
|
|
95
|
+
- No configuration key is added or changed. `downgradedSpans` is additive on the
|
|
96
|
+
repair result, and `unsupported-exact-spans-unquoted` is a new issue code in
|
|
97
|
+
the recorded node details.
|
|
98
|
+
- The eight-bullet and 25% bounds are not raised: they now count only the bullets
|
|
99
|
+
that still need removal. A span downgraded in place consumes no bullet.
|
|
100
|
+
- SQLite schema 16, migrations, the Pi JSONL boundary, tool contracts and the
|
|
101
|
+
reference adapter's public surface are untouched.
|
|
102
|
+
|
|
103
|
+
## Measured scope and limitations
|
|
104
|
+
|
|
105
|
+
- Verified by fixtures, not yet by a natural production failure: the first real
|
|
106
|
+
`compaction.custom_fallback` (or its absence) after updating the failing host
|
|
107
|
+
remains the end-to-end confirmation.
|
|
108
|
+
- The fallback is still possible — deliberately — when more than eight bullets
|
|
109
|
+
contain genuinely unsupported spans or when removal would exceed 25% of the
|
|
110
|
+
summary. Those are the cases where the evidence does not hold the information.
|
|
111
|
+
- A downgraded value is still stated, as prose. DS4 stops asserting exactness it
|
|
112
|
+
cannot verify but does not delete the fact, and the counters make the trade
|
|
113
|
+
visible.
|
|
114
|
+
- The repair now consumes the bounded class analysis (default 24000
|
|
115
|
+
evidence-source scans). A span left unanalysed is never downgraded, so budget
|
|
116
|
+
exhaustion keeps the previous fail-closed behaviour.
|
|
117
|
+
- `case-variant-present` is intentionally not downgraded, so summaries that only
|
|
118
|
+
differ by letter case still take the bullet-removal path. If production reports
|
|
119
|
+
show that class dominating, that is the next measured decision.
|
|
120
|
+
|
|
121
|
+
## Validation
|
|
122
|
+
|
|
123
|
+
- `npm run check` on the release commit: builds, `tsc --noEmit`, and
|
|
124
|
+
**100 test files / 649 tests passed** (7 new: 5 unit cases for the repair rule
|
|
125
|
+
and 2 integration cases, one of which pins the unchanged fallback on unrelated
|
|
126
|
+
composition).
|
|
127
|
+
- `npm run pack:check` in a clean consumer: `ds4-context-core@0.4.7`
|
|
128
|
+
(251 files), `ds4-context-reference-adapter@0.4.7` (7 files) and
|
|
129
|
+
`ds4-context-engine@0.4.7` (110 files).
|
|
130
|
+
- `npm run quality:compare`: candidate `task-weighted-v0.2-candidate` 0.9875
|
|
131
|
+
against the frozen baseline `static-ranking-v0.1` 0.808156, unchanged.
|
|
132
|
+
- `npm run schema:context-persistence` (`passed: true`).
|
|
133
|
+
- No provider calls are involved. `jev_verify` is not available in this
|
|
134
|
+
environment (project verification disabled), so the checks above were run
|
|
135
|
+
directly.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ds4-context-engine",
|
|
3
|
-
"version": "0.4.
|
|
3
|
+
"version": "0.4.7",
|
|
4
4
|
"description": "Non-destructive, provider-independent context management for Pi.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -63,7 +63,7 @@
|
|
|
63
63
|
]
|
|
64
64
|
},
|
|
65
65
|
"dependencies": {
|
|
66
|
-
"ds4-context-core": "0.4.
|
|
66
|
+
"ds4-context-core": "0.4.7",
|
|
67
67
|
"js-tiktoken": "1.0.21"
|
|
68
68
|
},
|
|
69
69
|
"peerDependencies": {
|
|
@@ -10,7 +10,9 @@ import {
|
|
|
10
10
|
classifyUnsupportedExactValueSpans,
|
|
11
11
|
groundSummaryFileSections,
|
|
12
12
|
validateSummary,
|
|
13
|
+
type ExactValuePruneResult,
|
|
13
14
|
type SummaryValidationInput,
|
|
15
|
+
type SummaryValidationIssue,
|
|
14
16
|
type SummaryValidationResult,
|
|
15
17
|
type UnsupportedSpanClassReport,
|
|
16
18
|
} from "ds4-context-core/compaction/summary-contract";
|
|
@@ -336,14 +338,7 @@ export async function generateValidatedSummary(
|
|
|
336
338
|
content = pruned.content;
|
|
337
339
|
validation = {
|
|
338
340
|
status: "warning",
|
|
339
|
-
issues: [
|
|
340
|
-
...repairedValidation.issues,
|
|
341
|
-
{
|
|
342
|
-
code: "unsupported-exact-bullets-pruned",
|
|
343
|
-
severity: "warning",
|
|
344
|
-
message: `Removed ${pruned.removedBullets} bullet(s) containing unsupported exact values`,
|
|
345
|
-
},
|
|
346
|
-
],
|
|
341
|
+
issues: [...repairedValidation.issues, ...exactRepairIssues(pruned)],
|
|
347
342
|
};
|
|
348
343
|
} else {
|
|
349
344
|
validation = repairedValidation;
|
|
@@ -370,6 +365,30 @@ export async function generateValidatedSummary(
|
|
|
370
365
|
return { content, validation, usage: sumUsage([...retryUsages, response.usage]) };
|
|
371
366
|
}
|
|
372
367
|
|
|
368
|
+
/**
|
|
369
|
+
* Record what the exact-value repair did, as counts only: the disputed spans
|
|
370
|
+
* never reach logs, notifications or diagnostics because they may contain
|
|
371
|
+
* sensitive source material.
|
|
372
|
+
*/
|
|
373
|
+
function exactRepairIssues(repair: ExactValuePruneResult): SummaryValidationIssue[] {
|
|
374
|
+
const issues: SummaryValidationIssue[] = [];
|
|
375
|
+
if (repair.removedBullets > 0) {
|
|
376
|
+
issues.push({
|
|
377
|
+
code: "unsupported-exact-bullets-pruned",
|
|
378
|
+
severity: "warning",
|
|
379
|
+
message: `Removed ${repair.removedBullets} bullet(s) containing unsupported exact values`,
|
|
380
|
+
});
|
|
381
|
+
}
|
|
382
|
+
if (repair.downgradedSpans > 0) {
|
|
383
|
+
issues.push({
|
|
384
|
+
code: "unsupported-exact-spans-unquoted",
|
|
385
|
+
severity: "warning",
|
|
386
|
+
message: `Retracted the quoting of ${repair.downgradedSpans} exact value(s) whose evidence rendering differs`,
|
|
387
|
+
});
|
|
388
|
+
}
|
|
389
|
+
return issues;
|
|
390
|
+
}
|
|
391
|
+
|
|
373
392
|
export function sumUsage(usages: readonly Usage[]): Usage {
|
|
374
393
|
const sum = (read: (usage: Usage) => number): number => usages.reduce((total, usage) => total + read(usage), 0);
|
|
375
394
|
const hasReasoning = usages.some((usage) => usage.reasoning !== undefined);
|