palimem 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- palimem-0.0.1/.github/workflows/ci.yml +16 -0
- palimem-0.0.1/.gitignore +18 -0
- palimem-0.0.1/LICENSE +21 -0
- palimem-0.0.1/LICENSE-CC-BY-4.0 +10 -0
- palimem-0.0.1/PKG-INFO +43 -0
- palimem-0.0.1/README.md +13 -0
- palimem-0.0.1/docs/PROPOSAL.md +256 -0
- palimem-0.0.1/docs/TASKS.md +194 -0
- palimem-0.0.1/pyproject.toml +37 -0
- palimem-0.0.1/scripts/check_secrets.sh +12 -0
- palimem-0.0.1/src/palimem/__init__.py +2 -0
- palimem-0.0.1/tests/test_import.py +5 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
on: [push, pull_request]
|
|
3
|
+
jobs:
|
|
4
|
+
test:
|
|
5
|
+
runs-on: ubuntu-latest
|
|
6
|
+
strategy:
|
|
7
|
+
matrix:
|
|
8
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
9
|
+
steps:
|
|
10
|
+
- uses: actions/checkout@v4
|
|
11
|
+
- uses: actions/setup-python@v5
|
|
12
|
+
with: { python-version: "${{ matrix.python-version }}" }
|
|
13
|
+
- run: pip install -e ".[dev]"
|
|
14
|
+
- run: ruff check .
|
|
15
|
+
- run: pytest -q
|
|
16
|
+
- run: ./scripts/check_secrets.sh --all
|
palimem-0.0.1/.gitignore
ADDED
palimem-0.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Chris
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
Creative Commons Attribution 4.0 International (CC BY 4.0)
|
|
2
|
+
https://creativecommons.org/licenses/by/4.0/legalcode
|
|
3
|
+
|
|
4
|
+
Applies to: the paper (paper/), the documentation (docs/), the contract text (contract/), and the data
|
|
5
|
+
produced by this project (frozen streams, gold, model outputs, result files), © 2026 Christian Leiva Beltran.
|
|
6
|
+
Attribution: Leiva Beltran, C. (2026). PALIMPSEST: Evaluating Retraction and Uncertainty in LLM Memory.
|
|
7
|
+
Zenodo. https://doi.org/10.5281/zenodo.23127764
|
|
8
|
+
|
|
9
|
+
Third-party material: Wikidata revision contents (bridge/data/s3, via scripts/s3_fetch.py) are CC0.
|
|
10
|
+
The LongMemEval dataset is not redistributed; see its source release for its terms.
|
palimem-0.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: palimem
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Justified memory for LLM agents: evidence-pinned beliefs, retraction that propagates, uncertainty kept rather than guessed. Pre-alpha.
|
|
5
|
+
Project-URL: Repository, https://github.com/chleiva/palimem
|
|
6
|
+
Project-URL: Study, https://doi.org/10.5281/zenodo.23127764
|
|
7
|
+
Author: Christian Leiva Beltran
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
License-File: LICENSE-CC-BY-4.0
|
|
11
|
+
Keywords: agents,belief-revision,llm,memory,truth-maintenance
|
|
12
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
16
|
+
Requires-Python: >=3.11
|
|
17
|
+
Provides-Extra: anthropic
|
|
18
|
+
Requires-Dist: anthropic>=0.40; extra == 'anthropic'
|
|
19
|
+
Provides-Extra: bedrock
|
|
20
|
+
Requires-Dist: boto3>=1.34; extra == 'bedrock'
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: jsonschema>=4; extra == 'dev'
|
|
23
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
24
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
25
|
+
Provides-Extra: mcp
|
|
26
|
+
Requires-Dist: mcp>=1.0; extra == 'mcp'
|
|
27
|
+
Provides-Extra: openai-compat
|
|
28
|
+
Requires-Dist: openai>=1.0; extra == 'openai-compat'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# palimem
|
|
32
|
+
|
|
33
|
+
**Justified memory for LLM agents.** Beliefs are held with the evidence that justifies them; withdrawing or correcting a report repairs every conclusion that depended on it; unresolved alternatives are kept rather than guessed.
|
|
34
|
+
|
|
35
|
+
> **Status: pre-alpha (0.0.x). Nothing here works yet.** This package name is reserved while the SDK is built in the open. Contracts may break at any 0.x release; 1.0 means the gate-G2 acceptance suite passes.
|
|
36
|
+
|
|
37
|
+
palimem is the SDK built on the revision kernel validated in the PALIMPSEST study
|
|
38
|
+
([DOI 10.5281/zenodo.23127764](https://doi.org/10.5281/zenodo.23127764), code at `chleiva/palimpsest`).
|
|
39
|
+
|
|
40
|
+
- Plan and review: [`docs/PROPOSAL.md`](docs/PROPOSAL.md)
|
|
41
|
+
- Task tracker: [`docs/TASKS.md`](docs/TASKS.md)
|
|
42
|
+
|
|
43
|
+
Licence: MIT (code), CC BY 4.0 (documentation and data).
|
palimem-0.0.1/README.md
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# palimem
|
|
2
|
+
|
|
3
|
+
**Justified memory for LLM agents.** Beliefs are held with the evidence that justifies them; withdrawing or correcting a report repairs every conclusion that depended on it; unresolved alternatives are kept rather than guessed.
|
|
4
|
+
|
|
5
|
+
> **Status: pre-alpha (0.0.x). Nothing here works yet.** This package name is reserved while the SDK is built in the open. Contracts may break at any 0.x release; 1.0 means the gate-G2 acceptance suite passes.
|
|
6
|
+
|
|
7
|
+
palimem is the SDK built on the revision kernel validated in the PALIMPSEST study
|
|
8
|
+
([DOI 10.5281/zenodo.23127764](https://doi.org/10.5281/zenodo.23127764), code at `chleiva/palimpsest`).
|
|
9
|
+
|
|
10
|
+
- Plan and review: [`docs/PROPOSAL.md`](docs/PROPOSAL.md)
|
|
11
|
+
- Task tracker: [`docs/TASKS.md`](docs/TASKS.md)
|
|
12
|
+
|
|
13
|
+
Licence: MIT (code), CC BY 4.0 (documentation and data).
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
# PALIMPSEST Memory — Solution Proposal (review of Design v0.3)
|
|
2
|
+
|
|
3
|
+
Date: 2026-10-04 · Reviewer: Claude (Sonnet 5.5) · Input: *PALIMPSEST Memory System — Solution Design v0.3* and the deposited repo at `~/palimpsest` (v1.0, 6 commits)
|
|
4
|
+
Companion file: `TASKS.md` (tracked, parallelisable work breakdown)
|
|
5
|
+
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
## 1. Verdict
|
|
9
|
+
|
|
10
|
+
The design is unusually rigorous for an agent-memory project. Three things make it credible:
|
|
11
|
+
|
|
12
|
+
- **Honest evidence tagging.** Every requirement is tagged *measured / designed / research*.
|
|
13
|
+
- **Clean layering.** Evidence log → admission → kernel → policy, each stage versioned separately.
|
|
14
|
+
- **Gates tied to frozen acceptance tests.**
|
|
15
|
+
|
|
16
|
+
Almost nothing in the open-source memory space (Mem0, Zep/Graphiti, Letta, LangMem) offers retraction propagation, refusal to commit, bitemporal queries or accountable provenance. That is a real, defensible niche.
|
|
17
|
+
|
|
18
|
+
It is **not yet a plan for a world-class open-source project**. It is a plan for a world-class *semantics and engineering core*. Four gaps stand between the two:
|
|
19
|
+
|
|
20
|
+
1. **Day-one usability.** A user needs a declared schema and a fixed-schema extractor before the system does anything useful.
|
|
21
|
+
2. **Trust boundary of the API.** The three-call surface lets the caller (an LLM) choose `source` and `origin`. That defeats the admission model.
|
|
22
|
+
3. **Exponential core.** The kernel is 2^n per key, and the proposed escape hatch (replay) is also 2^n.
|
|
23
|
+
4. **Evidence of agent-level value.** The headline "about 20 points" is mechanism-level and partly true by construction. Nothing yet shows an *agent* behaving better.
|
|
24
|
+
|
|
25
|
+
My proposal keeps the design's semantics and gates. It changes the *delivery strategy* (release train, thin facade, spec-first conformance suite) and adds four lanes the design lacks: agent-facing tooling, security, extractor and entity-resolution quality, and credibility evaluation.
|
|
26
|
+
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
## 2. What I verified versus took on faith
|
|
30
|
+
|
|
31
|
+
I read the design and inspected `~/palimpsest`. I did **not** re-run any of the study's experiments.
|
|
32
|
+
|
|
33
|
+
| Item | Status |
|
|
34
|
+
|---|---|
|
|
35
|
+
| Repo is real and matches the design's description (`palimpsest/core.py` 241 lines, `revise_stream/oracle_v1.py`, `oracle_v2.py`, `docs/SEMANTICS.md` v0.3, Zenodo deposit) | Verified |
|
|
36
|
+
| Test suite: 22 of 24 pass. The 2 failures (`test_scorer.py`) are `ModuleNotFoundError: jsonschema` in my interpreter, not repo bugs | Verified (cause) |
|
|
37
|
+
| `core.py` is **not** a standalone store. It imports `revise_stream.{oracle_v1, gold, model, timeline}`, takes a benchmark `Stream` object, and recomputes via the oracle. The design says Phase 1 "starts from `core.py`"; in practice the SDK kernel is a **port and decoupling**, not an extension | Verified |
|
|
38
|
+
| `oracle_v1._key_interps` enumerates `for mask in range(1 << n)`. **The replay oracle is itself 2^n per key.** The design's "overflow to a per-key replay path" therefore does not escape the exponential | Verified |
|
|
39
|
+
| Kernel time model is integer days; belief time is an arrival index. The design's `timestamp` + `precision: day/month/year` + `recorded_at` is new, unvalidated semantics | Verified |
|
|
40
|
+
| PyPI `palimpsest` is **already taken** (v0.0.2, last release Oct 2023, no description, owner unknown). `palimem` is free on PyPI and npm (404). `github.com/chleiva/palimem` already exists (public, contains only `LICENSE`) | Verified |
|
|
41
|
+
| `.secrets/anthropic_key` exists in the working tree but is gitignored; not in git history; **not** in the Zenodo zip listing | Verified (clean) |
|
|
42
|
+
| README badges and `CITATION` still use `OWNER` / pending push | Verified |
|
|
43
|
+
| All study numbers (40,030 queries, 6–12×, ~20 points, 0.78 poison, 0.167 vs 0.526 LongMemEval, etc.) | Taken from the document |
|
|
44
|
+
|
|
45
|
+
---
|
|
46
|
+
|
|
47
|
+
## 3. Concerns register
|
|
48
|
+
|
|
49
|
+
Severity: **C**ritical (changes the plan), **H**igh (must be resolved at or before G0), **M**edium, **L**ow.
|
|
50
|
+
|
|
51
|
+
### Critical
|
|
52
|
+
|
|
53
|
+
**C1 — Day-one usability gap.** v1 needs (a) a hand-declared schema whose key classes the paper calls "the largest single source of pipeline rework", and (b) a fixed-schema extractor. Open-schema extraction is *research* (R3.1). The non-goals also exclude episodic and summary memory, which is what most agent builders want first. So the first-hour experience of `pip install` → `m.observe("Alice moved to Paris")` fails unless a schema exists.
|
|
54
|
+
*Recommendation:*
|
|
55
|
+
- **Zero-config profile.** Undeclared keys default to `multi_set`, `open`, no inertia (the safest class: no spurious `established`).
|
|
56
|
+
- **Assisted schema drafting.** An LLM proposes `Attr` entries from sample text; a human approves and versions them. This is *not* learning, so it is allowed in v1 and is a stepping stone to R3.1.
|
|
57
|
+
- **Positioning.** Present it as "the belief layer that sits beside your episodic or vector memory", with an adapter that lets existing stores act as *sources*.
|
|
58
|
+
|
|
59
|
+
**C2 — API trust boundary.** `m.observe(text, source=..., origin=...)` lets the caller pick `source` and `origin`. If the caller is the LLM, it can label its own hypothesis `external_observation`, or invent a high-trust source. `actor` and `authority` have the same problem: for a library inside the agent process, the "principal" is whatever the caller says.
|
|
60
|
+
*Recommendation:* two API tiers.
|
|
61
|
+
- **Host API** (trusted code): full `append(report)` with explicit source, origin and actor.
|
|
62
|
+
- **Agent tool API** (what the LLM sees): `remember(text)`, `recall(query)`, `retract(report_id)`, `explain(...)`. The host binds `source`, `origin` and `actor` from connector or session metadata. Any `origin` the LLM supplies is ignored or downgraded to `agent_*`.
|
|
63
|
+
|
|
64
|
+
This must be in G0, not retrofitted.
|
|
65
|
+
|
|
66
|
+
**C3 — The exponential core, and an undefined overflow.** Write cost is 2^n and the cap is n ≤ 12 (4,096 environments per key). A long-lived agent re-asserting "employer is Acme" in every session reaches n = 12 quickly. Overflow "to replay" does not help, because replay enumerates `1 << n` subsets too. The design does not say what an answer *is* above the cap.
|
|
67
|
+
*Recommendation:*
|
|
68
|
+
- Define above-cap behaviour as `ResourceLimited(environment_budget)` for that key. Never silently degrade.
|
|
69
|
+
- Add **equivalence collapsing** before the kernel: reports with the same key, equivalent proposition and `origin_group`, and no distinguishing cue, collapse into one node. This is already implied by "counts once", so make it a kernel precondition. It should cut effective n by a large factor on real conversation logs.
|
|
70
|
+
- **Pull R4.1 forward** as a time-boxed spike (T-I1), not phase 4 in parallel. If a polynomial algorithm exists for the dominant restricted case (single-valued changeable key, same-key disputes), the product story changes. If none exists, we need to know before building storage around enumerated environments.
|
|
71
|
+
|
|
72
|
+
**C4 — Real-world error is dominated by extraction and entity resolution, not the kernel.** The paper's own numbers say last-write-wins overtakes justified belief at about 35% wrong extracted values or about 50% dropped change cues. LongMemEval showed key fragmentation. The design lumps canonicalisation, merging and extraction into one work package (E2.1) with no quality gate.
|
|
73
|
+
*Recommendation:*
|
|
74
|
+
- Make entity resolution (`find`, canonicalise, merge) its own lane with its own labelled evaluation set.
|
|
75
|
+
- Add an **extractor-quality gate (G-X)** with a declared minimum report-level F1 and cue accuracy *before any README claims about natural-language input*.
|
|
76
|
+
- Report an "extraction error budget" next to every benchmark number.
|
|
77
|
+
|
|
78
|
+
**C5 — Evidence base is mechanism-level and synthetic.** The ~20-point effect is on "warranted" downstream queries, which are *defined* as queries whose gold changes after a retraction, so a store without propagation is guaranteed to fail them. That is a valid mechanism test, but it is not evidence that an *agent* acts better. The comparison set is also all in-house: no Mem0, Zep/Graphiti or Letta baselines are run (the design forbids them in the *critical path*, which is not the same as forbidding them as *evaluation baselines*).
|
|
79
|
+
*Recommendation:*
|
|
80
|
+
- **Agent-level benchmark** (T-J1): tasks where a withdrawn or corrected fact changes the right action (customer record corrections, revoked permissions, retracted news). Metric: harmful-action rate and unnecessary-ask rate.
|
|
81
|
+
- **Baseline comparisons** (T-J2) against LWW, Mem0, Zep/Graphiti and Letta, as baselines only, on REVISE-STREAM plus the agent benchmark.
|
|
82
|
+
- **Publish REVISE-STREAM** as a standalone benchmark package. A world-class project usually owns a benchmark that others use.
|
|
83
|
+
|
|
84
|
+
**C6 — Distribution name collision. RESOLVED (2026-10-04).** `pip install palimpsest` is impossible, so the SDK ships as **`palimem`** (repo `github.com/chleiva/palimem`). The design's `pip install palimpsest` and `from palimpsest import Memory` become `pip install palimem` and `from palimem import Memory`. The study keeps the name PALIMPSEST; the SDK is the product built on it. Remaining actions: reserve `palimem` on PyPI and npm (T-0.1), and update every `palimpsest` import and README reference in the design text.
|
|
85
|
+
|
|
86
|
+
### High
|
|
87
|
+
|
|
88
|
+
**H1 — Sequencing: G0 is too big to be a single blocking gate.** G0 freezes the status ladder, three-stage pipeline, two-axis query, per-interval support, authority, resource contract and about 23 fixtures. Meanwhile Phase 1 "starts from `core.py`", which does not match the new semantics (see §2).
|
|
89
|
+
*Recommendation:* **release train** (§5), with `0.x` explicitly allowed to break the contract and `1.0` meaning "G2 passed". Parallel lanes start from interface stubs (see `TASKS.md`).
|
|
90
|
+
|
|
91
|
+
**H2 — Spec ambiguities that will cause rework (each needs a G0 decision).**
|
|
92
|
+
|
|
93
|
+
| # | Ambiguity |
|
|
94
|
+
|---|---|
|
|
95
|
+
| S-1 | `confirm` appears as a cue a source can emit, yet the admission model *derives* confirmation from equivalent reports of another origin group. Which is it, and may a `confirm` cue itself corroborate? |
|
|
96
|
+
| S-2 | Paper A-SELF: a correction from the **same origin** withdraws its target. Design default authority for `correct`/`withdraw` is the target's **own source**. Different sources sharing an origin pass the paper's rule but fail the design's. The `revise-stream-v1` profile must set authority to origin-group, or G1 cannot reach 0 disagreements. |
|
|
97
|
+
| S-3 | Paper `blocked` class (never admitted, never confirmable) versus design `quarantined` (admissible on confirmation). Map explicitly. |
|
|
98
|
+
| S-4 | The prose ladder lists three statuses (`established`, `unresolved`, `unknown`); the schema lists five (+`established_empty`, `established_false`). Pick one enum. |
|
|
99
|
+
| S-5 | `belief_as_of` defined over a wall-clock `recorded_at`. Ties, clock skew and ordering across restarts break determinism. Define it over the log sequence number, with timestamps as a lookup index. |
|
|
100
|
+
| S-6 | Behaviour above the environment cap (see C3). |
|
|
101
|
+
| S-7 | `principal`, "principal rule" and "actor's key scope" are used but never defined. No principal model exists. |
|
|
102
|
+
| S-8 | Paper A4 makes inertia a universal law; the design makes it a per-attribute flag. The compat profile must set it so that all 40,030 queries still match. |
|
|
103
|
+
| S-9 | Partial-date precision (`month`, `year`) semantics: interval-valued anchors versus the integer-day sweep. New and unvalidated. |
|
|
104
|
+
| S-10 | Open-world rules: a defeasible rule's *exception* is "blocked when the exception holds". Under open-world, does an *unknown* exception block? The paper's closed-world answer was no. |
|
|
105
|
+
| S-11 | Quarantined `attributed` reports, and nested `belief_of(belief_of(...))`. |
|
|
106
|
+
| S-12 | `explain` returns supports "two levels deep", but derivation chains can be deeper ("two or more steps downstream"). Define depth or make it a parameter. |
|
|
107
|
+
| S-13 | Tombstone fields (`key`, `actor`, `reason`) can themselves be personal data (see M3). |
|
|
108
|
+
|
|
109
|
+
**H3 — The "independent" acceptance suite and oracle share an author.** `oracle_v1`, `oracle_v2`, the store and the independent tests all have one author, with LLM-assisted reviews.
|
|
110
|
+
*Recommendation:* before G0 freeze, get **one external review** from someone in belief revision / truth maintenance, and add a **property-based differential fuzzer** (T-E4) that generates streams outside the generator's distribution.
|
|
111
|
+
|
|
112
|
+
**H4 — Store-wide dirty marker is an availability hazard.** When one append exhausts the traversal budget, *every* read returns `ResourceLimited(store_dirty)`. One noisy or hostile key can take the whole memory offline.
|
|
113
|
+
*Recommendation:* scope the dirty marker to the **connected component of the attribute dependency graph**, which is known statically from schema rules. Keep the store-wide marker only as a last resort.
|
|
114
|
+
|
|
115
|
+
**H5 — LongMemEval positioning.** The design is honest that justified belief scores 0.167 against LWW's 0.526 on knowledge-updates. The community will nevertheless compare numbers.
|
|
116
|
+
*Recommendation:*
|
|
117
|
+
- Ship **named policy presets** (`justified`, `recency` = P0cSU + LWW commit) with their measured numbers side by side.
|
|
118
|
+
- Own a retraction-aware benchmark (see C5), so the project is not only ever scored on a target it disclaims.
|
|
119
|
+
|
|
120
|
+
### Medium
|
|
121
|
+
|
|
122
|
+
**M1 — Language and ecosystem reach.** Python only, and no tool-facing surface for LLMs at all (MCP, tool schemas, framework adapters, guidance on how an LLM should read `ask` / `unresolved` / `resource_limited`).
|
|
123
|
+
*Recommendation:* lane F. Make the **conformance fixtures language-neutral JSON** so other-language ports can prove compatibility.
|
|
124
|
+
|
|
125
|
+
**M2 — Budget. DECIDED: $20 total for new LLM spend, low-cost models only** (for example gpt-oss-20b, Mistral Small or similar). This has consequences:
|
|
126
|
+
- **Extractor.** Provider-agnostic: an OpenAI-compatible interface is the default, Anthropic becomes an optional extra rather than the shipped implementation. Small models will extract less reliably than the study's extractor, so the G-X thresholds must be declared *per model*. The paper's own fragility points (about 35% wrong values, about 50% dropped change cues) are the budget to design against.
|
|
127
|
+
- **Local inference.** gpt-oss-20b can run locally if the machine has enough memory; that would make extractor iteration free. I have not checked the hardware and will not assume it.
|
|
128
|
+
- **Baselines (T-J2).** Scoped to LWW plus one or two third-party systems configured to use the same cheap model, on a subset of REVISE-STREAM and a small agent benchmark. All runs go through the cost ledger with a hard cap; nothing proceeds without a pre-run cost estimate.
|
|
129
|
+
- **Load week and the sealed held-out stratum** are symbolic and cost nothing; they are unaffected.
|
|
130
|
+
|
|
131
|
+
**M3 — Privacy.**
|
|
132
|
+
- A tombstone with `key` and `actor` can reveal sensitive facts (for example `(alice, health_condition)`). Store hashed or minimal tombstone metadata.
|
|
133
|
+
- `raw_ref` and extractor prompts leave the process whenever a hosted extractor is used (any provider); the README needs a data-flow note, and the no-LLM and local-model paths should be documented as the private options.
|
|
134
|
+
- Erasure from backups needs an operational procedure, not just a sentence.
|
|
135
|
+
- Legal review is required for the GDPR claims.
|
|
136
|
+
|
|
137
|
+
**M4 — Concurrency semantics.** "One process" is fine, but MCP servers and multi-agent setups need a defined single-writer discipline (SQLite WAL, a writer lock, read snapshots). Specify it so the barrier logic has a defined concurrency model.
|
|
138
|
+
|
|
139
|
+
**M5 — Derived-key rules under open-world.** Related to S-10. Static exactness checks cover disjoint closures but not the interaction of negative evidence and defeasible exceptions. Needs fixtures.
|
|
140
|
+
|
|
141
|
+
### Low
|
|
142
|
+
|
|
143
|
+
**L1 — Repo hygiene.**
|
|
144
|
+
- Move `.secrets/anthropic_key` out of the project tree (it is safely ignored today, but one bad `git add -f` or zip step leaks it).
|
|
145
|
+
- Replace the `OWNER` placeholders.
|
|
146
|
+
- Push the pending commits.
|
|
147
|
+
- Add `jsonschema` to a dev extra so a fresh checkout is green.
|
|
148
|
+
|
|
149
|
+
**L2 — Debuggability as adoption lever.** A provenance viewer or CLI (`palimpsest explain`, `inspect`, `diff`) is cheap given the data model and is the most demonstrable thing the system does.
|
|
150
|
+
|
|
151
|
+
---
|
|
152
|
+
|
|
153
|
+
## 4. Proposed solution
|
|
154
|
+
|
|
155
|
+
### 4.1 Architecture (design v0.3 unchanged except where marked)
|
|
156
|
+
|
|
157
|
+
```
|
|
158
|
+
┌────────────── HOST (trusted) ───────────────┐
|
|
159
|
+
agent / LLM ──► │ Agent Tool API: remember · recall · retract │
|
|
160
|
+
(untrusted) │ binds source/origin/actor from session │ ◄── NEW (C2)
|
|
161
|
+
└───────────────┬─────────────────────────────┘
|
|
162
|
+
▼
|
|
163
|
+
Extractor ──► Evidence log ──► Admission ──► Kernel ──► Store ──► Policy ──► Answer (v2)
|
|
164
|
+
(pluggable, append-only origin, per-key, versioned commit/ kernel_status
|
|
165
|
+
stamped) idempotent quarantine, open-world, beliefs, abstain/ + decision
|
|
166
|
+
authority, segments, barrier, ask + provenance
|
|
167
|
+
confirmation collapse ◄─ outbox
|
|
168
|
+
NEW (C3)
|
|
169
|
+
Entity resolution (find · canonicalise · merge) is its own recorded admission stage ◄── NEW lane (C4)
|
|
170
|
+
Learning loop and harness beside the pipeline; conformance fixtures are language-neutral JSON ◄── NEW (M1)
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### 4.2 Changes I recommend to the design
|
|
174
|
+
|
|
175
|
+
| # | Change | Reason |
|
|
176
|
+
|---|---|---|
|
|
177
|
+
| 1 | Two-tier API (host vs agent tool) in G0 | C2 |
|
|
178
|
+
| 2 | Equivalence collapsing as a kernel precondition; above-cap = `ResourceLimited(environment_budget)`; R4.1 spike first | C3 |
|
|
179
|
+
| 3 | `belief_as_of` over log sequence number | S-5 |
|
|
180
|
+
| 4 | Zero-config profile and assisted schema drafting in v1 | C1 |
|
|
181
|
+
| 5 | Entity resolution as a standalone lane with its own eval set | C4 |
|
|
182
|
+
| 6 | Extractor-quality gate G-X and "extraction error budget" reporting | C4 |
|
|
183
|
+
| 7 | Component-scoped dirty marker | H4 |
|
|
184
|
+
| 8 | Policy presets (`justified`, `recency`) shipped and documented | H5 |
|
|
185
|
+
| 9 | Security lane: threat model, poisoning evaluation, privacy review | C2, M3 |
|
|
186
|
+
| 10 | Credibility lane: agent-level benchmark, external baselines, standalone REVISE-STREAM | C5 |
|
|
187
|
+
| 11 | Release train `0.1 → 1.0` instead of one big v1 | H1 |
|
|
188
|
+
| 12 | Language-neutral conformance fixtures | M1, H3 |
|
|
189
|
+
|
|
190
|
+
### 4.3 Gates
|
|
191
|
+
|
|
192
|
+
Existing gates are kept as written in v0.3 (G0–G4). Added gates:
|
|
193
|
+
|
|
194
|
+
- **G-S (security).** Threat model published; poisoning evaluation shows quarantine and confirmation lower injected-commit rate below a declared limit; the agent tool API cannot set `origin` / `source` (tested).
|
|
195
|
+
- **G-X (extraction).** On a labelled natural-language set, report-level F1 and operator-cue accuracy meet declared minimums; every README benchmark carries the extraction error budget.
|
|
196
|
+
- **G-A (agent-level).** On the agent benchmark, the system lowers harmful-action rate against LWW and at least one third-party baseline, at a stated cost in unnecessary asks.
|
|
197
|
+
|
|
198
|
+
### 4.4 Release train
|
|
199
|
+
|
|
200
|
+
| Release | Content | Passes |
|
|
201
|
+
|---|---|---|
|
|
202
|
+
| **0.1** | Decoupled kernel, in-memory and SQLite backend, closed-world compat + basic open-world ladder, three-call facade, differential CI against the frozen sets | G0-lite, differential gate |
|
|
203
|
+
| **0.2** | Admission (origin, quarantine, authority, confirmation), two-axis query, `ResourceLimited`, agent tool API | G1 |
|
|
204
|
+
| **0.3** | Generation barrier, completion jobs, outbox, `subscribe` / `notify`, versioned inputs | G1 (full) |
|
|
205
|
+
| **0.4** | Extractor, entity resolution, `find`, assisted schema drafting, MCP server, framework adapters | G-X, G-S |
|
|
206
|
+
| **1.0** | Declared performance targets met on the load week, README obligations, packaging | G2, G-A |
|
|
207
|
+
| **1.x / 2.x (research)** | Schema induction, learned policy, calibration, consolidation | G3, G4 |
|
|
208
|
+
|
|
209
|
+
### 4.5 Parallelisation
|
|
210
|
+
|
|
211
|
+
After a short **Wave 0** (spec decisions plus work that needs no spec), five lanes run independently against interface stubs: kernel (B), store (C), admission and policy (D), harness (E) and agent surface (F). Lanes G (extraction), H (security), I (research) and J (evaluation) can start in Wave 0 because they do not depend on the final contract. Dependencies and wave assignment are in `TASKS.md`.
|
|
212
|
+
|
|
213
|
+
I can run these lanes as parallel sub-agents in isolated git worktrees once you approve the plan and the repo strategy. Each lane would deliver code plus the fixtures it owns, and the harness lane's differential CI would be the merge gate.
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
## 5. Success criteria for "world-class"
|
|
218
|
+
|
|
219
|
+
- Conformance: **0** status/value/alternative disagreements with the oracle on every frozen query, through the v1 adapter, on every PR.
|
|
220
|
+
- Safety: published threat model; injected-commit rate and extractor error budget reported in the README.
|
|
221
|
+
- Performance: declared absolute p99 query and append latencies, memory per million reports and recovery time, met on a one-week load test.
|
|
222
|
+
- Credibility: agent-level benchmark and third-party baselines published, including results that go against the system.
|
|
223
|
+
- Adoption: `pip install` to first useful answer in under five minutes with zero-config; MCP server working with at least two agent frameworks; provenance CLI.
|
|
224
|
+
- Openness: language-neutral conformance suite, RFC process for contract changes, semantic-versioned contracts.
|
|
225
|
+
|
|
226
|
+
---
|
|
227
|
+
|
|
228
|
+
## 6. Assumptions (change any of these and the plan changes)
|
|
229
|
+
|
|
230
|
+
Decided by you on 2026-10-04:
|
|
231
|
+
|
|
232
|
+
1. The SDK lives in a **new repo, `github.com/chleiva/palimem`** (public, currently only `LICENSE`); this directory is its working copy. `~/palimpsest` stays the frozen study and benchmark repo, consumed as a pinned dependency for the harness.
|
|
233
|
+
2. Distribution name **`palimem`**.
|
|
234
|
+
3. **Release train from 0.1** (§4.4).
|
|
235
|
+
4. **$20 of LLM spend for the evaluation work, low-cost models only** (M2).
|
|
236
|
+
|
|
237
|
+
Still my assumptions:
|
|
238
|
+
|
|
239
|
+
5. The **import name is `palimem`** (`from palimem import Memory`), matching the distribution. Say so if you want `palimpsest` kept as the import name.
|
|
240
|
+
6. The $20 is the whole remaining budget for new work, not on top of the $11.15 left in the study ledger.
|
|
241
|
+
7. Python 3.11+, MIT code, CC BY 4.0 for docs and data, as in the design.
|
|
242
|
+
|
|
243
|
+
## 7. Open questions
|
|
244
|
+
|
|
245
|
+
Answered 2026-10-04: repo (new, `chleiva/palimem`), distribution name (`palimem`), release strategy (train from 0.1), budget ($20, cheap models).
|
|
246
|
+
|
|
247
|
+
Still open, defaults noted:
|
|
248
|
+
|
|
249
|
+
- **Import name** (default `palimem`) and whether the $20 is inclusive of the $11.15 ledger (default yes).
|
|
250
|
+
- **Extraction model.** Which cheap model is the default for development: a hosted gpt-oss-20b or Mistral Small endpoint, or a local one? (decides the T-G2 default and the G-X thresholds)
|
|
251
|
+
- **Publishing.** Reserving `palimem` on PyPI and npm are outward-facing; I will not do either without a go-ahead.
|
|
252
|
+
- **Positioning.** Belief layer beside episodic memory (default) versus all-in-one memory.
|
|
253
|
+
- **Authority model.** Minimal principal model (string ids plus per-attribute rules; default) versus pluggable.
|
|
254
|
+
- **TypeScript client.** Not in v1 (default); conformance fixtures keep the door open.
|
|
255
|
+
- **External reviewer.** Who reviews G0 before freeze? (needed for H3)
|
|
256
|
+
- **Legal.** Is a legal review of the deletion and GDPR design available?
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
# PALIMPSEST Memory — Task Tracker
|
|
2
|
+
|
|
3
|
+
Companion to `PROPOSAL.md`. Update the checkbox and `Status` as work proceeds.
|
|
4
|
+
|
|
5
|
+
**Size:** S ≤ 2 days · M ≈ 1 week · L > 1 week (relative, not a schedule commitment; the design deliberately avoids dates for research).
|
|
6
|
+
**Wave:** 0 = start now, no spec dependency · 1 = after G0-lite (spec decisions A4/A5 frozen) · 2 = after the lane's wave-1 dependencies · 3 = gates and release.
|
|
7
|
+
**Parallel:** tasks in the same wave with no listed dependency between them can run concurrently (separate worktrees).
|
|
8
|
+
|
|
9
|
+
Status legend: `[ ]` todo · `[~]` in progress · `[x]` done · `[!]` blocked
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
## Lane 0 — Hygiene (Wave 0, all parallel, all S)
|
|
14
|
+
|
|
15
|
+
| ✓ | ID | Task | Deps | Acceptance |
|
|
16
|
+
|---|---|---|---|---|
|
|
17
|
+
| [~] | T-0.1 | Distribution name **decided: `palimem`** (free on PyPI and npm as of 2026-10-04). Remaining: reserve it on PyPI and npm (needs your go-ahead, outward-facing); confirm import name `palimem` | — | Name reserved on PyPI; import name confirmed |
|
|
18
|
+
| [~] | T-0.2 | Repo **decided: `github.com/chleiva/palimem`** (exists, public, only `LICENSE`). Remaining: clone it into this directory, add `LICENSE-CC-BY-4.0`, README stub, `.gitignore`, pre-commit secret scan | T-0.1 | Working copy cloned; baseline files committed |
|
|
19
|
+
| [ ] | T-0.3 | In `~/palimpsest`: push pending commits, replace `OWNER` placeholders, add `jsonschema` to dev requirements, move `.secrets/anthropic_key` outside the tree | — | Fresh clone: `pytest` is 24/24 green; no secrets in tree |
|
|
20
|
+
| [ ] | T-0.4 | Open issues for E1.1–E1.5 and every task below (after the user approves) | T-0.2 | Issues exist and link to this file |
|
|
21
|
+
|
|
22
|
+
---
|
|
23
|
+
|
|
24
|
+
## Lane A — Spec and contracts (G0)
|
|
25
|
+
|
|
26
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
27
|
+
|---|---|---|---|---|---|---|
|
|
28
|
+
| [ ] | T-A1 | **Resolve spec ambiguities S-1 … S-13** (PROPOSAL §3 H2): one decision record each, in `docs/decisions/` | — | 0 | M | 13 records; each states choice, reason, effect on compat profile |
|
|
29
|
+
| [ ] | T-A2 | **Two-tier API and trust boundary** (C2): host API vs agent tool API; what the host binds; downgrade rules | T-A1 | 0 | S | Spec section + 5 negative fixtures (LLM-supplied origin ignored, etc.) |
|
|
30
|
+
| [ ] | T-A3 | Executable types: `Report`, `Proposition`, `Attr`, `Candidate`, `Support`, `Segment`, `Belief`, `BeliefView`, `Query`, `Answer` (dataclasses + validation, stdlib only) | T-A1 | 1 | M | Types import; round-trip to/from JSON; mypy-clean |
|
|
31
|
+
| [ ] | T-A4 | JSON Schemas for all records and the v2 output contract (language-neutral) | T-A3 | 1 | S | `jsonschema` validates every fixture |
|
|
32
|
+
| [ ] | T-A5 | **Conformance fixtures** as JSON: the 23 independent tests of design v0.3 + negative-evidence-under-rules fixtures (S-10/M5) + trust-boundary fixtures (T-A2) | T-A3 | 1 | L | Fixture runner executes against a stub; all fixtures have expected output |
|
|
33
|
+
| [ ] | T-A6 | Versioning policy: contract semver, store-format version, RFC process | — | 0 | S | `docs/VERSIONING.md` |
|
|
34
|
+
| [ ] | T-A7 | **External review** of G0 by one belief-revision / TMS reviewer (H3) | T-A1, T-A5 | 1 | S (calendar-bound) | Written review, issues triaged |
|
|
35
|
+
| [ ] | **G0** | Contracts and fixtures frozen as versioned artefacts | T-A1…A7 | 3 | — | Tagged `contracts-v2.0-rc1` |
|
|
36
|
+
|
|
37
|
+
---
|
|
38
|
+
|
|
39
|
+
## Lane B — Kernel (needs T-A3 types; can start against drafts)
|
|
40
|
+
|
|
41
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
42
|
+
|---|---|---|---|---|---|---|
|
|
43
|
+
| [ ] | T-B1 | **Decouple kernel from the benchmark `Stream`**: pure function `justify(evidence_set, attr, semantic_cfg) → Belief` ported from `oracle_v1` / `timeline`; no imports from `revise_stream` | T-A3 | 1 | L | Kernel tests pass on hand-built reports; no benchmark imports |
|
|
44
|
+
| [ ] | T-B2 | Open-world status ladder, negative evidence (`not_value`, `not_member`, `enumeration([])`), completeness modes | T-B1 | 2 | M | Fixtures "no evidence / explicit empty / explicit denial" and "positive + negative → unresolved" green |
|
|
45
|
+
| [ ] | T-B3 | Valid-time segments; partial-date precision; per-attr inertia | T-B1 | 2 | L | Temporal fixtures green; compat profile reproduces integer-day semantics |
|
|
46
|
+
| [ ] | T-B4 | Per-candidate, per-interval subset-minimal environments; provenance contract | T-B1 | 2 | M | Closes the 2–5% provenance gap vs oracle; "disjoint intervals" fixture green |
|
|
47
|
+
| [ ] | T-B5 | Static exactness check at schema load (same-key relations, disjoint rule closures); {A,B,∅} counter-example as test | T-B1 | 2 | S | Counter-example refused; global-oracle fallback hook present |
|
|
48
|
+
| [ ] | T-B6 | **Equivalence collapsing** before the kernel (same key, equivalent proposition, same `origin_group`, no distinguishing cue) | T-B1 | 2 | M | Same answers on frozen sets with measurably smaller n; property test |
|
|
49
|
+
| [ ] | T-B7 | Environment-budget behaviour: above cap → `ResourceLimited(environment_budget)`; no silent degradation | T-B6, T-A1(S-6) | 2 | S | Budget fixture green |
|
|
50
|
+
| [ ] | T-B8 | A-SU same-origin self-update as semantic config (`P0c`, `P0cSU`) | T-B1 | 2 | S | Both configs match deposited results |
|
|
51
|
+
| [ ] | T-B9 | Derived keys: rule evaluation under open-world and defeasible exceptions (S-10) | T-B2, T-A1 | 2 | M | Rule fixtures green; chain depth ≥ 3 tested |
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## Lane C — Store (needs T-A3; backends against a shared interface)
|
|
56
|
+
|
|
57
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
58
|
+
|---|---|---|---|---|---|---|
|
|
59
|
+
| [ ] | T-C1 | Backend interface (five operations) and **in-memory reference backend** | T-A3 | 1 | M | Interface test-suite passes on the reference backend |
|
|
60
|
+
| [ ] | T-C2 | SQLite backend: tables, WAL, writer lock, single transaction per append, idempotency keys, crash recovery | T-C1 | 2 | L | Crash-injection test: no duplicate, no partial revision |
|
|
61
|
+
| [ ] | T-C3 | Log-sequence `belief_as_of` resolution and version history | T-C1, T-A1(S-5) | 2 | M | Retrospective-correction fixture green on both axes |
|
|
62
|
+
| [ ] | T-C4 | Generation barrier, `required` / `completed` generation, durable completion jobs, **component-scoped** dirty marker | T-C2, T-B1 | 2 | L | All four barrier fixtures green; store-wide marker only as fallback |
|
|
63
|
+
| [ ] | T-C5 | Notification outbox and `subscribe` / `notify` (at-least-once, stable `event_id`) | T-C2 | 2 | M | Crash-after-commit fixture green |
|
|
64
|
+
| [ ] | T-C6 | Versioned inputs: schema, rule, semantic, admission, policy versions; historical queries use versions current at `belief_as_of` | T-C3 | 2 | M | Historical-version fixture green |
|
|
65
|
+
| [ ] | T-C7 | Entity merges as reversible, pinned records | T-C2, T-G3 | 2 | M | "Mistaken merge reversed" fixture green |
|
|
66
|
+
| [ ] | T-C8 | Deletion vs withdrawal; minimal/hashed tombstones; backup-erasure procedure | T-C2, T-H4 | 2 | M | Deletion fixture green; tombstone leaks no sensitive key text |
|
|
67
|
+
| [ ] | T-C9 | JSONL export/import of the evidence log; store-format migrations | T-C2 | 2 | M | Export → import reproduces identical beliefs |
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## Lane D — Admission and policy
|
|
72
|
+
|
|
73
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
74
|
+
|---|---|---|---|---|---|---|
|
|
75
|
+
| [ ] | T-D1 | Admission pipeline: origin check, quarantine, confirmation, origin groups, recorded decisions with `admission_version` | T-A3, T-A1(S-1,S-3) | 1 | M | Quarantine and agent-origin fixtures green |
|
|
76
|
+
| [ ] | T-D2 | Principal and authority model; `allege` downgrade | T-A1(S-7), T-D1 | 2 | M | "Unauthorised withdrawal" fixture green; same-origin rule per S-2 |
|
|
77
|
+
| [ ] | T-D3 | Policy object and `decide()`; presets: `justified`, `recency` (P0cSU + LWW commit), `lww` | T-A3 | 1 | M | Presets reproduce the study's regime table through the adapter |
|
|
78
|
+
| [ ] | T-D4 | `inquiry` generation (competing candidates, deciding keys, resolver source classes) | T-D3, T-B9 | 2 | M | Inquiry fixture names resolver |
|
|
79
|
+
| [ ] | T-D5 | Attribution handling (`belief_of`) and its fixtures | T-D1 | 2 | S | "Two reports that Alice believes P" green |
|
|
80
|
+
|
|
81
|
+
---
|
|
82
|
+
|
|
83
|
+
## Lane E — Harness and compatibility (Wave 0 start; **merge gate for everything**)
|
|
84
|
+
|
|
85
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
86
|
+
|---|---|---|---|---|---|---|
|
|
87
|
+
| [ ] | T-E1 | **Differential CI first**: harness that runs the existing store against the oracle on the frozen sets, failing on any status/value/alternative disagreement; pin `~/palimpsest` as a versioned dependency | T-0.2 | 0 | M | CI red when a seeded bug is injected; green on baseline |
|
|
88
|
+
| [ ] | T-E2 | `frozen.json` checksum guard against the Zenodo deposit (silent gold drift) | T-E1 | 0 | S | Altered file fails the build |
|
|
89
|
+
| [ ] | T-E3 | v1 compatibility adapter (`revise-stream-v1` profile) from v2 Answer to v1 contract | T-A3 | 1 | M | Adapter unit tests; frozen Setting 1 reproduces deposited numbers |
|
|
90
|
+
| [ ] | T-E4 | **Property-based differential fuzzer** that generates streams outside the benchmark generator's distribution; shrinks counter-examples | T-E1 | 1 | M | Finds a seeded bug; runs nightly |
|
|
91
|
+
| [ ] | T-E5 | Absolute performance targets declared and a benchmark suite (p99 query, p99 append-to-visible, memory per million reports, recovery time) | T-C2 | 2 | M | Targets written before the load week; suite runs in CI nightly |
|
|
92
|
+
| [ ] | T-E6 | LLM cost ledger and **hard cap of $20** carried into the new repo (port `budget.py`); pre-run cost estimate required for every paid run | T-0.2 | 0 | S | Calls without the ledger fail; run is refused if the estimate exceeds the remaining cap |
|
|
93
|
+
|
|
94
|
+
---
|
|
95
|
+
|
|
96
|
+
## Lane F — Agent surface (adoption lane)
|
|
97
|
+
|
|
98
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
99
|
+
|---|---|---|---|---|---|---|
|
|
100
|
+
| [ ] | T-F1 | Three-call facade `Memory.observe / ask / withdraw` over the host API | T-A3, T-C1 | 1 | S | Quickstart runs on the reference backend |
|
|
101
|
+
| [ ] | T-F2 | **Agent tool API**: `remember`, `recall`, `retract`, `explain` with host-bound source/origin/actor | T-A2, T-F1 | 2 | M | Trust-boundary fixtures green |
|
|
102
|
+
| [ ] | T-F3 | MCP server over the agent tool API | T-F2 | 2 | M | Works with at least two MCP clients |
|
|
103
|
+
| [ ] | T-F4 | Framework adapters (pick two first: Claude Agent SDK, LangGraph or OpenAI Agents SDK) | T-F2 | 2 | M | Example agents run end to end |
|
|
104
|
+
| [ ] | T-F5 | "How an LLM should read an Answer" guide and tool descriptions (`ask`, `unresolved`, `resource_limited`) | T-F2 | 2 | S | Tested against a prompted agent on fixture answers |
|
|
105
|
+
| [ ] | T-F6 | CLI: `explain`, `inspect`, `diff`, `export` (provenance viewer) | T-C3, T-B4 | 2 | M | Demo recorded for README |
|
|
106
|
+
| [ ] | T-F7 | Zero-config profile (undeclared keys default to `multi_set`, `open`, no inertia) | T-A1, T-B2 | 2 | S | Works without a schema file; never returns `established` from absence |
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## Lane G — Extraction and entity resolution
|
|
111
|
+
|
|
112
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
113
|
+
|---|---|---|---|---|---|---|
|
|
114
|
+
| [ ] | T-G1 | Extractor interface; no-LLM path for typed input | T-A3 | 1 | S | Typed reports pass through unchanged |
|
|
115
|
+
| [ ] | T-G2 | **Provider-agnostic extractor** (OpenAI-compatible endpoint by default; low-cost models such as gpt-oss-20b or Mistral Small; Anthropic as optional extra) with per-report stamp (model, prompt hash, raw ref) | T-G1, T-E6 | 2 | M | Re-run on frozen Setting 2 inputs through the adapter; diff vs deposit reported per model (empty diff is required only for the study's original extractor outputs, which come from cache) |
|
|
116
|
+
| [ ] | T-G3 | Entity resolution: `find`, canonicalisation, merge decisions as recorded admission steps | T-D1 | 2 | L | Name-variant fixtures; merge pass is a versioned step, not a patch |
|
|
117
|
+
| [ ] | T-G4 | **Extractor-quality evaluation set** (labelled NL → typed reports, with cue labels) and G-X thresholds | T-G1 | 1 | L | Dataset and scorer published; thresholds declared before first run |
|
|
118
|
+
| [ ] | T-G5 | Temporal expression normalisation ("last week", "since March") | T-G2, T-B3 | 2 | M | Accuracy on the T-G4 temporal slice |
|
|
119
|
+
| [ ] | T-G6 | Assisted schema drafting: LLM proposes `Attr` entries, human approves, versioned | T-G2, T-A3 | 2 | M | Draft → review → load flow; every approved attr is versioned |
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
## Lane H — Security and privacy
|
|
124
|
+
|
|
125
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
126
|
+
|---|---|---|---|---|---|---|
|
|
127
|
+
| [ ] | T-H1 | **Threat model**: poisoning, prompt-injection into reports, spoofed source/origin, authority abuse, dirty-marker DoS, resource exhaustion, privacy | — | 0 | M | `docs/THREAT_MODEL.md`; each threat has a mitigation or an accepted-risk note |
|
|
128
|
+
| [ ] | T-H2 | Poisoning evaluation: extend the existing `poison_streams` to confirm quarantine + confirmation lower the 0.78 injected-commit rate | T-H1, T-D1 | 2 | M | Declared limit set up front; result reported either way |
|
|
129
|
+
| [ ] | T-H3 | `SECURITY.md`, dependency and SBOM policy, input hardening (SQL, size limits) | T-H1 | 1 | S | Policy files; fuzz on parsers |
|
|
130
|
+
| [ ] | T-H4 | Privacy and GDPR design review (tombstones, backups, data flow to extractor providers) | T-H1 | 1 | M | Written findings; legal reviewer sign-off if available |
|
|
131
|
+
| [ ] | **G-S** | Security gate | T-H1…H4, T-F2 | 3 | — | Per PROPOSAL §4.3 |
|
|
132
|
+
|
|
133
|
+
---
|
|
134
|
+
|
|
135
|
+
## Lane I — Research (does not block engineering except T-I1)
|
|
136
|
+
|
|
137
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
138
|
+
|---|---|---|---|---|---|---|
|
|
139
|
+
| [ ] | T-I1 | **R4.1 spike, time-boxed (default 2 weeks)**: polynomial algorithm or restricted class for justification maintenance on the dominant single-valued changeable case; BDD / d-DNNF feasibility | — | 0 | M | Go / no-go memo; informs T-B6, T-B7 and T-C2 |
|
|
140
|
+
| [ ] | T-I2 | R3.1 schema induction (builds on T-G6) | T-G6, T-G4 | 3 | L | Per design phase 3 |
|
|
141
|
+
| [ ] | T-I3 | R3.2 learned policy; **seal the held-out stratum before fitting** | T-D3, T-E1 | 3 | L | G3 comparison rule |
|
|
142
|
+
| [ ] | T-I4 | R3.3 calibrated confidence (new method) | T-I3 | 3 | L | Passes the harness |
|
|
143
|
+
| [ ] | T-I5 | R4.2 and R4.3 consolidation and above-budget stratum | T-I1 | 3 | L | G4 including withdrawal-cascade fixtures |
|
|
144
|
+
|
|
145
|
+
---
|
|
146
|
+
|
|
147
|
+
## Lane J — Credibility evaluation
|
|
148
|
+
|
|
149
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
150
|
+
|---|---|---|---|---|---|---|
|
|
151
|
+
| [ ] | T-J1 | **Agent-level benchmark design**: tasks where a retracted or corrected fact changes the right action; metrics = harmful-action rate, unnecessary-ask rate, cost | — | 0 | M | Protocol and scoring pre-registered before runs |
|
|
152
|
+
| [ ] | T-J2 | Baseline comparisons as *evaluation baselines only*: LWW (free) plus one or two of Mem0 / Zep-Graphiti / Letta, each configured with the same low-cost model, on a REVISE-STREAM subset and T-J1 | T-J1, T-E6 | 2 | L | Cost estimate approved before running; results published including losses |
|
|
153
|
+
| [ ] | T-J3 | Package REVISE-STREAM as a standalone benchmark with its own README and scorer | T-0.3 | 1 | M | `pip install` and run in a fresh env |
|
|
154
|
+
| [ ] | T-J4 | Real-workload load week (declared targets) | T-E5, T-C2 | 3 | M | G2 targets met |
|
|
155
|
+
| [ ] | **G-A** | Agent-level gate | T-J1, T-J2 | 3 | — | Per PROPOSAL §4.3 |
|
|
156
|
+
| [ ] | **G-X** | Extraction gate | T-G2, T-G4 | 3 | — | Per PROPOSAL §4.3 |
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
160
|
+
## Lane K — Release and community
|
|
161
|
+
|
|
162
|
+
| ✓ | ID | Task | Deps | Wave | Size | Acceptance |
|
|
163
|
+
|---|---|---|---|---|---|---|
|
|
164
|
+
| [ ] | T-K1 | `pyproject.toml` for **`palimem`**, extras (`openai-compat`, `anthropic`, `mcp`, `dev` incl. `jsonschema`, `pytest`), CI matrix 3.11–3.13, trusted-publishing release workflow | T-0.2 | 1 | S | Test release to TestPyPI |
|
|
165
|
+
| [ ] | T-K2 | README with the six obligations (value, non-goals, LongMemEval note, when/when-not, regime table, quickstart) | T-F1 | 2 | M | Review checklist against design §Open-source delivery |
|
|
166
|
+
| [ ] | T-K3 | Docs site and examples/notebooks (retraction, bitemporal, inquiry, MCP) | T-F3 | 2 | M | Examples run in CI |
|
|
167
|
+
| [ ] | T-K4 | `CONTRIBUTING`, `GOVERNANCE`, RFC process, issue templates | T-A6 | 1 | S | Files present |
|
|
168
|
+
| [ ] | T-K5 | Fresh-environment gate: `pip install` → quickstart on frozen Setting 1 → deposited numbers | T-E3, T-K1 | 3 | S | Part of G2 |
|
|
169
|
+
|
|
170
|
+
---
|
|
171
|
+
|
|
172
|
+
## Release train and gates
|
|
173
|
+
|
|
174
|
+
| Release | Gate(s) | Tasks that must be done |
|
|
175
|
+
|---|---|---|
|
|
176
|
+
| 0.1 | differential CI, G0-lite | 0.1–0.3, A1, A2, A3, A6, B1, C1, E1–E3, F1, K1, K4 |
|
|
177
|
+
| 0.2 | G1 | + B2–B8, C2, C3, D1–D3, D5, E4, F2 |
|
|
178
|
+
| 0.3 | G1 (full) | + C4–C6, C9 |
|
|
179
|
+
| 0.4 | G-S, G-X | + F3–F7, G1–G6, H1–H4, C7, C8, D4 |
|
|
180
|
+
| 1.0 | G2, G-A | + E5, J1, J2, J4, K2, K3, K5 |
|
|
181
|
+
| 1.x+ | G3, G4 | Lane I |
|
|
182
|
+
|
|
183
|
+
---
|
|
184
|
+
|
|
185
|
+
## Suggested parallel plan (what could run concurrently, if approved)
|
|
186
|
+
|
|
187
|
+
| Worktree | Starts with | Notes |
|
|
188
|
+
|---|---|---|
|
|
189
|
+
| `lane-e-harness` | T-E1, T-E2, T-E6 | Must land first; becomes the merge gate for the others |
|
|
190
|
+
| `lane-a-spec` | T-A1, T-A2, T-A6 | Human-in-the-loop: the decisions need the author's input |
|
|
191
|
+
| `lane-h-security` | T-H1 | No dependencies |
|
|
192
|
+
| `lane-i-research` | T-I1 | Time-boxed; outcome may reshape lanes B and C |
|
|
193
|
+
| `lane-j-eval` | T-J1 | No dependencies |
|
|
194
|
+
| `lane-b-kernel`, `lane-c-store`, `lane-d-admission`, `lane-f-agent`, `lane-g-extract` | After T-A3 types land | Run against stubs and the draft fixtures |
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.24"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "palimem"
|
|
7
|
+
version = "0.0.1"
|
|
8
|
+
description = "Justified memory for LLM agents: evidence-pinned beliefs, retraction that propagates, uncertainty kept rather than guessed. Pre-alpha."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Christian Leiva Beltran" }]
|
|
13
|
+
keywords = ["llm", "agents", "memory", "belief-revision", "truth-maintenance"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [] # core is standard library only
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
bedrock = ["boto3>=1.34"]
|
|
24
|
+
openai-compat = ["openai>=1.0"]
|
|
25
|
+
anthropic = ["anthropic>=0.40"]
|
|
26
|
+
mcp = ["mcp>=1.0"]
|
|
27
|
+
dev = ["pytest>=8", "jsonschema>=4", "ruff>=0.5"]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Repository = "https://github.com/chleiva/palimem"
|
|
31
|
+
Study = "https://doi.org/10.5281/zenodo.23127764"
|
|
32
|
+
|
|
33
|
+
[tool.hatch.build.targets.wheel]
|
|
34
|
+
packages = ["src/palimem"]
|
|
35
|
+
|
|
36
|
+
[tool.pytest.ini_options]
|
|
37
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Fail if staged (or, with --all, tracked) files look like they contain credentials.
|
|
3
|
+
set -euo pipefail
|
|
4
|
+
if [ "${1:-}" = "--all" ]; then files=$(git ls-files); else files=$(git diff --cached --name-only --diff-filter=ACM); fi
|
|
5
|
+
[ -z "$files" ] && exit 0
|
|
6
|
+
pat='(AKIA[0-9A-Z]{16}|sk-[A-Za-z0-9_-]{20,}|ASIA[0-9A-Z]{16}|pypi-[A-Za-z0-9_-]{30,}|npm_[A-Za-z0-9]{30,}|ghp_[A-Za-z0-9]{30,}|-----BEGIN [A-Z ]*PRIVATE KEY-----)'
|
|
7
|
+
bad=0
|
|
8
|
+
for f in $files; do
|
|
9
|
+
[ -f "$f" ] || continue
|
|
10
|
+
if grep -IEn "$pat" "$f" >/dev/null 2>&1; then echo "possible secret in $f"; bad=1; fi
|
|
11
|
+
done
|
|
12
|
+
exit $bad
|