pyboltzmann 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of pyboltzmann might be problematic. Click here for more details.
- boltzmann/__init__.py +160 -0
- boltzmann/blocks/__init__.py +50 -0
- boltzmann/blocks/base.py +215 -0
- boltzmann/blocks/canonical.py +76 -0
- boltzmann/blocks/episodic.py +52 -0
- boltzmann/blocks/memory_type.py +53 -0
- boltzmann/blocks/procedural.py +62 -0
- boltzmann/blocks/provenance.py +305 -0
- boltzmann/blocks/semantic.py +79 -0
- boltzmann/brain.py +1877 -0
- boltzmann/conformance/__init__.py +23 -0
- boltzmann/conformance/golden.py +48 -0
- boltzmann/conformance/suite.py +438 -0
- boltzmann/conformance/vectors/__init__.py +1 -0
- boltzmann/conformance/vectors/block_ids.json +185 -0
- boltzmann/conformance/vectors/inclusion_proofs.json +158 -0
- boltzmann/conformance/vectors/merkle_roots.json +156 -0
- boltzmann/conformance/vectors/serialization.json +72 -0
- boltzmann/constants.py +22 -0
- boltzmann/distribution/__init__.py +37 -0
- boltzmann/distribution/layers.py +203 -0
- boltzmann/distribution/local.py +164 -0
- boltzmann/distribution/manifest.py +391 -0
- boltzmann/distribution/media_types.py +112 -0
- boltzmann/distribution/oras_client.py +345 -0
- boltzmann/distribution/registry.py +111 -0
- boltzmann/exceptions.py +147 -0
- boltzmann/identity/__init__.py +32 -0
- boltzmann/identity/digest.py +185 -0
- boltzmann/identity/hashing.py +88 -0
- boltzmann/identity/serialization.py +154 -0
- boltzmann/identity/time.py +68 -0
- boltzmann/indices/__init__.py +10 -0
- boltzmann/indices/base.py +164 -0
- boltzmann/ingest/__init__.py +58 -0
- boltzmann/ingest/commit.py +49 -0
- boltzmann/ingest/pipelines.py +113 -0
- boltzmann/ingest/proposer.py +119 -0
- boltzmann/ingest/register.py +71 -0
- boltzmann/ingest/schema.py +195 -0
- boltzmann/ingest/task.py +90 -0
- boltzmann/ingest/validation.py +234 -0
- boltzmann/ingest/validators.py +472 -0
- boltzmann/merkle/__init__.py +28 -0
- boltzmann/merkle/diff.py +86 -0
- boltzmann/merkle/layout.py +60 -0
- boltzmann/merkle/proof.py +113 -0
- boltzmann/merkle/tree.py +227 -0
- boltzmann/module/__init__.py +16 -0
- boltzmann/module/composition.py +255 -0
- boltzmann/module/ledger.py +173 -0
- boltzmann/module/module.py +245 -0
- boltzmann/module/snapshot.py +219 -0
- boltzmann/protocol/__init__.py +27 -0
- boltzmann/protocol/operations.py +472 -0
- boltzmann/py.typed +0 -0
- boltzmann/query/__init__.py +26 -0
- boltzmann/query/evidence.py +133 -0
- boltzmann/query/planner.py +49 -0
- boltzmann/query/request.py +124 -0
- boltzmann/query/scan.py +325 -0
- boltzmann/retention/__init__.py +41 -0
- boltzmann/retention/cascade.py +192 -0
- boltzmann/retention/policy.py +122 -0
- boltzmann/retention/reachability.py +154 -0
- boltzmann/retention/requests.py +254 -0
- boltzmann/store/__init__.py +13 -0
- boltzmann/store/base.py +302 -0
- boltzmann/store/memory.py +138 -0
- boltzmann/store/oci_layout.py +294 -0
- boltzmann/utils/__init__.py +7 -0
- boltzmann/utils/logging.py +41 -0
- pyboltzmann-0.1.0.dist-info/METADATA +185 -0
- pyboltzmann-0.1.0.dist-info/RECORD +76 -0
- pyboltzmann-0.1.0.dist-info/WHEEL +4 -0
- pyboltzmann-0.1.0.dist-info/licenses/LICENSE +21 -0
boltzmann/__init__.py
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Boltzmann: an SDK for the Boltzmann Protocol.
|
|
2
|
+
|
|
3
|
+
The brain conserves, validates, and retrieves knowledge. An external LLM processes, contextualizes,
|
|
4
|
+
and uses it.
|
|
5
|
+
|
|
6
|
+
Knowledge is stored as typed, content-addressed blocks organized into five memory modules --
|
|
7
|
+
canonical, episodic, semantic, procedural, and provenance. Each module bundles its blocks, a Merkle
|
|
8
|
+
DAG that pins the exact composition of a version, and the indices needed to query it.
|
|
9
|
+
|
|
10
|
+
**This package is a protocol SDK, not a brain.** It provides three things:
|
|
11
|
+
|
|
12
|
+
1. **Identity and verification**, implemented, because every conforming client must compute the same
|
|
13
|
+
values: the canonical serialization, ``block_id``, the Merkle root, inclusion proofs.
|
|
14
|
+
2. **The types the protocol exchanges**, implemented, because they are wire formats: block schemas,
|
|
15
|
+
``ProcessingTask``, ``boltzmann.candidates/v1``, ``EvidenceBundle``, ``Query``, the OCI manifest.
|
|
16
|
+
3. **The interfaces an implementation satisfies**, declared: ``BrainReader``, ``BrainWriter``,
|
|
17
|
+
``BrainRetention``, ``BrainDistribution``, plus ``BlockStore``, ``Index``, ``QueryPlanner``,
|
|
18
|
+
``Validator``, ``CandidateProposer``, ``RegistryClient``, ``MerkleLayout``,
|
|
19
|
+
``NormalizationPipeline``.
|
|
20
|
+
|
|
21
|
+
Ingestion, query, retention, and distribution are **not implemented here**. Where the paper leaves
|
|
22
|
+
something to the implementation -- ranking, fusion, index engines, cascade depth, retention
|
|
23
|
+
thresholds -- so does this SDK.
|
|
24
|
+
|
|
25
|
+
It embeds no language model. Interpretation enters through ``CandidateProposer`` and nowhere else.
|
|
26
|
+
|
|
27
|
+
Reference: *Boltzmann Brain: A Versioned, Distributable, and Model-Agnostic Knowledge Architecture*
|
|
28
|
+
(Gaussia, 2026).
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from boltzmann.blocks import (
|
|
32
|
+
Actor,
|
|
33
|
+
ActorKind,
|
|
34
|
+
Block,
|
|
35
|
+
CanonicalBlock,
|
|
36
|
+
EpisodicBlock,
|
|
37
|
+
MemoryType,
|
|
38
|
+
NormalizedView,
|
|
39
|
+
ProceduralBlock,
|
|
40
|
+
Producer,
|
|
41
|
+
ProvenanceBlock,
|
|
42
|
+
Relation,
|
|
43
|
+
RemovalMechanism,
|
|
44
|
+
SemanticBlock,
|
|
45
|
+
SemanticKind,
|
|
46
|
+
Step,
|
|
47
|
+
)
|
|
48
|
+
from boltzmann.brain import Brain, BrainState
|
|
49
|
+
from boltzmann.exceptions import BoltzmannError
|
|
50
|
+
from boltzmann.identity import BlockId, MerkleRoot, OciDigest
|
|
51
|
+
from boltzmann.identity.time import utc_timestamp
|
|
52
|
+
from boltzmann.indices import Index, IndexKind
|
|
53
|
+
from boltzmann.ingest import (
|
|
54
|
+
Candidate,
|
|
55
|
+
CandidateProposer,
|
|
56
|
+
CandidateSet,
|
|
57
|
+
CommitResult,
|
|
58
|
+
ProcessingTask,
|
|
59
|
+
RegistrationRequest,
|
|
60
|
+
RegistrationResult,
|
|
61
|
+
ValidationReport,
|
|
62
|
+
ValidationStatus,
|
|
63
|
+
Validator,
|
|
64
|
+
)
|
|
65
|
+
from boltzmann.merkle import InclusionProof, MerkleLayout, MerkleTree
|
|
66
|
+
from boltzmann.module import Composition, Ledger, Module, ModuleRef, Snapshot
|
|
67
|
+
from boltzmann.protocol import (
|
|
68
|
+
PROTOCOL_VERSION,
|
|
69
|
+
BoltzmannProtocol,
|
|
70
|
+
BrainDistribution,
|
|
71
|
+
BrainReader,
|
|
72
|
+
BrainRetention,
|
|
73
|
+
BrainWriter,
|
|
74
|
+
)
|
|
75
|
+
from boltzmann.query import EvidenceBundle, Match, Query, QueryFilters, QueryHints, QueryPlanner, RetrievalMode
|
|
76
|
+
from boltzmann.retention import (
|
|
77
|
+
CascadePlan,
|
|
78
|
+
DropRequest,
|
|
79
|
+
DropResult,
|
|
80
|
+
ProducerDropRequest,
|
|
81
|
+
PruneReport,
|
|
82
|
+
RedactionResult,
|
|
83
|
+
ResolvabilityReport,
|
|
84
|
+
RetentionPolicy,
|
|
85
|
+
SupersessionResult,
|
|
86
|
+
)
|
|
87
|
+
from boltzmann.store import BlockStore, MemoryBlockStore, OciLayoutStore
|
|
88
|
+
|
|
89
|
+
__version__ = "0.1.0"
|
|
90
|
+
|
|
91
|
+
__all__ = [
|
|
92
|
+
"PROTOCOL_VERSION",
|
|
93
|
+
"Actor",
|
|
94
|
+
"ActorKind",
|
|
95
|
+
"Block",
|
|
96
|
+
"BlockId",
|
|
97
|
+
"BlockStore",
|
|
98
|
+
"BoltzmannError",
|
|
99
|
+
"BoltzmannProtocol",
|
|
100
|
+
"Brain",
|
|
101
|
+
"BrainDistribution",
|
|
102
|
+
"BrainReader",
|
|
103
|
+
"BrainRetention",
|
|
104
|
+
"BrainState",
|
|
105
|
+
"BrainWriter",
|
|
106
|
+
"Candidate",
|
|
107
|
+
"CascadePlan",
|
|
108
|
+
"CandidateProposer",
|
|
109
|
+
"CandidateSet",
|
|
110
|
+
"CanonicalBlock",
|
|
111
|
+
"CommitResult",
|
|
112
|
+
"Composition",
|
|
113
|
+
"DropRequest",
|
|
114
|
+
"DropResult",
|
|
115
|
+
"ProducerDropRequest",
|
|
116
|
+
"PruneReport",
|
|
117
|
+
"RedactionResult",
|
|
118
|
+
"ResolvabilityReport",
|
|
119
|
+
"SupersessionResult",
|
|
120
|
+
"EpisodicBlock",
|
|
121
|
+
"EvidenceBundle",
|
|
122
|
+
"InclusionProof",
|
|
123
|
+
"Ledger",
|
|
124
|
+
"Index",
|
|
125
|
+
"IndexKind",
|
|
126
|
+
"Match",
|
|
127
|
+
"MemoryBlockStore",
|
|
128
|
+
"MemoryType",
|
|
129
|
+
"MerkleLayout",
|
|
130
|
+
"MerkleRoot",
|
|
131
|
+
"MerkleTree",
|
|
132
|
+
"Module",
|
|
133
|
+
"ModuleRef",
|
|
134
|
+
"NormalizedView",
|
|
135
|
+
"OciDigest",
|
|
136
|
+
"OciLayoutStore",
|
|
137
|
+
"ProceduralBlock",
|
|
138
|
+
"ProcessingTask",
|
|
139
|
+
"Producer",
|
|
140
|
+
"ProvenanceBlock",
|
|
141
|
+
"Query",
|
|
142
|
+
"QueryFilters",
|
|
143
|
+
"QueryHints",
|
|
144
|
+
"QueryPlanner",
|
|
145
|
+
"RegistrationRequest",
|
|
146
|
+
"RegistrationResult",
|
|
147
|
+
"Relation",
|
|
148
|
+
"RemovalMechanism",
|
|
149
|
+
"RetentionPolicy",
|
|
150
|
+
"RetrievalMode",
|
|
151
|
+
"SemanticBlock",
|
|
152
|
+
"SemanticKind",
|
|
153
|
+
"Snapshot",
|
|
154
|
+
"Step",
|
|
155
|
+
"ValidationReport",
|
|
156
|
+
"ValidationStatus",
|
|
157
|
+
"Validator",
|
|
158
|
+
"__version__",
|
|
159
|
+
"utc_timestamp",
|
|
160
|
+
]
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Typed knowledge blocks: the durable unit of the protocol."""
|
|
2
|
+
|
|
3
|
+
from boltzmann.blocks.base import ENVELOPE_KEYS, Block
|
|
4
|
+
from boltzmann.blocks.canonical import CanonicalBlock, NormalizedView
|
|
5
|
+
from boltzmann.blocks.episodic import EpisodicBlock
|
|
6
|
+
from boltzmann.blocks.memory_type import MemoryType
|
|
7
|
+
from boltzmann.blocks.procedural import ProceduralBlock, Step
|
|
8
|
+
from boltzmann.blocks.provenance import (
|
|
9
|
+
Actor,
|
|
10
|
+
ActorKind,
|
|
11
|
+
DemotionRecord,
|
|
12
|
+
DerivationRecord,
|
|
13
|
+
NormalizationRecord,
|
|
14
|
+
Producer,
|
|
15
|
+
ProducerKind,
|
|
16
|
+
ProvenanceBlock,
|
|
17
|
+
ProvenanceEntry,
|
|
18
|
+
RegistrationRecord,
|
|
19
|
+
RemovalMechanism,
|
|
20
|
+
RemovalRecord,
|
|
21
|
+
SupersessionRecord,
|
|
22
|
+
)
|
|
23
|
+
from boltzmann.blocks.semantic import Relation, SemanticBlock, SemanticKind
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"ENVELOPE_KEYS",
|
|
27
|
+
"Actor",
|
|
28
|
+
"ActorKind",
|
|
29
|
+
"Block",
|
|
30
|
+
"CanonicalBlock",
|
|
31
|
+
"DemotionRecord",
|
|
32
|
+
"DerivationRecord",
|
|
33
|
+
"EpisodicBlock",
|
|
34
|
+
"MemoryType",
|
|
35
|
+
"NormalizationRecord",
|
|
36
|
+
"NormalizedView",
|
|
37
|
+
"ProceduralBlock",
|
|
38
|
+
"Producer",
|
|
39
|
+
"ProducerKind",
|
|
40
|
+
"ProvenanceBlock",
|
|
41
|
+
"ProvenanceEntry",
|
|
42
|
+
"RegistrationRecord",
|
|
43
|
+
"Relation",
|
|
44
|
+
"RemovalMechanism",
|
|
45
|
+
"RemovalRecord",
|
|
46
|
+
"SemanticBlock",
|
|
47
|
+
"SemanticKind",
|
|
48
|
+
"Step",
|
|
49
|
+
"SupersessionRecord",
|
|
50
|
+
]
|
boltzmann/blocks/base.py
ADDED
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""The block envelope and the computation of ``block_id`` (paper Section 6.1).
|
|
2
|
+
|
|
3
|
+
A block is a logical unit of knowledge with a deterministic serialization:
|
|
4
|
+
|
|
5
|
+
.. math::
|
|
6
|
+
|
|
7
|
+
block\\_id = SHA\\text{-}256(canonical\\_serialization(block))
|
|
8
|
+
|
|
9
|
+
What gets hashed is the **envelope**, not the payload alone, so that the memory
|
|
10
|
+
type and the schema version are bound into the identity:
|
|
11
|
+
|
|
12
|
+
.. code-block:: json
|
|
13
|
+
|
|
14
|
+
{
|
|
15
|
+
"boltzmann": 1,
|
|
16
|
+
"memory_type": "semantic",
|
|
17
|
+
"payload": {},
|
|
18
|
+
"schema_version": 1,
|
|
19
|
+
"serialization": "jcs/1"
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
Everything derived or mutable stays outside: physical location, index state, and
|
|
23
|
+
the time a block happened to be registered are not part of what the block *is*.
|
|
24
|
+
|
|
25
|
+
Two conventions keep identity unambiguous:
|
|
26
|
+
|
|
27
|
+
* An absent optional field is dropped from the payload rather than serialized as
|
|
28
|
+
``null``, so ``{"a": 1}`` and ``{"a": 1, "b": null}`` are the same block.
|
|
29
|
+
Optional collections therefore default to ``None``, never to ``[]``.
|
|
30
|
+
* Blocks are frozen. Correcting a block means creating a new one; the previous
|
|
31
|
+
one may be kept for history, audit, or rollback.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import json
|
|
37
|
+
from abc import ABC
|
|
38
|
+
from typing import Any, ClassVar, Self
|
|
39
|
+
|
|
40
|
+
from pydantic import BaseModel, ConfigDict, model_validator
|
|
41
|
+
|
|
42
|
+
from boltzmann.blocks.memory_type import MemoryType
|
|
43
|
+
from boltzmann.constants import PROTOCOL_VERSION
|
|
44
|
+
from boltzmann.exceptions import BlockIntegrityError, BlockSchemaError
|
|
45
|
+
from boltzmann.identity.digest import BlockId
|
|
46
|
+
from boltzmann.identity.serialization import SERIALIZATION_ID, canonicalize, reject_non_deterministic
|
|
47
|
+
|
|
48
|
+
ENVELOPE_KEYS = frozenset({"boltzmann", "memory_type", "payload", "schema_version", "serialization"})
|
|
49
|
+
"""The exact set of keys a block envelope carries."""
|
|
50
|
+
|
|
51
|
+
_REGISTRY: dict[tuple[MemoryType, int], type[Block]] = {}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class Block(BaseModel, ABC):
|
|
55
|
+
"""
|
|
56
|
+
Base class for the five typed knowledge blocks.
|
|
57
|
+
|
|
58
|
+
A subclass declares its memory type and schema version as class variables and
|
|
59
|
+
its fields as the payload. Identity, serialization, and decoding come for free.
|
|
60
|
+
|
|
61
|
+
Attributes:
|
|
62
|
+
MEMORY_TYPE (MemoryType): Which memory module this block belongs to.
|
|
63
|
+
SCHEMA_VERSION (int): Version of this block's payload schema.
|
|
64
|
+
SERIALIZATION (str): Canonical serialization used to compute ``block_id``.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
model_config = ConfigDict(frozen=True, extra="forbid")
|
|
68
|
+
|
|
69
|
+
MEMORY_TYPE: ClassVar[MemoryType]
|
|
70
|
+
SCHEMA_VERSION: ClassVar[int] = 1
|
|
71
|
+
SERIALIZATION: ClassVar[str] = SERIALIZATION_ID
|
|
72
|
+
|
|
73
|
+
def __init_subclass__(cls, **kwargs: Any) -> None:
|
|
74
|
+
"""Register concrete subclasses so stored bytes can be decoded back."""
|
|
75
|
+
super().__init_subclass__(**kwargs)
|
|
76
|
+
memory_type = getattr(cls, "MEMORY_TYPE", None)
|
|
77
|
+
if memory_type is None or ABC in cls.__bases__:
|
|
78
|
+
return
|
|
79
|
+
key = (memory_type, cls.SCHEMA_VERSION)
|
|
80
|
+
registered = _REGISTRY.get(key)
|
|
81
|
+
if registered is not None and registered is not cls:
|
|
82
|
+
raise BlockSchemaError(
|
|
83
|
+
f"{cls.__name__} claims ({memory_type}, schema_version={cls.SCHEMA_VERSION}), "
|
|
84
|
+
f"already held by {registered.__name__}"
|
|
85
|
+
)
|
|
86
|
+
_REGISTRY[key] = cls
|
|
87
|
+
|
|
88
|
+
@model_validator(mode="after")
|
|
89
|
+
def _reject_non_deterministic_payload(self) -> Self:
|
|
90
|
+
"""Fail at construction, not at hashing, when a value has no canonical form."""
|
|
91
|
+
reject_non_deterministic(self.payload())
|
|
92
|
+
return self
|
|
93
|
+
|
|
94
|
+
# --- Identity -------------------------------------------------------------
|
|
95
|
+
|
|
96
|
+
def payload(self) -> dict[str, Any]:
|
|
97
|
+
"""
|
|
98
|
+
The block's payload as a JSON-shaped mapping.
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
dict[str, Any]: The payload, with absent optional fields dropped.
|
|
102
|
+
"""
|
|
103
|
+
return self.model_dump(mode="json", exclude_none=True)
|
|
104
|
+
|
|
105
|
+
def envelope(self) -> dict[str, Any]:
|
|
106
|
+
"""
|
|
107
|
+
The full envelope that ``block_id`` is computed over.
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
dict[str, Any]: The envelope mapping.
|
|
111
|
+
"""
|
|
112
|
+
return {
|
|
113
|
+
"boltzmann": PROTOCOL_VERSION,
|
|
114
|
+
"memory_type": self.MEMORY_TYPE.value,
|
|
115
|
+
"payload": self.payload(),
|
|
116
|
+
"schema_version": self.SCHEMA_VERSION,
|
|
117
|
+
"serialization": self.SERIALIZATION,
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
def canonical_bytes(self) -> bytes:
|
|
121
|
+
"""
|
|
122
|
+
The exact bytes that are hashed and stored.
|
|
123
|
+
|
|
124
|
+
Returns:
|
|
125
|
+
bytes: The canonically serialized envelope.
|
|
126
|
+
"""
|
|
127
|
+
return canonicalize(self.envelope(), self.SERIALIZATION)
|
|
128
|
+
|
|
129
|
+
@property
|
|
130
|
+
def block_id(self) -> BlockId:
|
|
131
|
+
"""The block's content-addressed identity."""
|
|
132
|
+
return BlockId.of(self.canonical_bytes())
|
|
133
|
+
|
|
134
|
+
# --- Decoding -------------------------------------------------------------
|
|
135
|
+
|
|
136
|
+
@staticmethod
|
|
137
|
+
def decode(data: bytes) -> Block:
|
|
138
|
+
"""
|
|
139
|
+
Decode stored bytes back into a typed block.
|
|
140
|
+
|
|
141
|
+
Beyond parsing, this checks that ``data`` is already in canonical form: if
|
|
142
|
+
re-serializing the decoded block does not reproduce ``data`` byte for byte,
|
|
143
|
+
the stored bytes would hash to a different ``block_id`` than they claim, so
|
|
144
|
+
they are rejected rather than silently normalized.
|
|
145
|
+
|
|
146
|
+
Args:
|
|
147
|
+
data (bytes): Canonically serialized envelope bytes.
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
Block: The decoded block.
|
|
151
|
+
|
|
152
|
+
Raises:
|
|
153
|
+
BlockSchemaError: If the envelope is malformed or its type is unknown.
|
|
154
|
+
BlockIntegrityError: If ``data`` is not in canonical form.
|
|
155
|
+
"""
|
|
156
|
+
try:
|
|
157
|
+
envelope = json.loads(data)
|
|
158
|
+
except json.JSONDecodeError as error:
|
|
159
|
+
raise BlockSchemaError(f"block envelope is not valid JSON: {error}") from error
|
|
160
|
+
|
|
161
|
+
if not isinstance(envelope, dict):
|
|
162
|
+
raise BlockSchemaError(f"block envelope must be an object, got {type(envelope).__name__}")
|
|
163
|
+
|
|
164
|
+
missing = ENVELOPE_KEYS - envelope.keys()
|
|
165
|
+
unexpected = envelope.keys() - ENVELOPE_KEYS
|
|
166
|
+
if missing or unexpected:
|
|
167
|
+
details = []
|
|
168
|
+
if missing:
|
|
169
|
+
details.append(f"missing {sorted(missing)}")
|
|
170
|
+
if unexpected:
|
|
171
|
+
details.append(f"unexpected {sorted(unexpected)}")
|
|
172
|
+
raise BlockSchemaError(f"malformed block envelope: {', '.join(details)}")
|
|
173
|
+
|
|
174
|
+
if envelope["boltzmann"] != PROTOCOL_VERSION:
|
|
175
|
+
raise BlockSchemaError(
|
|
176
|
+
f"block declares protocol version {envelope['boltzmann']!r}, this client implements {PROTOCOL_VERSION}"
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
try:
|
|
180
|
+
memory_type = MemoryType(envelope["memory_type"])
|
|
181
|
+
except ValueError as error:
|
|
182
|
+
raise BlockSchemaError(f"unknown memory type {envelope['memory_type']!r}") from error
|
|
183
|
+
|
|
184
|
+
schema_version = envelope["schema_version"]
|
|
185
|
+
block_class = _REGISTRY.get((memory_type, schema_version))
|
|
186
|
+
if block_class is None:
|
|
187
|
+
known = sorted(version for kind, version in _REGISTRY if kind is memory_type)
|
|
188
|
+
raise BlockSchemaError(
|
|
189
|
+
f"no schema registered for {memory_type} version {schema_version!r}; this client knows {known}"
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
if envelope["serialization"] != block_class.SERIALIZATION:
|
|
193
|
+
raise BlockSchemaError(
|
|
194
|
+
f"block declares serialization {envelope['serialization']!r}, "
|
|
195
|
+
f"{block_class.__name__} uses {block_class.SERIALIZATION!r}"
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
block = block_class.model_validate(envelope["payload"])
|
|
199
|
+
if block.canonical_bytes() != data:
|
|
200
|
+
raise BlockIntegrityError(
|
|
201
|
+
f"stored bytes for a {memory_type} block are not in canonical "
|
|
202
|
+
f"{block_class.SERIALIZATION} form, so they do not hash to the block_id they claim"
|
|
203
|
+
)
|
|
204
|
+
return block
|
|
205
|
+
|
|
206
|
+
@staticmethod
|
|
207
|
+
def registry() -> dict[tuple[MemoryType, int], type[Block]]:
|
|
208
|
+
"""
|
|
209
|
+
The registered block schemas.
|
|
210
|
+
|
|
211
|
+
Returns:
|
|
212
|
+
dict[tuple[MemoryType, int], type[Block]]: Map of memory type and
|
|
213
|
+
schema version to the class that implements it.
|
|
214
|
+
"""
|
|
215
|
+
return dict(_REGISTRY)
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Canonical memory: what material was observed (paper Section 5).
|
|
2
|
+
|
|
3
|
+
The canonical module preserves observed evidence, not truth claims. A canonical
|
|
4
|
+
block asserts that a source was incorporated and that it *contains* certain
|
|
5
|
+
bytes; it does not declare those bytes correct. Because every semantic and
|
|
6
|
+
procedural interpretation cites canonical evidence through provenance, this
|
|
7
|
+
module is the root of re-derivation.
|
|
8
|
+
|
|
9
|
+
**A canonical block is a pure statement about observed bytes.** Its identity
|
|
10
|
+
depends on nothing else, which is what makes re-registering an identical original
|
|
11
|
+
a genuine no-op: two actors ingesting the same PDF compute the same ``block_id``
|
|
12
|
+
and the second registration adds no block. Everything historical or
|
|
13
|
+
actor-dependent -- who incorporated it, when, from where, under what license or
|
|
14
|
+
retention policy, and which earlier edition it supersedes -- is recorded as a
|
|
15
|
+
provenance edge instead. See :mod:`boltzmann.blocks.provenance`.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from typing import ClassVar
|
|
21
|
+
|
|
22
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
23
|
+
|
|
24
|
+
from boltzmann.blocks.base import Block
|
|
25
|
+
from boltzmann.blocks.memory_type import MemoryType
|
|
26
|
+
from boltzmann.identity.digest import OciDigest
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class NormalizedView(BaseModel):
|
|
30
|
+
"""
|
|
31
|
+
A deterministic transform of an original blob.
|
|
32
|
+
|
|
33
|
+
A normalized view is plain text, Markdown, or a structured extract produced by
|
|
34
|
+
a named deterministic pipeline whose identity is recorded in provenance.
|
|
35
|
+
Normalized views live in canonical rather than semantic memory because they are
|
|
36
|
+
still evidence for ingestion, not consolidated knowledge.
|
|
37
|
+
|
|
38
|
+
Attributes:
|
|
39
|
+
blob (OciDigest): Content address of the normalized bytes.
|
|
40
|
+
media_type (str): IANA media type of the normalized bytes.
|
|
41
|
+
size (int): Length of the normalized bytes in bytes.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
model_config = ConfigDict(frozen=True, extra="forbid")
|
|
45
|
+
|
|
46
|
+
blob: OciDigest
|
|
47
|
+
media_type: str = Field(min_length=1)
|
|
48
|
+
size: int = Field(ge=0)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class CanonicalBlock(Block):
|
|
52
|
+
"""
|
|
53
|
+
Evidence that a source was incorporated and preserved.
|
|
54
|
+
|
|
55
|
+
Note which level of identity ``blob`` carries. The observed bytes of a source
|
|
56
|
+
are a transportable file, not a unit of knowledge, so they are addressed by an
|
|
57
|
+
:class:`~boltzmann.identity.digest.OciDigest`; the ``block_id`` of this block is
|
|
58
|
+
the knowledge-level statement *about* those bytes. Read together with
|
|
59
|
+
``media_type`` and ``size``, this block is an OCI descriptor over the evidence
|
|
60
|
+
plus an optional normalized view of it -- which is why publishing a brain is a
|
|
61
|
+
copy rather than a conversion.
|
|
62
|
+
|
|
63
|
+
Attributes:
|
|
64
|
+
blob (OciDigest): Content address of the original bytes as observed.
|
|
65
|
+
media_type (str): IANA media type of the original.
|
|
66
|
+
size (int): Length of the original in bytes.
|
|
67
|
+
normalized_view (NormalizedView | None): Optional deterministic transform
|
|
68
|
+
of the original, addressed by its own hash.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
MEMORY_TYPE: ClassVar[MemoryType] = MemoryType.CANONICAL
|
|
72
|
+
|
|
73
|
+
blob: OciDigest
|
|
74
|
+
media_type: str = Field(min_length=1)
|
|
75
|
+
size: int = Field(ge=0)
|
|
76
|
+
normalized_view: NormalizedView | None = None
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Episodic memory: what happened in a concrete context (paper Section 5).
|
|
2
|
+
|
|
3
|
+
Records events and experiences with time, context, participants, outcome, and
|
|
4
|
+
evidence. This is the only module that stays append-only: corrections generate new
|
|
5
|
+
episodes or supersession relations rather than drops that rewrite the past.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import ClassVar, Self
|
|
11
|
+
|
|
12
|
+
from pydantic import Field, model_validator
|
|
13
|
+
|
|
14
|
+
from boltzmann.blocks.base import Block
|
|
15
|
+
from boltzmann.blocks.memory_type import MemoryType
|
|
16
|
+
from boltzmann.identity.digest import BlockId
|
|
17
|
+
from boltzmann.identity.time import Timestamp, parse_timestamp
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class EpisodicBlock(Block):
|
|
21
|
+
"""
|
|
22
|
+
A concrete experience, situated in time.
|
|
23
|
+
|
|
24
|
+
Attributes:
|
|
25
|
+
summary (str): What happened, in one statement.
|
|
26
|
+
occurred_at (Timestamp): When the episode happened, in canonical UTC form.
|
|
27
|
+
ended_at (Timestamp | None): When it finished, for episodes with duration.
|
|
28
|
+
context (str | None): The setting the episode belongs to, such as a course,
|
|
29
|
+
a project, or a conversation.
|
|
30
|
+
participants (list[str] | None): Who took part.
|
|
31
|
+
outcome (str | None): How it resolved.
|
|
32
|
+
evidence (list[BlockId] | None): Canonical blocks that attest to the episode.
|
|
33
|
+
tags (list[str] | None): Free-form labels for filtering.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
MEMORY_TYPE: ClassVar[MemoryType] = MemoryType.EPISODIC
|
|
37
|
+
|
|
38
|
+
summary: str = Field(min_length=1)
|
|
39
|
+
occurred_at: Timestamp
|
|
40
|
+
ended_at: Timestamp | None = None
|
|
41
|
+
context: str | None = None
|
|
42
|
+
participants: list[str] | None = None
|
|
43
|
+
outcome: str | None = None
|
|
44
|
+
evidence: list[BlockId] | None = None
|
|
45
|
+
tags: list[str] | None = None
|
|
46
|
+
|
|
47
|
+
@model_validator(mode="after")
|
|
48
|
+
def _check_interval(self) -> Self:
|
|
49
|
+
"""An episode cannot end before it started."""
|
|
50
|
+
if self.ended_at is not None and parse_timestamp(self.ended_at) < parse_timestamp(self.occurred_at):
|
|
51
|
+
raise ValueError(f"ended_at {self.ended_at} precedes occurred_at {self.occurred_at}")
|
|
52
|
+
return self
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""The five memory modules (paper Section 5).
|
|
2
|
+
|
|
3
|
+
Each module is named by the question it answers:
|
|
4
|
+
|
|
5
|
+
============ ==========================================================
|
|
6
|
+
canonical What material was observed?
|
|
7
|
+
episodic What happened in a concrete context?
|
|
8
|
+
semantic What general knowledge was consolidated?
|
|
9
|
+
procedural How is a task performed?
|
|
10
|
+
provenance Where did it come from and how was it transformed?
|
|
11
|
+
============ ==========================================================
|
|
12
|
+
|
|
13
|
+
The enum lives in the kernel rather than in the module layer because a block's
|
|
14
|
+
memory type sits inside the hashed envelope: it is part of that block's identity,
|
|
15
|
+
not a property of where the block happens to be stored.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class MemoryType(StrEnum):
|
|
22
|
+
"""A memory module, and therefore the type of a knowledge block."""
|
|
23
|
+
|
|
24
|
+
CANONICAL = "canonical"
|
|
25
|
+
EPISODIC = "episodic"
|
|
26
|
+
SEMANTIC = "semantic"
|
|
27
|
+
PROCEDURAL = "procedural"
|
|
28
|
+
PROVENANCE = "provenance"
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def is_append_only(self) -> bool:
|
|
32
|
+
"""
|
|
33
|
+
Whether the module refuses ``drop``.
|
|
34
|
+
|
|
35
|
+
The episodic module is a chronological record of what happened, so
|
|
36
|
+
corrections append new episodes or supersession relations rather than
|
|
37
|
+
rewriting the past (paper Section 10.3).
|
|
38
|
+
"""
|
|
39
|
+
return self is MemoryType.EPISODIC
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def is_droppable(self) -> bool:
|
|
43
|
+
"""Whether blocks may be excluded from this module's composition."""
|
|
44
|
+
return not self.is_append_only
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def is_derived(self) -> bool:
|
|
48
|
+
"""
|
|
49
|
+
Whether blocks of this type are interpretations that cite canonical evidence.
|
|
50
|
+
|
|
51
|
+
Derived blocks are what a canonical drop cascades to (paper Section 10.3).
|
|
52
|
+
"""
|
|
53
|
+
return self in {MemoryType.SEMANTIC, MemoryType.PROCEDURAL}
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Procedural memory: how a task is performed (paper Section 5).
|
|
2
|
+
|
|
3
|
+
Represents action sequences, conditions, decisions, alternative paths, and success
|
|
4
|
+
criteria -- for example how to obtain the Fourier coefficients of a given function.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import ClassVar
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
12
|
+
|
|
13
|
+
from boltzmann.blocks.base import Block
|
|
14
|
+
from boltzmann.blocks.memory_type import MemoryType
|
|
15
|
+
from boltzmann.identity.digest import BlockId
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Step(BaseModel):
|
|
19
|
+
"""
|
|
20
|
+
One action in a procedure.
|
|
21
|
+
|
|
22
|
+
Attributes:
|
|
23
|
+
action (str): What to do.
|
|
24
|
+
condition (str | None): When this step applies, for branching procedures.
|
|
25
|
+
alternatives (list[str] | None): Other ways to accomplish the same step.
|
|
26
|
+
uses (list[BlockId] | None): Semantic blocks the step relies on, such as the
|
|
27
|
+
formula it applies.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
model_config = ConfigDict(frozen=True, extra="forbid")
|
|
31
|
+
|
|
32
|
+
action: str = Field(min_length=1)
|
|
33
|
+
condition: str | None = None
|
|
34
|
+
alternatives: list[str] | None = None
|
|
35
|
+
uses: list[BlockId] | None = None
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ProceduralBlock(Block):
|
|
39
|
+
"""
|
|
40
|
+
A way of performing a task.
|
|
41
|
+
|
|
42
|
+
Attributes:
|
|
43
|
+
label (str): Short name of the procedure.
|
|
44
|
+
goal (str): What the procedure accomplishes.
|
|
45
|
+
steps (list[Step]): The ordered actions. Order is significant, so it is part
|
|
46
|
+
of the block's identity.
|
|
47
|
+
preconditions (list[str] | None): What must hold before starting.
|
|
48
|
+
success_criteria (list[str] | None): How to tell the procedure worked.
|
|
49
|
+
subject (str | None): Domain the procedure belongs to, for filtering.
|
|
50
|
+
evidence (list[BlockId] | None): Canonical blocks this procedure cites.
|
|
51
|
+
A canonical drop cascades to every block that lists it here.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
MEMORY_TYPE: ClassVar[MemoryType] = MemoryType.PROCEDURAL
|
|
55
|
+
|
|
56
|
+
label: str = Field(min_length=1)
|
|
57
|
+
goal: str = Field(min_length=1)
|
|
58
|
+
steps: list[Step] = Field(min_length=1)
|
|
59
|
+
preconditions: list[str] | None = None
|
|
60
|
+
success_criteria: list[str] | None = None
|
|
61
|
+
subject: str | None = None
|
|
62
|
+
evidence: list[BlockId] | None = None
|