scl-lang 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
scl/__init__.py ADDED
@@ -0,0 +1,26 @@
1
+ """
2
+ SCL — Semantic Compression Language.
3
+
4
+ Grammar: @anchor → verb [key: value, key: value]
5
+
6
+ >>> from scl import SCLRecord, Anchor, Relation, Scope
7
+ >>> from scl.parser import parse_document
8
+ >>> from scl.delta import Delta, SemanticState, DeltaStream
9
+ >>> from scl.gossip import Peer, HierarchicalSwarm
10
+
11
+ Modules:
12
+ types — Anchor, Relation, Scope, SCLRecord, SCLDocument
13
+ parser — text → SCL AST
14
+ emitter — SCL AST → text
15
+ grammar — formal BNF specification
16
+ delta — CRDT-style semantic state deltas
17
+ gossip — hierarchical gossip protocol (100K+ agents)
18
+ eval — executable SCL rules engine
19
+ iblt — Invertible Bloom Lookup Tables for O(d) sync
20
+ braille — compact braille encoding layer
21
+ """
22
+
23
+ from .types import Anchor, Relation, Scope, SCLRecord, SCLDocument
24
+
25
+ __version__ = "0.1.0"
26
+ __all__ = ["Anchor", "Relation", "Scope", "SCLRecord", "SCLDocument"]
@@ -0,0 +1,14 @@
1
+ """
2
+ Braille — compact encoding layer for SCL.
3
+
4
+ Uses the 256 Unicode Braille characters (U+2800–U+28FF) as a bijective
5
+ byte-to-character encoding. Each byte maps to exactly one Braille character.
6
+
7
+ Modules:
8
+ codec — encode/decode bytes ↔ Braille Unicode
9
+ fingerprint — SCLRecord → fixed-width Braille hash
10
+ """
11
+
12
+ from .codec import encode, decode, encode_hex, encode_int
13
+
14
+ __all__ = ["encode", "decode", "encode_hex", "encode_int"]
scl/braille/codec.py ADDED
@@ -0,0 +1,162 @@
1
+ """
2
+ Braille Codec — encode/decode arbitrary bytes as Braille Unicode.
3
+
4
+ The 256 Unicode Braille characters (U+2800 to U+28FF) map 1:1 to byte values
5
+ 0x00–0xFF. Each character's Unicode codepoint offset from U+2800 IS the byte
6
+ value. This gives us a natural, lossless, bijective encoding.
7
+
8
+ Properties:
9
+ - Bijective: every byte sequence has exactly one Braille representation
10
+ - Lossless: decode(encode(x)) == x for all x
11
+ - Fixed density: 1 byte = 1 Braille character = 3 UTF-8 bytes
12
+
13
+ Token efficiency:
14
+ Most LLM tokenizers treat Braille characters as single tokens or small
15
+ multi-byte sequences, making this encoding competitive with or better than
16
+ hex/base64 for transmitting binary data through language models.
17
+
18
+ Visual examples:
19
+ 0x00 → ⠀ (blank) 0x01 → ⠁ 0x41 ('A') → ⡁ 0xFF → ⣿ (all dots)
20
+ """
21
+
22
+ # The base codepoint for Braille patterns
23
+ _BRAILLE_BASE = 0x2800
24
+
25
+ # Pre-compute lookup tables for speed
26
+ _BYTE_TO_BRAILLE: list[str] = [chr(_BRAILLE_BASE + i) for i in range(256)]
27
+ _BRAILLE_TO_BYTE: dict[str, int] = {chr(_BRAILLE_BASE + i): i for i in range(256)}
28
+
29
+
30
+ def encode(data: bytes) -> str:
31
+ """Encode arbitrary bytes as a Braille Unicode string.
32
+
33
+ Args:
34
+ data: Arbitrary byte sequence.
35
+
36
+ Returns:
37
+ String of Braille characters, one per input byte.
38
+
39
+ Examples:
40
+ >>> encode(b'\\x00')
41
+ '⠀'
42
+ >>> encode(b'\\xff')
43
+ '⣿'
44
+ >>> encode(b'hello')
45
+ '⡨⡥⡬⡬⡯'
46
+ """
47
+ return "".join(_BYTE_TO_BRAILLE[b] for b in data)
48
+
49
+
50
+ def decode(braille: str) -> bytes:
51
+ """Decode a Braille Unicode string back to bytes.
52
+
53
+ Args:
54
+ braille: String of Braille characters.
55
+
56
+ Returns:
57
+ Original byte sequence.
58
+
59
+ Raises:
60
+ ValueError: If string contains non-Braille characters.
61
+
62
+ Examples:
63
+ >>> decode('⡨⡥⡬⡬⡯')
64
+ b'hello'
65
+ """
66
+ result = bytearray(len(braille))
67
+ for i, char in enumerate(braille):
68
+ cp = ord(char) - _BRAILLE_BASE
69
+ if cp < 0 or cp > 255:
70
+ raise ValueError(
71
+ f"Character at position {i} is not a Braille pattern: "
72
+ f"U+{ord(char):04X} ({char!r})"
73
+ )
74
+ result[i] = cp
75
+ return bytes(result)
76
+
77
+
78
+ def encode_hex(hex_str: str) -> str:
79
+ """Encode a hex string as Braille.
80
+
81
+ Args:
82
+ hex_str: Hex string (with or without '0x' prefix, spaces allowed).
83
+
84
+ Returns:
85
+ Braille-encoded string.
86
+
87
+ Examples:
88
+ >>> encode_hex('deadbeef')
89
+ '⣞⣭⣾⣯'
90
+ """
91
+ hex_str = hex_str.replace(" ", "").replace("0x", "").replace("0X", "")
92
+ return encode(bytes.fromhex(hex_str))
93
+
94
+
95
+ def encode_int(n: int, width: int = 4) -> str:
96
+ """Encode an integer as fixed-width Braille (big-endian).
97
+
98
+ Args:
99
+ n: Non-negative integer.
100
+ width: Number of bytes (= number of Braille characters).
101
+
102
+ Returns:
103
+ Fixed-width Braille string.
104
+
105
+ Raises:
106
+ ValueError: If n is negative or doesn't fit in width bytes.
107
+
108
+ Examples:
109
+ >>> encode_int(0, width=2)
110
+ '⠀⠀'
111
+ >>> encode_int(255, width=1)
112
+ '⣿'
113
+ >>> encode_int(65535, width=2)
114
+ '⣿⣿'
115
+ """
116
+ if n < 0:
117
+ raise ValueError(f"Cannot encode negative integer: {n}")
118
+ max_val = (1 << (width * 8)) - 1
119
+ if n > max_val:
120
+ raise ValueError(
121
+ f"Integer {n} does not fit in {width} bytes (max {max_val})"
122
+ )
123
+ return encode(n.to_bytes(width, byteorder="big"))
124
+
125
+
126
+ def decode_int(braille: str) -> int:
127
+ """Decode a Braille string back to an integer (big-endian).
128
+
129
+ Args:
130
+ braille: Braille-encoded integer.
131
+
132
+ Returns:
133
+ The decoded integer.
134
+ """
135
+ return int.from_bytes(decode(braille), byteorder="big")
136
+
137
+
138
+ def is_braille(char: str) -> bool:
139
+ """Check if a character is a Braille pattern (U+2800–U+28FF)."""
140
+ cp = ord(char)
141
+ return _BRAILLE_BASE <= cp <= _BRAILLE_BASE + 255
142
+
143
+
144
+ def braille_len(braille: str) -> int:
145
+ """Return the number of bytes encoded in a Braille string.
146
+
147
+ Same as len(braille) since the encoding is 1:1, but explicit
148
+ about the semantic meaning.
149
+ """
150
+ return len(braille)
151
+
152
+
153
+ def to_hex(braille: str) -> str:
154
+ """Convert Braille string to hex representation.
155
+
156
+ Useful for debugging and display.
157
+
158
+ Examples:
159
+ >>> to_hex('⣞⣭⣾⣯')
160
+ 'deadbeef'
161
+ """
162
+ return decode(braille).hex()
@@ -0,0 +1,146 @@
1
+ """
2
+ Braille Fingerprints — SCLRecord → fixed-width Braille hash.
3
+
4
+ Maps SCL records and documents to compact, fixed-width Braille strings
5
+ using SHA-256 hashing. These fingerprints serve as:
6
+
7
+ 1. **Dedup keys** — identical records produce identical fingerprints
8
+ 2. **LSH approximation** — similar records often share prefix bits
9
+ 3. **Gossip payloads** — 4-char fingerprint = 32 bits, cheap to propagate
10
+ 4. **Convergence checks** — Hamming distance between fingerprints estimates
11
+ semantic divergence between agents
12
+
13
+ At scale (N → ∞):
14
+ - Agents broadcast fingerprints instead of full state
15
+ - Cluster-heads aggregate fingerprints for sub-swarms
16
+ - Hamming distance < ε ⟹ agents are in agreement
17
+ - Hamming distance > threshold ⟹ trigger challenger verification
18
+ """
19
+
20
+ import hashlib
21
+
22
+ from .codec import encode
23
+ from ..types import SCLRecord, SCLDocument
24
+
25
+
26
+ def fingerprint(record: SCLRecord, width: int = 4) -> str:
27
+ """SCL record → fixed-width Braille fingerprint.
28
+
29
+ Process:
30
+ 1. Serialize record to canonical bytes (record.to_bytes())
31
+ 2. SHA-256 hash
32
+ 3. Take first `width` bytes of hash
33
+ 4. Encode as Braille
34
+
35
+ Args:
36
+ record: The SCL record to fingerprint.
37
+ width: Number of Braille characters (= bytes of hash). Default 4 (32 bits).
38
+
39
+ Returns:
40
+ String of `width` Braille characters.
41
+
42
+ Examples:
43
+ >>> from src.scl.types import Anchor, Relation, Scope, SCLRecord
44
+ >>> r = SCLRecord(Anchor('router'), Relation('select'), Scope({'model': 'qwen3:4b'}))
45
+ >>> len(fingerprint(r))
46
+ 4
47
+ >>> fingerprint(r) == fingerprint(r) # deterministic
48
+ True
49
+ """
50
+ h = hashlib.sha256(record.to_bytes()).digest()
51
+ return encode(h[:width])
52
+
53
+
54
+ def fingerprint_document(doc: SCLDocument, width: int = 8) -> str:
55
+ """Full SCL document → single Braille fingerprint.
56
+
57
+ Hashes the concatenated binary of all records.
58
+
59
+ Args:
60
+ doc: The SCL document to fingerprint.
61
+ width: Number of Braille characters. Default 8 (64 bits).
62
+
63
+ Returns:
64
+ String of `width` Braille characters.
65
+ """
66
+ h = hashlib.sha256(doc.to_bytes()).digest()
67
+ return encode(h[:width])
68
+
69
+
70
+ def fingerprint_batch(records: list[SCLRecord], width: int = 8) -> str:
71
+ """Multiple SCL records → single Braille fingerprint.
72
+
73
+ Convenience wrapper: creates a temporary document and fingerprints it.
74
+ """
75
+ doc = SCLDocument(records=records)
76
+ return fingerprint_document(doc, width=width)
77
+
78
+
79
+ def fingerprint_match(fp1: str, fp2: str) -> bool:
80
+ """Compare two fingerprints for exact equality."""
81
+ return fp1 == fp2
82
+
83
+
84
+ def similarity(fp1: str, fp2: str) -> float:
85
+ """Hamming-distance-based similarity between two fingerprints.
86
+
87
+ Compares bit-by-bit. Returns 0.0 (completely different) to 1.0 (identical).
88
+
89
+ At scale, this is the convergence metric:
90
+ - similarity > 0.9 → agents agree, no action needed
91
+ - similarity 0.5–0.9 → partial agreement, may need verification
92
+ - similarity < 0.5 → disagreement, trigger challenger/swarm
93
+
94
+ Args:
95
+ fp1: First Braille fingerprint.
96
+ fp2: Second Braille fingerprint.
97
+
98
+ Returns:
99
+ Float in [0.0, 1.0].
100
+
101
+ Raises:
102
+ ValueError: If fingerprints have different lengths.
103
+ """
104
+ if len(fp1) != len(fp2):
105
+ raise ValueError(
106
+ f"Fingerprints must have same length: {len(fp1)} vs {len(fp2)}"
107
+ )
108
+
109
+ if not fp1:
110
+ return 1.0
111
+
112
+ total_bits = len(fp1) * 8
113
+ matching_bits = 0
114
+
115
+ for c1, c2 in zip(fp1, fp2):
116
+ b1 = ord(c1) - 0x2800
117
+ b2 = ord(c2) - 0x2800
118
+ # XOR gives bits that differ; popcount gives number of differing bits
119
+ xor = b1 ^ b2
120
+ differing = bin(xor).count("1")
121
+ matching_bits += 8 - differing
122
+
123
+ return matching_bits / total_bits
124
+
125
+
126
+ def hamming_distance(fp1: str, fp2: str) -> int:
127
+ """Raw Hamming distance in bits between two fingerprints.
128
+
129
+ Args:
130
+ fp1: First Braille fingerprint.
131
+ fp2: Second Braille fingerprint.
132
+
133
+ Returns:
134
+ Number of differing bits.
135
+ """
136
+ if len(fp1) != len(fp2):
137
+ raise ValueError(
138
+ f"Fingerprints must have same length: {len(fp1)} vs {len(fp2)}"
139
+ )
140
+
141
+ distance = 0
142
+ for c1, c2 in zip(fp1, fp2):
143
+ b1 = ord(c1) - 0x2800
144
+ b2 = ord(c2) - 0x2800
145
+ distance += bin(b1 ^ b2).count("1")
146
+ return distance