scl-lang 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scl/__init__.py +26 -0
- scl/braille/__init__.py +14 -0
- scl/braille/codec.py +162 -0
- scl/braille/fingerprint.py +146 -0
- scl/delta.py +489 -0
- scl/emitter.py +134 -0
- scl/eval.py +720 -0
- scl/gossip.py +1190 -0
- scl/grammar.py +129 -0
- scl/iblt.py +581 -0
- scl/intent.py +408 -0
- scl/ontology.py +350 -0
- scl/parser.py +246 -0
- scl/types.py +440 -0
- scl_lang-0.1.0.dist-info/METADATA +174 -0
- scl_lang-0.1.0.dist-info/RECORD +17 -0
- scl_lang-0.1.0.dist-info/WHEEL +4 -0
scl/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""
|
|
2
|
+
SCL — Semantic Compression Language.
|
|
3
|
+
|
|
4
|
+
Grammar: @anchor → verb [key: value, key: value]
|
|
5
|
+
|
|
6
|
+
>>> from scl import SCLRecord, Anchor, Relation, Scope
|
|
7
|
+
>>> from scl.parser import parse_document
|
|
8
|
+
>>> from scl.delta import Delta, SemanticState, DeltaStream
|
|
9
|
+
>>> from scl.gossip import Peer, HierarchicalSwarm
|
|
10
|
+
|
|
11
|
+
Modules:
|
|
12
|
+
types — Anchor, Relation, Scope, SCLRecord, SCLDocument
|
|
13
|
+
parser — text → SCL AST
|
|
14
|
+
emitter — SCL AST → text
|
|
15
|
+
grammar — formal BNF specification
|
|
16
|
+
delta — CRDT-style semantic state deltas
|
|
17
|
+
gossip — hierarchical gossip protocol (100K+ agents)
|
|
18
|
+
eval — executable SCL rules engine
|
|
19
|
+
iblt — Invertible Bloom Lookup Tables for O(d) sync
|
|
20
|
+
braille — compact braille encoding layer
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from .types import Anchor, Relation, Scope, SCLRecord, SCLDocument
|
|
24
|
+
|
|
25
|
+
__version__ = "0.1.0"
|
|
26
|
+
__all__ = ["Anchor", "Relation", "Scope", "SCLRecord", "SCLDocument"]
|
scl/braille/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Braille — compact encoding layer for SCL.
|
|
3
|
+
|
|
4
|
+
Uses the 256 Unicode Braille characters (U+2800–U+28FF) as a bijective
|
|
5
|
+
byte-to-character encoding. Each byte maps to exactly one Braille character.
|
|
6
|
+
|
|
7
|
+
Modules:
|
|
8
|
+
codec — encode/decode bytes ↔ Braille Unicode
|
|
9
|
+
fingerprint — SCLRecord → fixed-width Braille hash
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .codec import encode, decode, encode_hex, encode_int
|
|
13
|
+
|
|
14
|
+
__all__ = ["encode", "decode", "encode_hex", "encode_int"]
|
scl/braille/codec.py
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Braille Codec — encode/decode arbitrary bytes as Braille Unicode.
|
|
3
|
+
|
|
4
|
+
The 256 Unicode Braille characters (U+2800 to U+28FF) map 1:1 to byte values
|
|
5
|
+
0x00–0xFF. Each character's Unicode codepoint offset from U+2800 IS the byte
|
|
6
|
+
value. This gives us a natural, lossless, bijective encoding.
|
|
7
|
+
|
|
8
|
+
Properties:
|
|
9
|
+
- Bijective: every byte sequence has exactly one Braille representation
|
|
10
|
+
- Lossless: decode(encode(x)) == x for all x
|
|
11
|
+
- Fixed density: 1 byte = 1 Braille character = 3 UTF-8 bytes
|
|
12
|
+
|
|
13
|
+
Token efficiency:
|
|
14
|
+
Most LLM tokenizers treat Braille characters as single tokens or small
|
|
15
|
+
multi-byte sequences, making this encoding competitive with or better than
|
|
16
|
+
hex/base64 for transmitting binary data through language models.
|
|
17
|
+
|
|
18
|
+
Visual examples:
|
|
19
|
+
0x00 → ⠀ (blank) 0x01 → ⠁ 0x41 ('A') → ⡁ 0xFF → ⣿ (all dots)
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
# The base codepoint for Braille patterns
|
|
23
|
+
_BRAILLE_BASE = 0x2800
|
|
24
|
+
|
|
25
|
+
# Pre-compute lookup tables for speed
|
|
26
|
+
_BYTE_TO_BRAILLE: list[str] = [chr(_BRAILLE_BASE + i) for i in range(256)]
|
|
27
|
+
_BRAILLE_TO_BYTE: dict[str, int] = {chr(_BRAILLE_BASE + i): i for i in range(256)}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def encode(data: bytes) -> str:
|
|
31
|
+
"""Encode arbitrary bytes as a Braille Unicode string.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
data: Arbitrary byte sequence.
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
String of Braille characters, one per input byte.
|
|
38
|
+
|
|
39
|
+
Examples:
|
|
40
|
+
>>> encode(b'\\x00')
|
|
41
|
+
'⠀'
|
|
42
|
+
>>> encode(b'\\xff')
|
|
43
|
+
'⣿'
|
|
44
|
+
>>> encode(b'hello')
|
|
45
|
+
'⡨⡥⡬⡬⡯'
|
|
46
|
+
"""
|
|
47
|
+
return "".join(_BYTE_TO_BRAILLE[b] for b in data)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def decode(braille: str) -> bytes:
|
|
51
|
+
"""Decode a Braille Unicode string back to bytes.
|
|
52
|
+
|
|
53
|
+
Args:
|
|
54
|
+
braille: String of Braille characters.
|
|
55
|
+
|
|
56
|
+
Returns:
|
|
57
|
+
Original byte sequence.
|
|
58
|
+
|
|
59
|
+
Raises:
|
|
60
|
+
ValueError: If string contains non-Braille characters.
|
|
61
|
+
|
|
62
|
+
Examples:
|
|
63
|
+
>>> decode('⡨⡥⡬⡬⡯')
|
|
64
|
+
b'hello'
|
|
65
|
+
"""
|
|
66
|
+
result = bytearray(len(braille))
|
|
67
|
+
for i, char in enumerate(braille):
|
|
68
|
+
cp = ord(char) - _BRAILLE_BASE
|
|
69
|
+
if cp < 0 or cp > 255:
|
|
70
|
+
raise ValueError(
|
|
71
|
+
f"Character at position {i} is not a Braille pattern: "
|
|
72
|
+
f"U+{ord(char):04X} ({char!r})"
|
|
73
|
+
)
|
|
74
|
+
result[i] = cp
|
|
75
|
+
return bytes(result)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def encode_hex(hex_str: str) -> str:
|
|
79
|
+
"""Encode a hex string as Braille.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
hex_str: Hex string (with or without '0x' prefix, spaces allowed).
|
|
83
|
+
|
|
84
|
+
Returns:
|
|
85
|
+
Braille-encoded string.
|
|
86
|
+
|
|
87
|
+
Examples:
|
|
88
|
+
>>> encode_hex('deadbeef')
|
|
89
|
+
'⣞⣭⣾⣯'
|
|
90
|
+
"""
|
|
91
|
+
hex_str = hex_str.replace(" ", "").replace("0x", "").replace("0X", "")
|
|
92
|
+
return encode(bytes.fromhex(hex_str))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def encode_int(n: int, width: int = 4) -> str:
|
|
96
|
+
"""Encode an integer as fixed-width Braille (big-endian).
|
|
97
|
+
|
|
98
|
+
Args:
|
|
99
|
+
n: Non-negative integer.
|
|
100
|
+
width: Number of bytes (= number of Braille characters).
|
|
101
|
+
|
|
102
|
+
Returns:
|
|
103
|
+
Fixed-width Braille string.
|
|
104
|
+
|
|
105
|
+
Raises:
|
|
106
|
+
ValueError: If n is negative or doesn't fit in width bytes.
|
|
107
|
+
|
|
108
|
+
Examples:
|
|
109
|
+
>>> encode_int(0, width=2)
|
|
110
|
+
'⠀⠀'
|
|
111
|
+
>>> encode_int(255, width=1)
|
|
112
|
+
'⣿'
|
|
113
|
+
>>> encode_int(65535, width=2)
|
|
114
|
+
'⣿⣿'
|
|
115
|
+
"""
|
|
116
|
+
if n < 0:
|
|
117
|
+
raise ValueError(f"Cannot encode negative integer: {n}")
|
|
118
|
+
max_val = (1 << (width * 8)) - 1
|
|
119
|
+
if n > max_val:
|
|
120
|
+
raise ValueError(
|
|
121
|
+
f"Integer {n} does not fit in {width} bytes (max {max_val})"
|
|
122
|
+
)
|
|
123
|
+
return encode(n.to_bytes(width, byteorder="big"))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def decode_int(braille: str) -> int:
|
|
127
|
+
"""Decode a Braille string back to an integer (big-endian).
|
|
128
|
+
|
|
129
|
+
Args:
|
|
130
|
+
braille: Braille-encoded integer.
|
|
131
|
+
|
|
132
|
+
Returns:
|
|
133
|
+
The decoded integer.
|
|
134
|
+
"""
|
|
135
|
+
return int.from_bytes(decode(braille), byteorder="big")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def is_braille(char: str) -> bool:
|
|
139
|
+
"""Check if a character is a Braille pattern (U+2800–U+28FF)."""
|
|
140
|
+
cp = ord(char)
|
|
141
|
+
return _BRAILLE_BASE <= cp <= _BRAILLE_BASE + 255
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def braille_len(braille: str) -> int:
|
|
145
|
+
"""Return the number of bytes encoded in a Braille string.
|
|
146
|
+
|
|
147
|
+
Same as len(braille) since the encoding is 1:1, but explicit
|
|
148
|
+
about the semantic meaning.
|
|
149
|
+
"""
|
|
150
|
+
return len(braille)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def to_hex(braille: str) -> str:
|
|
154
|
+
"""Convert Braille string to hex representation.
|
|
155
|
+
|
|
156
|
+
Useful for debugging and display.
|
|
157
|
+
|
|
158
|
+
Examples:
|
|
159
|
+
>>> to_hex('⣞⣭⣾⣯')
|
|
160
|
+
'deadbeef'
|
|
161
|
+
"""
|
|
162
|
+
return decode(braille).hex()
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Braille Fingerprints — SCLRecord → fixed-width Braille hash.
|
|
3
|
+
|
|
4
|
+
Maps SCL records and documents to compact, fixed-width Braille strings
|
|
5
|
+
using SHA-256 hashing. These fingerprints serve as:
|
|
6
|
+
|
|
7
|
+
1. **Dedup keys** — identical records produce identical fingerprints
|
|
8
|
+
2. **LSH approximation** — similar records often share prefix bits
|
|
9
|
+
3. **Gossip payloads** — 4-char fingerprint = 32 bits, cheap to propagate
|
|
10
|
+
4. **Convergence checks** — Hamming distance between fingerprints estimates
|
|
11
|
+
semantic divergence between agents
|
|
12
|
+
|
|
13
|
+
At scale (N → ∞):
|
|
14
|
+
- Agents broadcast fingerprints instead of full state
|
|
15
|
+
- Cluster-heads aggregate fingerprints for sub-swarms
|
|
16
|
+
- Hamming distance < ε ⟹ agents are in agreement
|
|
17
|
+
- Hamming distance > threshold ⟹ trigger challenger verification
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
|
|
22
|
+
from .codec import encode
|
|
23
|
+
from ..types import SCLRecord, SCLDocument
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def fingerprint(record: SCLRecord, width: int = 4) -> str:
|
|
27
|
+
"""SCL record → fixed-width Braille fingerprint.
|
|
28
|
+
|
|
29
|
+
Process:
|
|
30
|
+
1. Serialize record to canonical bytes (record.to_bytes())
|
|
31
|
+
2. SHA-256 hash
|
|
32
|
+
3. Take first `width` bytes of hash
|
|
33
|
+
4. Encode as Braille
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
record: The SCL record to fingerprint.
|
|
37
|
+
width: Number of Braille characters (= bytes of hash). Default 4 (32 bits).
|
|
38
|
+
|
|
39
|
+
Returns:
|
|
40
|
+
String of `width` Braille characters.
|
|
41
|
+
|
|
42
|
+
Examples:
|
|
43
|
+
>>> from src.scl.types import Anchor, Relation, Scope, SCLRecord
|
|
44
|
+
>>> r = SCLRecord(Anchor('router'), Relation('select'), Scope({'model': 'qwen3:4b'}))
|
|
45
|
+
>>> len(fingerprint(r))
|
|
46
|
+
4
|
|
47
|
+
>>> fingerprint(r) == fingerprint(r) # deterministic
|
|
48
|
+
True
|
|
49
|
+
"""
|
|
50
|
+
h = hashlib.sha256(record.to_bytes()).digest()
|
|
51
|
+
return encode(h[:width])
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def fingerprint_document(doc: SCLDocument, width: int = 8) -> str:
|
|
55
|
+
"""Full SCL document → single Braille fingerprint.
|
|
56
|
+
|
|
57
|
+
Hashes the concatenated binary of all records.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
doc: The SCL document to fingerprint.
|
|
61
|
+
width: Number of Braille characters. Default 8 (64 bits).
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
String of `width` Braille characters.
|
|
65
|
+
"""
|
|
66
|
+
h = hashlib.sha256(doc.to_bytes()).digest()
|
|
67
|
+
return encode(h[:width])
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def fingerprint_batch(records: list[SCLRecord], width: int = 8) -> str:
|
|
71
|
+
"""Multiple SCL records → single Braille fingerprint.
|
|
72
|
+
|
|
73
|
+
Convenience wrapper: creates a temporary document and fingerprints it.
|
|
74
|
+
"""
|
|
75
|
+
doc = SCLDocument(records=records)
|
|
76
|
+
return fingerprint_document(doc, width=width)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def fingerprint_match(fp1: str, fp2: str) -> bool:
|
|
80
|
+
"""Compare two fingerprints for exact equality."""
|
|
81
|
+
return fp1 == fp2
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def similarity(fp1: str, fp2: str) -> float:
|
|
85
|
+
"""Hamming-distance-based similarity between two fingerprints.
|
|
86
|
+
|
|
87
|
+
Compares bit-by-bit. Returns 0.0 (completely different) to 1.0 (identical).
|
|
88
|
+
|
|
89
|
+
At scale, this is the convergence metric:
|
|
90
|
+
- similarity > 0.9 → agents agree, no action needed
|
|
91
|
+
- similarity 0.5–0.9 → partial agreement, may need verification
|
|
92
|
+
- similarity < 0.5 → disagreement, trigger challenger/swarm
|
|
93
|
+
|
|
94
|
+
Args:
|
|
95
|
+
fp1: First Braille fingerprint.
|
|
96
|
+
fp2: Second Braille fingerprint.
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
Float in [0.0, 1.0].
|
|
100
|
+
|
|
101
|
+
Raises:
|
|
102
|
+
ValueError: If fingerprints have different lengths.
|
|
103
|
+
"""
|
|
104
|
+
if len(fp1) != len(fp2):
|
|
105
|
+
raise ValueError(
|
|
106
|
+
f"Fingerprints must have same length: {len(fp1)} vs {len(fp2)}"
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
if not fp1:
|
|
110
|
+
return 1.0
|
|
111
|
+
|
|
112
|
+
total_bits = len(fp1) * 8
|
|
113
|
+
matching_bits = 0
|
|
114
|
+
|
|
115
|
+
for c1, c2 in zip(fp1, fp2):
|
|
116
|
+
b1 = ord(c1) - 0x2800
|
|
117
|
+
b2 = ord(c2) - 0x2800
|
|
118
|
+
# XOR gives bits that differ; popcount gives number of differing bits
|
|
119
|
+
xor = b1 ^ b2
|
|
120
|
+
differing = bin(xor).count("1")
|
|
121
|
+
matching_bits += 8 - differing
|
|
122
|
+
|
|
123
|
+
return matching_bits / total_bits
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def hamming_distance(fp1: str, fp2: str) -> int:
|
|
127
|
+
"""Raw Hamming distance in bits between two fingerprints.
|
|
128
|
+
|
|
129
|
+
Args:
|
|
130
|
+
fp1: First Braille fingerprint.
|
|
131
|
+
fp2: Second Braille fingerprint.
|
|
132
|
+
|
|
133
|
+
Returns:
|
|
134
|
+
Number of differing bits.
|
|
135
|
+
"""
|
|
136
|
+
if len(fp1) != len(fp2):
|
|
137
|
+
raise ValueError(
|
|
138
|
+
f"Fingerprints must have same length: {len(fp1)} vs {len(fp2)}"
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
distance = 0
|
|
142
|
+
for c1, c2 in zip(fp1, fp2):
|
|
143
|
+
b1 = ord(c1) - 0x2800
|
|
144
|
+
b2 = ord(c2) - 0x2800
|
|
145
|
+
distance += bin(b1 ^ b2).count("1")
|
|
146
|
+
return distance
|