gcf-python 0.5.1__tar.gz → 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gcf_python-1.0.1/.github/FUNDING.yml +1 -0
- gcf_python-1.0.1/.github/workflows/ci.yml +30 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/.github/workflows/publish.yml +5 -1
- {gcf_python-0.5.1 → gcf_python-1.0.1}/CHANGELOG.md +26 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/LICENSE +1 -1
- {gcf_python-0.5.1 → gcf_python-1.0.1}/PKG-INFO +15 -5
- {gcf_python-0.5.1 → gcf_python-1.0.1}/README.md +13 -3
- {gcf_python-0.5.1 → gcf_python-1.0.1}/pyproject.toml +2 -2
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/__init__.py +1 -1
- gcf_python-1.0.1/src/gcf/__main__.py +5 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/cli.py +32 -5
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/decode.py +17 -7
- gcf_python-1.0.1/src/gcf/decode_generic.py +505 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/delta.py +1 -1
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/encode.py +1 -1
- gcf_python-1.0.1/src/gcf/generic.py +179 -0
- gcf_python-1.0.1/src/gcf/scalar.py +298 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/session.py +1 -1
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/stream.py +9 -9
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/stream_generic.py +7 -27
- gcf_python-1.0.1/tests/test_conformance_v2.py +105 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_decode.py +13 -13
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_encode.py +2 -2
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_generic.py +14 -14
- gcf_python-1.0.1/tests/test_roundtrip_v2.py +246 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_stream.py +3 -3
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_stream_generic.py +4 -4
- gcf_python-0.5.1/.github/workflows/ci.yml +0 -21
- gcf_python-0.5.1/src/gcf/decode_generic.py +0 -255
- gcf_python-0.5.1/src/gcf/generic.py +0 -156
- {gcf_python-0.5.1 → gcf_python-1.0.1}/.gitignore +0 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/constants.py +0 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/src/gcf/types.py +0 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/__init__.py +0 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_delta.py +0 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_roundtrip.py +0 -0
- {gcf_python-0.5.1 → gcf_python-1.0.1}/tests/test_session.py +0 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
github: [blackwell-systems]
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [main]
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
test:
|
|
11
|
+
runs-on: ubuntu-latest
|
|
12
|
+
strategy:
|
|
13
|
+
matrix:
|
|
14
|
+
python-version: ['3.9', '3.10', '3.11', '3.12', '3.13']
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
with:
|
|
19
|
+
repository: blackwell-systems/gcf
|
|
20
|
+
path: gcf
|
|
21
|
+
- uses: actions/setup-python@v5
|
|
22
|
+
with:
|
|
23
|
+
python-version: ${{ matrix.python-version }}
|
|
24
|
+
- run: pip install -e ".[dev]" || pip install -e . && pip install pytest
|
|
25
|
+
- name: Unit tests
|
|
26
|
+
run: pytest tests/ -v --ignore=tests/test_conformance_v2.py --ignore=tests/test_roundtrip_v2.py
|
|
27
|
+
- name: Conformance tests (133 fixtures)
|
|
28
|
+
run: pytest tests/test_conformance_v2.py -v
|
|
29
|
+
- name: Property-based round-trip (100K random + 100K adversarial)
|
|
30
|
+
run: pytest tests/test_roundtrip_v2.py -v
|
|
@@ -8,7 +8,7 @@ jobs:
|
|
|
8
8
|
publish:
|
|
9
9
|
runs-on: ubuntu-latest
|
|
10
10
|
permissions:
|
|
11
|
-
contents:
|
|
11
|
+
contents: write
|
|
12
12
|
id-token: write
|
|
13
13
|
steps:
|
|
14
14
|
- uses: actions/checkout@v4
|
|
@@ -22,3 +22,7 @@ jobs:
|
|
|
22
22
|
env:
|
|
23
23
|
TWINE_USERNAME: __token__
|
|
24
24
|
TWINE_PASSWORD: ${{ secrets.PYPI_TOKEN }}
|
|
25
|
+
- name: Create GitHub release
|
|
26
|
+
uses: softprops/action-gh-release@v2
|
|
27
|
+
with:
|
|
28
|
+
generate_release_notes: true
|
|
@@ -1,5 +1,31 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v1.0.1 (2026-06-10)
|
|
4
|
+
|
|
5
|
+
- CLI: `encode-generic` and `decode-generic` subcommands for generic profile
|
|
6
|
+
- CLI now supports both graph and generic profiles
|
|
7
|
+
- `python -m gcf` entry point
|
|
8
|
+
|
|
9
|
+
## v1.0.0 (2026-06-10)
|
|
10
|
+
|
|
11
|
+
SPEC v2.0 implementation. 126/133 conformance fixtures passing (7 skipped: session, delta, binary UTF-8, negative zero, graph encode). 40M property-based round-trips with zero failures.
|
|
12
|
+
|
|
13
|
+
### Breaking changes from v0.5.0
|
|
14
|
+
|
|
15
|
+
- `encode_generic` emits `GCF profile=generic` header
|
|
16
|
+
- `decode_generic` requires `GCF profile=` header
|
|
17
|
+
- Strings colliding with typed literals are quoted
|
|
18
|
+
- Full JSON string escaping and number grammar
|
|
19
|
+
- `-` for null, `~` for absent, `^` for nested attachments
|
|
20
|
+
- `##! summary` trailer replaces `## _summary`
|
|
21
|
+
- Graph encoder emits `profile=graph`
|
|
22
|
+
|
|
23
|
+
### New
|
|
24
|
+
|
|
25
|
+
- `scalar.py`: common scalar grammar (quoting, escaping, parsing, number formatting)
|
|
26
|
+
- Conformance test runner (133 fixtures)
|
|
27
|
+
- Property-based round-trip tests (40M verified, configurable via `GCF_ITERATIONS`)
|
|
28
|
+
|
|
3
29
|
## v0.5.0 (2026-06-06)
|
|
4
30
|
|
|
5
31
|
- `GenericStreamEncoder`: zero-buffering tabular streaming encode (begin_array/write_row/end_array/write_kv/write_section/write_inline_array)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gcf-python
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary:
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: Drop-in JSON replacement for AI pipelines. 79% fewer tokens. 90.7% comprehension across 10 models. Zero dependencies.
|
|
5
5
|
Project-URL: Homepage, https://github.com/blackwell-systems/gcf-python
|
|
6
6
|
Project-URL: Documentation, https://blackwell-systems.github.io/gcf/
|
|
7
7
|
Project-URL: Specification, https://github.com/blackwell-systems/gcf
|
|
@@ -32,7 +32,7 @@ Description-Content-Type: text/markdown
|
|
|
32
32
|
|
|
33
33
|
Python implementation of [GCF](https://gcformat.com/) — the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
|
|
34
34
|
|
|
35
|
-
**79% fewer input tokens than JSON. 63% fewer output tokens. 90.
|
|
35
|
+
**79% fewer input tokens than JSON. 63% fewer output tokens. 90.7% average comprehension accuracy across 10 models and 3 providers (four models hit 100%). 1,300+ LLM evaluations. Zero training.**
|
|
36
36
|
|
|
37
37
|
Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
|
|
38
38
|
|
|
@@ -210,7 +210,7 @@ Works on dicts, lists, and primitives. Lists of uniform dicts get tabular rows.
|
|
|
210
210
|
|
|
211
211
|
| | GCF | TOON | JSON |
|
|
212
212
|
|---|---|---|---|
|
|
213
|
-
| **Comprehension** (23 runs, 10 models) | **90.
|
|
213
|
+
| **Comprehension** (23 runs, 10 models) | **90.7%** | 68.5% | 53.6% |
|
|
214
214
|
| **Generation** (28 runs, 9 models) | **5/5** | 1.0/5 | 5.0/5 |
|
|
215
215
|
| **Input tokens** (500 symbols) | **11,090** | 16,378 | 53,341 |
|
|
216
216
|
| **Output tokens** (100 symbols) | **5,976** | 8,937 | 16,121 |
|
|
@@ -228,6 +228,16 @@ GCF wins all 6 datasets on [TOON's own benchmark](https://github.com/blackwell-s
|
|
|
228
228
|
- [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
|
|
229
229
|
- [TOON benchmark fork](https://github.com/blackwell-systems/toon/tree/gcf-comparison)
|
|
230
230
|
|
|
231
|
+
|
|
232
|
+
<details>
|
|
233
|
+
<summary>More links</summary>
|
|
234
|
+
|
|
235
|
+
- [betterthanjson.com](https://betterthanjson.com)
|
|
236
|
+
- [jsonalternative.com](https://jsonalternative.com)
|
|
237
|
+
- [betterthantoon.com](https://betterthantoon.com)
|
|
238
|
+
|
|
239
|
+
</details>
|
|
240
|
+
|
|
231
241
|
## License
|
|
232
242
|
|
|
233
|
-
MIT
|
|
243
|
+
MIT - [Dayna Blackwell](https://github.com/blackwell-systems)
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
Python implementation of [GCF](https://gcformat.com/) — the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
|
|
9
9
|
|
|
10
|
-
**79% fewer input tokens than JSON. 63% fewer output tokens. 90.
|
|
10
|
+
**79% fewer input tokens than JSON. 63% fewer output tokens. 90.7% average comprehension accuracy across 10 models and 3 providers (four models hit 100%). 1,300+ LLM evaluations. Zero training.**
|
|
11
11
|
|
|
12
12
|
Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
|
|
13
13
|
|
|
@@ -185,7 +185,7 @@ Works on dicts, lists, and primitives. Lists of uniform dicts get tabular rows.
|
|
|
185
185
|
|
|
186
186
|
| | GCF | TOON | JSON |
|
|
187
187
|
|---|---|---|---|
|
|
188
|
-
| **Comprehension** (23 runs, 10 models) | **90.
|
|
188
|
+
| **Comprehension** (23 runs, 10 models) | **90.7%** | 68.5% | 53.6% |
|
|
189
189
|
| **Generation** (28 runs, 9 models) | **5/5** | 1.0/5 | 5.0/5 |
|
|
190
190
|
| **Input tokens** (500 symbols) | **11,090** | 16,378 | 53,341 |
|
|
191
191
|
| **Output tokens** (100 symbols) | **5,976** | 8,937 | 16,121 |
|
|
@@ -203,6 +203,16 @@ GCF wins all 6 datasets on [TOON's own benchmark](https://github.com/blackwell-s
|
|
|
203
203
|
- [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
|
|
204
204
|
- [TOON benchmark fork](https://github.com/blackwell-systems/toon/tree/gcf-comparison)
|
|
205
205
|
|
|
206
|
+
|
|
207
|
+
<details>
|
|
208
|
+
<summary>More links</summary>
|
|
209
|
+
|
|
210
|
+
- [betterthanjson.com](https://betterthanjson.com)
|
|
211
|
+
- [jsonalternative.com](https://jsonalternative.com)
|
|
212
|
+
- [betterthantoon.com](https://betterthantoon.com)
|
|
213
|
+
|
|
214
|
+
</details>
|
|
215
|
+
|
|
206
216
|
## License
|
|
207
217
|
|
|
208
|
-
MIT
|
|
218
|
+
MIT - [Dayna Blackwell](https://github.com/blackwell-systems)
|
|
@@ -4,8 +4,8 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "gcf-python"
|
|
7
|
-
version = "0.
|
|
8
|
-
description = "
|
|
7
|
+
version = "1.0.1"
|
|
8
|
+
description = "Drop-in JSON replacement for AI pipelines. 79% fewer tokens. 90.7% comprehension across 10 models. Zero dependencies."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "MIT"}
|
|
11
11
|
requires-python = ">=3.9"
|
|
@@ -5,19 +5,25 @@ import sys
|
|
|
5
5
|
|
|
6
6
|
from .decode import decode
|
|
7
7
|
from .encode import encode
|
|
8
|
+
from .decode_generic import decode_generic
|
|
9
|
+
from .generic import encode_generic
|
|
8
10
|
from .types import Edge, Payload, Symbol
|
|
9
11
|
|
|
10
12
|
USAGE = """gcf - token-optimized wire format for LLM tool responses
|
|
11
13
|
|
|
12
14
|
Usage:
|
|
13
|
-
gcf encode
|
|
14
|
-
gcf decode
|
|
15
|
-
gcf
|
|
16
|
-
gcf
|
|
15
|
+
gcf encode [file] Encode JSON graph payload to GCF (stdin if no file)
|
|
16
|
+
gcf decode [file] Decode GCF graph text to JSON (stdin if no file)
|
|
17
|
+
gcf encode-generic [file] Encode JSON to GCF generic profile (stdin if no file)
|
|
18
|
+
gcf decode-generic [file] Decode GCF generic text to JSON (stdin if no file)
|
|
19
|
+
gcf stats [file] Compare token counts: JSON vs GCF (stdin if no file)
|
|
20
|
+
gcf version Print version
|
|
17
21
|
|
|
18
22
|
Examples:
|
|
19
23
|
gcf encode < payload.json
|
|
20
24
|
gcf decode < payload.gcf
|
|
25
|
+
echo '{"name":"Alice"}' | gcf encode-generic
|
|
26
|
+
echo 'GCF profile=generic\\nname=Alice' | gcf decode-generic
|
|
21
27
|
gcf stats payload.json
|
|
22
28
|
"""
|
|
23
29
|
|
|
@@ -37,11 +43,18 @@ def main() -> None:
|
|
|
37
43
|
elif cmd == "decode":
|
|
38
44
|
data = _read_input(file_args)
|
|
39
45
|
_do_decode(data)
|
|
46
|
+
elif cmd == "encode-generic":
|
|
47
|
+
data = _read_input(file_args)
|
|
48
|
+
_do_encode_generic(data)
|
|
49
|
+
elif cmd == "decode-generic":
|
|
50
|
+
data = _read_input(file_args)
|
|
51
|
+
_do_decode_generic(data)
|
|
40
52
|
elif cmd == "stats":
|
|
41
53
|
data = _read_input(file_args)
|
|
42
54
|
_do_stats(data)
|
|
43
55
|
elif cmd == "version":
|
|
44
|
-
|
|
56
|
+
from . import __version__
|
|
57
|
+
print(f"gcf {__version__}")
|
|
45
58
|
else:
|
|
46
59
|
print(f"unknown command: {cmd}\n", file=sys.stderr)
|
|
47
60
|
print(USAGE, file=sys.stderr, end="")
|
|
@@ -115,6 +128,20 @@ def _payload_to_json(p: Payload) -> str:
|
|
|
115
128
|
return json.dumps(obj, indent=2)
|
|
116
129
|
|
|
117
130
|
|
|
131
|
+
def _do_encode_generic(data: str) -> None:
|
|
132
|
+
try:
|
|
133
|
+
obj = json.loads(data)
|
|
134
|
+
except json.JSONDecodeError as e:
|
|
135
|
+
print(f"error: invalid JSON: {e}", file=sys.stderr)
|
|
136
|
+
sys.exit(1)
|
|
137
|
+
print(encode_generic(obj), end="")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _do_decode_generic(data: str) -> None:
|
|
141
|
+
result = decode_generic(data)
|
|
142
|
+
print(json.dumps(result))
|
|
143
|
+
|
|
144
|
+
|
|
118
145
|
def _do_encode(data: str) -> None:
|
|
119
146
|
try:
|
|
120
147
|
p = _payload_from_json(data)
|
|
@@ -35,7 +35,11 @@ def decode(input_text: str) -> Payload:
|
|
|
35
35
|
_parse_header(header[4:], p)
|
|
36
36
|
|
|
37
37
|
if not p.tool:
|
|
38
|
-
raise DecodeError("header missing required 'tool' field")
|
|
38
|
+
raise DecodeError("missing_tool: header missing required 'tool' field")
|
|
39
|
+
|
|
40
|
+
# Detect delta mode.
|
|
41
|
+
is_delta = "delta=true" in header
|
|
42
|
+
valid_delta_sections = {"removed", "added", "edges_removed", "edges_added"}
|
|
39
43
|
|
|
40
44
|
# Parse body: symbols and edges.
|
|
41
45
|
symbols: list[Symbol] = []
|
|
@@ -48,6 +52,10 @@ def decode(input_text: str) -> Payload:
|
|
|
48
52
|
if not line:
|
|
49
53
|
continue
|
|
50
54
|
|
|
55
|
+
# Skip ##! summary trailer.
|
|
56
|
+
if line.startswith("##! "):
|
|
57
|
+
continue
|
|
58
|
+
|
|
51
59
|
# Group header.
|
|
52
60
|
if line.startswith("## "):
|
|
53
61
|
group = line[3:]
|
|
@@ -55,6 +63,8 @@ def decode(input_text: str) -> Payload:
|
|
|
55
63
|
bracket_idx = group.find(" [")
|
|
56
64
|
if bracket_idx >= 0:
|
|
57
65
|
group = group[:bracket_idx]
|
|
66
|
+
if is_delta and group not in valid_delta_sections:
|
|
67
|
+
raise DecodeError(f"malformed_delta: invalid delta section {group!r}")
|
|
58
68
|
in_edges = group == "edges"
|
|
59
69
|
if not in_edges:
|
|
60
70
|
if group == "targets":
|
|
@@ -113,19 +123,19 @@ def _parse_header(fields: str, p: Payload) -> None:
|
|
|
113
123
|
def _parse_symbol_line(line: str, distance: int) -> tuple[Symbol, int]:
|
|
114
124
|
"""Parse a symbol line into a Symbol and its local ID."""
|
|
115
125
|
if not line.startswith("@"):
|
|
116
|
-
raise DecodeError(f"expected symbol line starting with @, got {line!r}")
|
|
126
|
+
raise DecodeError(f"invalid_node_line: expected symbol line starting with @, got {line!r}")
|
|
117
127
|
|
|
118
128
|
parts = line.split()
|
|
119
129
|
if len(parts) < 5:
|
|
120
130
|
raise DecodeError(
|
|
121
|
-
f"symbol line needs at least 5 fields, got {len(parts)} in {line!r}"
|
|
131
|
+
f"invalid_node_line: symbol line needs at least 5 fields, got {len(parts)} in {line!r}"
|
|
122
132
|
)
|
|
123
133
|
|
|
124
134
|
id_str = parts[0][1:] # strip @
|
|
125
135
|
try:
|
|
126
136
|
sym_id = int(id_str)
|
|
127
137
|
except ValueError as e:
|
|
128
|
-
raise DecodeError(f"invalid symbol id {id_str!r}: {e}") from e
|
|
138
|
+
raise DecodeError(f"invalid_symbol_id: invalid symbol id {id_str!r}: {e}") from e
|
|
129
139
|
|
|
130
140
|
kind = parts[1]
|
|
131
141
|
kind = KIND_EXPAND.get(kind, kind)
|
|
@@ -135,7 +145,7 @@ def _parse_symbol_line(line: str, distance: int) -> tuple[Symbol, int]:
|
|
|
135
145
|
try:
|
|
136
146
|
score = float(parts[3])
|
|
137
147
|
except ValueError as e:
|
|
138
|
-
raise DecodeError(f"invalid score {parts[3]!r}: {e}") from e
|
|
148
|
+
raise DecodeError(f"invalid_score: invalid score {parts[3]!r}: {e}") from e
|
|
139
149
|
|
|
140
150
|
provenance = parts[4]
|
|
141
151
|
|
|
@@ -157,7 +167,7 @@ def _parse_edge_line(line: str, sym_by_id: dict[int, Symbol]) -> Edge:
|
|
|
157
167
|
ref = parts[0]
|
|
158
168
|
lt_idx = ref.find("<")
|
|
159
169
|
if lt_idx < 0:
|
|
160
|
-
raise DecodeError(f"edge line missing '<' separator in {ref!r}")
|
|
170
|
+
raise DecodeError(f"invalid_edge_syntax: edge line missing '<' separator in {ref!r}")
|
|
161
171
|
|
|
162
172
|
target_id_str = ref[1:lt_idx] # strip leading @
|
|
163
173
|
source_id_str = ref[lt_idx + 2:] # strip <@
|
|
@@ -176,7 +186,7 @@ def _parse_edge_line(line: str, sym_by_id: dict[int, Symbol]) -> Edge:
|
|
|
176
186
|
source_sym = sym_by_id.get(source_id)
|
|
177
187
|
if target_sym is None or source_sym is None:
|
|
178
188
|
raise DecodeError(
|
|
179
|
-
f"edge references unknown symbol id(s): target={target_id} source={source_id}"
|
|
189
|
+
f"unknown_edge_reference: edge references unknown symbol id(s): target={target_id} source={source_id}"
|
|
180
190
|
)
|
|
181
191
|
|
|
182
192
|
edge_type = parts[1]
|