gcf-python 2.3.0__tar.gz → 2.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gcf_python-2.3.0 → gcf_python-2.5.0}/CHANGELOG.md +39 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/PKG-INFO +47 -11
- {gcf_python-2.3.0 → gcf_python-2.5.0}/README.md +46 -10
- gcf_python-2.5.0/assets/divider-wave-2.png +0 -0
- gcf_python-2.5.0/assets/divider.png +0 -0
- gcf_python-2.5.0/assets/gcf-hero-wire-delta.png +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/pyproject.toml +1 -1
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/__init__.py +5 -1
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/decode_generic.py +76 -8
- gcf_python-2.5.0/src/gcf/delta.py +237 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/encode.py +36 -14
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/generic.py +79 -7
- gcf_python-2.5.0/src/gcf/keyed_map.py +86 -0
- gcf_python-2.5.0/src/gcf/packroot.py +53 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/scalar.py +11 -3
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/session.py +26 -16
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/stream.py +22 -13
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/stream_generic.py +54 -3
- gcf_python-2.5.0/tests/test_conformance_v2.py +457 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_delta.py +5 -5
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_encode.py +3 -2
- gcf_python-2.5.0/tests/test_generic_delta_fuzz.py +60 -0
- gcf_python-2.5.0/tests/test_keyed_map_fuzz.py +281 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_roundtrip.py +5 -5
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_roundtrip_v2.py +68 -8
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_session.py +3 -2
- gcf_python-2.5.0/tests/test_stream_fielddecl.py +110 -0
- gcf_python-2.3.0/src/gcf/delta.py +0 -54
- gcf_python-2.3.0/tests/test_conformance_v2.py +0 -202
- {gcf_python-2.3.0 → gcf_python-2.5.0}/.github/FUNDING.yml +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/.github/workflows/ci.yml +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/.github/workflows/publish.yml +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/.gitignore +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/LICENSE +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/assets/gcf-python-diagram.png +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/__main__.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/cli.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/constants.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/decode.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/generic_delta.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/src/gcf/types.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/__init__.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_decode.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_generic.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_generic_delta.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_generic_delta_session.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_stream.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/tests/test_stream_generic.py +0 -0
- {gcf_python-2.3.0 → gcf_python-2.5.0}/uv.lock +0 -0
|
@@ -1,5 +1,43 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v2.5.0 (2026-08-07)
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
- Keyed-tabular map encoding (SPEC 7.2a): a JSON object whose values are all objects forming a tabular set is encoded as a keyed table (`## [N:]{key,...}`) - the shared value fields are declared once, with one key-prefixed row per member. Canonical by default, supported in nested and streaming positions, and integrated with generic delta using the map key as the identity.
|
|
7
|
+
|
|
8
|
+
### Changed
|
|
9
|
+
- Negative zero is canonicalized to `0` for both integer and floating-point values (SPEC 2.3.1).
|
|
10
|
+
- Canonical-output alignment across all six SDKs: object key ordering, graph header fields, and symbol ordering follow the specification and reference implementation exactly.
|
|
11
|
+
|
|
12
|
+
### Testing
|
|
13
|
+
- Conformance runners assert re-encode idempotence (`encode(decode(x)) == x`) for the generic, graph, and delta profiles; a differential cross-SDK fuzz was added to the verification suite.
|
|
14
|
+
|
|
15
|
+
## v2.4.0 (2026-07-12)
|
|
16
|
+
|
|
17
|
+
### Fixes
|
|
18
|
+
|
|
19
|
+
- The conformance runner now hard-fails on any unhandled operation (instead of silently skipping it) and exercises session, delta, roundtrip, and pack-root fixtures end to end; the graph delta wire decode and verify path is now covered, so no operations remain allow-listed.
|
|
20
|
+
- Implemented the graph delta wire decoder and verifier (`decode_delta` / `verify_delta`): parse a `GCF profile=graph delta=true` wire back into removed/added symbols and edge changes, apply them atomically to a base snapshot, recompute `pack_root`, and reject a wrong `new_root` with `root_mismatch` (SPEC 10.4). The `## added` encoder now emits the trailing `distance` field (SPEC 3.4.1, Section 10.1). The shared `graph-delta` fixtures now run end to end: 001 (encode, gains the trailing distance), 002 (verified apply), 003 (`root_mismatch` rejection).
|
|
21
|
+
- **Session encoding correctness fix.** `encode_with_session` assigned per-response local IDs instead of stable session-global IDs, so the cross-call dedup references (`@N # previously transmitted`) pointed at the wrong symbols, and the header emitted zero-valued `budget`/`tokens`/`edges`. Both are fixed to match the reference; graph session output is now byte-identical across all six SDKs. This had gone undetected because the conformance runner skipped the shared graph-session fixtures (now wired).
|
|
22
|
+
- Added the graph-profile PackRoot (`pack_root(symbols, edges)`, gcf-pack-root-v1, SPEC 10.2): the content-addressed sha256 over canonical, independently-sorted symbol/edge records, byte-identical to gcf-go/rust/typescript/swift/kotlin. The conformance runner now exercises the shared `graph-pack-root` fixtures, which it had been skipping (so this primitive was previously unimplemented and untested).
|
|
23
|
+
- Buffered graph encoder now matches the reference byte-for-byte: symbols are ordered by distance then descending score with local IDs assigned in output order, and the header omits `budget`/`tokens`/`edges` when zero (previously symbols kept input order and zero-valued header fields were always emitted). The conformance runner now exercises the shared `graph-encode` fixtures (001-003), which it had been skipping - which is how this divergence went uncaught.
|
|
24
|
+
- Buffered graph encoder: order edges by source ID, then target ID, then edge type (SPEC 16.1), instead of emitting them in input order. Decode-invariant (edges are a set) and does not affect `pack_root` (which sorts edge records independently), so no content addresses change. Pinned by shared fixture `graph-encode/003`. Streaming edges remain in producer-arrival order.
|
|
25
|
+
- Decoder: reject an orphan `.field` attachment (a `.field` whose name is neither a `^`-marked column of its row nor a `>`-containing field name, SPEC 7.4.6.1.4) instead of silently absorbing it as an undeclared extra field. Such a stray attachment previously decoded to a record no encoder produces, silently injecting a field onto the last-parsed row (a lossless round-trip hole); now rejected per SPEC 16.5 (`orphan_attachment`).
|
|
26
|
+
- Decoder: reject an orphan positional inline body (a pipe-delimited line with no eligible `^{}` attachment-marker cell) instead of silently dropping it. The object-body parser previously skipped any unrecognized line, so a stray positional body (e.g. a second `Bob|b@t.com` after a row's one inline cell was filled) vanished with no error (silent data loss); now rejected per SPEC 16.5 (`orphan_inline_attachment`).
|
|
27
|
+
- Graph streaming trailer: the edge count is now always the last `counts` entry, even when the stream has no edges (positional `counts=2,1,0`; labeled `counts=…,edges:0`). A zero-edge stream previously dropped it, violating the SPEC §8.4 / §8.4.1 rule that the edge count is always present and last (the invariant that keeps the positional form unambiguous). The graph trailer is decoder-ignored, so this changes producer output only.
|
|
28
|
+
|
|
29
|
+
### Streaming: opt-in labeled trailer counts (SPEC §8.4.1)
|
|
30
|
+
|
|
31
|
+
- New `labeled_trailer_counts` keyword on `StreamEncoder`. When set, the `##! summary` graph streaming trailer emits `counts=` in the labeled form `label:count` per group (e.g. `counts=targets:2,related:1,edges:3`) instead of the default positional values-only form (`counts=2,1,3`). Default false is byte-identical to prior output.
|
|
32
|
+
- Opt-in and non-breaking: a producer-side comprehension aid for known weak consumers. The trailer counts remain informational (decoder-ignored) in both forms; neither changes the decoded payload. Mirrors the `gcf-go` reference.
|
|
33
|
+
|
|
34
|
+
### Conformance and docs
|
|
35
|
+
|
|
36
|
+
- Streaming graph trailer now emits `distance_N` group counts in pure group-header emission order (dropping a fixed `targets,related,extended` prefix), matching the other SDKs and deterministic per SPEC 16.1. Byte-identical for contract-conformant (ascending-distance) input; pinned by shared fixtures `streaming-v2/010`–`011`.
|
|
37
|
+
- The conformance runner now executes the `graph-stream-encode` fixtures (streaming-encode parity, previously decode-only): fixture 004 (positional trailer) and 005 (labeled trailer).
|
|
38
|
+
- README: corrected the streaming example trailer from the defunct `## _summary … sections=` to the real `##! summary … counts=`; README now leads with the project diagram.
|
|
39
|
+
- Added a generic-delta fuzz test (decoder never crashes; string round-trip).
|
|
40
|
+
|
|
3
41
|
## v2.3.0 (2026-07-12)
|
|
4
42
|
|
|
5
43
|
### Generic-profile delta encoding (SPEC §10a)
|
|
@@ -18,6 +56,7 @@
|
|
|
18
56
|
- Unit suite mirroring `gcf-go`: self-proving round-trip (diff -> encode -> apply -> recomputed root), determinism / row-order invariance, no-type-collision canonicalization, every invariant/error path, full-payload wire round-trip, the complete server -> wire -> consumer end-to-end loop, and malformed-wire-fails-closed.
|
|
19
57
|
- Conformance runner support for `generic-pack-root`, `generic-delta`, `generic-delta-verify`, `generic-delta-decode` (12 shared fixtures); verified to produce identical pack roots and delta wire to `gcf-go`.
|
|
20
58
|
- Session helper suite (`test_generic_delta_session.py`) mirroring `gcf-go`: FixedN cadence pattern, size-guard triggering, schema-change forced full, FixedN(15)-over-30-turns count, and the load-bearing consumer-stays-in-sync check under both policies. Conformance runner support for `generic-delta-session` (3 shared fixtures: fixed-N, size-guard, schema-change).
|
|
59
|
+
- Generic-delta fuzz (`test_generic_delta_fuzz.py`), mirroring `gcf-go`: the decoder never crashes on arbitrary/mutated input, and arbitrary UTF-8 string cells (including multi-byte and control characters) survive the full-wire round-trip with the pack root preserved.
|
|
21
60
|
|
|
22
61
|
## v2.2.2 (2026-07-10)
|
|
23
62
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gcf-python
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.5.0
|
|
4
4
|
Summary: The AI-native wire format for structured data. 50-92% fewer tokens than JSON, with multi-turn delta encoding for agent loops. 100% comprehension on every frontier model. Zero dependencies.
|
|
5
5
|
Project-URL: Homepage, https://github.com/blackwell-systems/gcf-python
|
|
6
6
|
Project-URL: Documentation, https://gcformat.com/
|
|
@@ -24,15 +24,39 @@ Requires-Python: >=3.9
|
|
|
24
24
|
Description-Content-Type: text/markdown
|
|
25
25
|
|
|
26
26
|
<p align="center">
|
|
27
|
-
<a href="https://
|
|
28
|
-
<a href="
|
|
27
|
+
<a href="https://gcformat.com/playground.html"><img src="https://img.shields.io/badge/playground-live-2563eb?style=for-the-badge" alt="Playground"></a>
|
|
28
|
+
<a href="https://gcformat.com/guide/benchmarks.html"><img src="https://img.shields.io/badge/benchmarks-2%2C500%2B%20evals-22c55e?style=for-the-badge" alt="Benchmarks"></a>
|
|
29
|
+
<a href="https://pypi.org/project/gcf-python/"><img src="https://img.shields.io/pypi/v/gcf-python?style=for-the-badge&logo=python&logoColor=white&color=3776AB" alt="PyPI"></a>
|
|
30
|
+
<a href="https://github.com/blackwell-systems/gcf-python/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-333?style=for-the-badge" alt="License"></a>
|
|
31
|
+
</p>
|
|
32
|
+
|
|
33
|
+
<p align="center">
|
|
34
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-hero-wire-delta.png" alt="gcf-python" width="760">
|
|
29
35
|
</p>
|
|
30
36
|
|
|
31
37
|
# gcf-python
|
|
32
38
|
|
|
33
|
-
Python implementation of [GCF](https://gcformat.com/)
|
|
39
|
+
Python implementation of [GCF](https://gcformat.com/), the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
|
|
40
|
+
|
|
41
|
+
<p align="center">
|
|
42
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider-wave-2.png" alt="" width="100%">
|
|
43
|
+
</p>
|
|
44
|
+
|
|
45
|
+
<p align="center">
|
|
46
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-python-diagram.png" alt="gcf-python" width="80%">
|
|
47
|
+
</p>
|
|
48
|
+
|
|
49
|
+
<p align="center">
|
|
50
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider.png" alt="" width="100%">
|
|
51
|
+
</p>
|
|
52
|
+
|
|
53
|
+
**Built for the agentic loop, where the same structured context crosses the model boundary turn after turn.** A single payload is 50-92% smaller than JSON, but GCF also deduplicates repeated structure across turns and sends only deltas when context changes, so by the 5th overlapping call each response costs 99% fewer tokens than JSON, and a 10-call session runs 94.4% cheaper than re-sending JSON every turn. Session dedup and delta both need local IDs and a multi-turn design that neither JSON nor TOON has.
|
|
54
|
+
|
|
55
|
+
- **100% comprehension on every frontier model**, zero training. 29% fewer tokens than TOON and 56% fewer than JSON across 16 datasets; 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%).
|
|
56
|
+
- **Proven lossless** across 43,000,000,000+ round-trips in 5 formats and 6 languages. Zero runtime dependencies.
|
|
57
|
+
- **One format, four properties no other single format holds at once:** schema-free, lossless, token-compact (50-92% vs JSON), and model-readable with zero training. JSON is verbose, Protobuf needs a schema, MessagePack is binary, and TOON isn't reliably lossless.
|
|
34
58
|
|
|
35
|
-
|
|
59
|
+
2,500+ LLM evaluations. [Full benchmarks](https://gcformat.com/guide/benchmarks.html).
|
|
36
60
|
|
|
37
61
|
Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
|
|
38
62
|
|
|
@@ -105,7 +129,7 @@ out1 = encode_with_session(payload1, sess) # full declarations
|
|
|
105
129
|
out2 = encode_with_session(payload2, sess) # reused symbols as "@N # previously transmitted"
|
|
106
130
|
```
|
|
107
131
|
|
|
108
|
-
By the 5th call in a session:
|
|
132
|
+
By the 5th call in a session: 86% fewer tokens than JSON from dedup alone, 99% stacked with delta encoding.
|
|
109
133
|
|
|
110
134
|
## Streaming Encode
|
|
111
135
|
|
|
@@ -119,7 +143,7 @@ enc = StreamEncoder(sys.stdout, "context_for_task", token_budget=5000)
|
|
|
119
143
|
enc.write_symbol(Symbol(qualified_name="pkg.Auth", kind="function", score=0.95, provenance="lsp", distance=0))
|
|
120
144
|
enc.write_symbol(Symbol(qualified_name="pkg.Server", kind="function", score=0.60, provenance="lsp", distance=1))
|
|
121
145
|
enc.write_edge(Edge(source="pkg.Server", target="pkg.Auth", edge_type="calls"))
|
|
122
|
-
enc.close() # emits
|
|
146
|
+
enc.close() # emits ##! summary trailer
|
|
123
147
|
```
|
|
124
148
|
|
|
125
149
|
Output:
|
|
@@ -131,7 +155,7 @@ GCF tool=context_for_task budget=5000
|
|
|
131
155
|
@1 fn pkg.Server 0.60 lsp
|
|
132
156
|
## edges [?]
|
|
133
157
|
@0<@1 calls
|
|
134
|
-
|
|
158
|
+
##! summary symbols=2 edges=1 counts=1,1,1
|
|
135
159
|
```
|
|
136
160
|
|
|
137
161
|
The writer is any object with a `write(s: str)` method. Thread-safe. Standard `decode()` handles streaming output with no changes.
|
|
@@ -250,7 +274,7 @@ for snapshot in stream: # each turn's current GenericSet
|
|
|
250
274
|
|
|
251
275
|
## Benchmarks
|
|
252
276
|
|
|
253
|
-
2,
|
|
277
|
+
2,500+ LLM evaluations across 11 models, 4 providers, and 50+ independent test runs.
|
|
254
278
|
|
|
255
279
|
| | GCF | TOON | JSON |
|
|
256
280
|
|---|---|---|---|
|
|
@@ -280,11 +304,23 @@ GCF wins 15/16 datasets on the expanded [token efficiency benchmark](https://git
|
|
|
280
304
|
|
|
281
305
|
**Zero runtime dependencies. Permanently.** All six implementations depend only on their language's standard library. No transitive dependencies. No supply chain risk. This is a permanent commitment: GCF will never take on external runtime dependencies. MIT licensed. All implementations support both generic profile (`encodeGeneric`) and graph profile (`encode`). CLI included in all 6 languages.
|
|
282
306
|
|
|
283
|
-
**Specification:** [SPEC v3.
|
|
307
|
+
**Specification:** [SPEC v3.4.1 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 204 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.4.0+ (Go v1.5.0). Cross-language 6x6 matrix verified.
|
|
284
308
|
|
|
285
309
|
## Adopted by
|
|
286
310
|
|
|
287
|
-
|
|
311
|
+
| Project | |
|
|
312
|
+
|---------|--|
|
|
313
|
+
| **[Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp)** | 47K★ · the Google Chrome DevTools team's MCP server; exposes live browser state (DOM, network, console, performance) to AI coding agents |
|
|
314
|
+
| **[Speakeasy](https://speakeasy.com)** | OpenAPI tooling (customers include Google, Verizon, Mistral AI, DocuSign, Vercel); GCF is a native output format in their `oq` CLI |
|
|
315
|
+
| **[OmniRoute](https://omniroute.online)** | 17K★ · AI gateway, registry, and proxy between AI clients and model providers; GCF vendored into its compression engine |
|
|
316
|
+
| **[NetClaw](https://github.com/automateyournetwork/netclaw)** | 610★ · AI-powered network automation (113 skills, 66 MCP integrations); replaced TOON with GCF across every MCP server |
|
|
317
|
+
| **[ctx](https://github.com/stevesolun/ctx)** | 552★ · real-time context selector for Claude Code; surfaces only the relevant tools from a 103K-node knowledge graph |
|
|
318
|
+
| **[Lynkr](https://github.com/Fast-Editor/Lynkr)** | 531★ · local LLM gateway for AI coding clients; GCF as a drop-in tool-result compressor alongside TOON |
|
|
319
|
+
| **[Open Data Products SDK](https://opendataproducts.org/sdk/)** | Linux Foundation · Python toolkit and MCP server for data-product standards; GCF sidecars for agent context |
|
|
320
|
+
| **[NeuroNest](https://neuronest.cc)** | agent-first IDE; first commercial GCF adoption, across four encoding surfaces with session dedup and delta |
|
|
321
|
+
| **[Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter)** | JSON-to-GCF Converter extension in the Raycast Store, for the macOS productivity launcher |
|
|
322
|
+
|
|
323
|
+
[See all adopters →](https://gcformat.com/ecosystem/adopters.html)
|
|
288
324
|
|
|
289
325
|
## License
|
|
290
326
|
|
|
@@ -1,13 +1,37 @@
|
|
|
1
1
|
<p align="center">
|
|
2
|
-
<a href="https://
|
|
3
|
-
<a href="
|
|
2
|
+
<a href="https://gcformat.com/playground.html"><img src="https://img.shields.io/badge/playground-live-2563eb?style=for-the-badge" alt="Playground"></a>
|
|
3
|
+
<a href="https://gcformat.com/guide/benchmarks.html"><img src="https://img.shields.io/badge/benchmarks-2%2C500%2B%20evals-22c55e?style=for-the-badge" alt="Benchmarks"></a>
|
|
4
|
+
<a href="https://pypi.org/project/gcf-python/"><img src="https://img.shields.io/pypi/v/gcf-python?style=for-the-badge&logo=python&logoColor=white&color=3776AB" alt="PyPI"></a>
|
|
5
|
+
<a href="https://github.com/blackwell-systems/gcf-python/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-333?style=for-the-badge" alt="License"></a>
|
|
6
|
+
</p>
|
|
7
|
+
|
|
8
|
+
<p align="center">
|
|
9
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-hero-wire-delta.png" alt="gcf-python" width="760">
|
|
4
10
|
</p>
|
|
5
11
|
|
|
6
12
|
# gcf-python
|
|
7
13
|
|
|
8
|
-
Python implementation of [GCF](https://gcformat.com/)
|
|
14
|
+
Python implementation of [GCF](https://gcformat.com/), the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
|
|
15
|
+
|
|
16
|
+
<p align="center">
|
|
17
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider-wave-2.png" alt="" width="100%">
|
|
18
|
+
</p>
|
|
19
|
+
|
|
20
|
+
<p align="center">
|
|
21
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-python-diagram.png" alt="gcf-python" width="80%">
|
|
22
|
+
</p>
|
|
23
|
+
|
|
24
|
+
<p align="center">
|
|
25
|
+
<img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider.png" alt="" width="100%">
|
|
26
|
+
</p>
|
|
27
|
+
|
|
28
|
+
**Built for the agentic loop, where the same structured context crosses the model boundary turn after turn.** A single payload is 50-92% smaller than JSON, but GCF also deduplicates repeated structure across turns and sends only deltas when context changes, so by the 5th overlapping call each response costs 99% fewer tokens than JSON, and a 10-call session runs 94.4% cheaper than re-sending JSON every turn. Session dedup and delta both need local IDs and a multi-turn design that neither JSON nor TOON has.
|
|
29
|
+
|
|
30
|
+
- **100% comprehension on every frontier model**, zero training. 29% fewer tokens than TOON and 56% fewer than JSON across 16 datasets; 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%).
|
|
31
|
+
- **Proven lossless** across 43,000,000,000+ round-trips in 5 formats and 6 languages. Zero runtime dependencies.
|
|
32
|
+
- **One format, four properties no other single format holds at once:** schema-free, lossless, token-compact (50-92% vs JSON), and model-readable with zero training. JSON is verbose, Protobuf needs a schema, MessagePack is binary, and TOON isn't reliably lossless.
|
|
9
33
|
|
|
10
|
-
|
|
34
|
+
2,500+ LLM evaluations. [Full benchmarks](https://gcformat.com/guide/benchmarks.html).
|
|
11
35
|
|
|
12
36
|
Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
|
|
13
37
|
|
|
@@ -80,7 +104,7 @@ out1 = encode_with_session(payload1, sess) # full declarations
|
|
|
80
104
|
out2 = encode_with_session(payload2, sess) # reused symbols as "@N # previously transmitted"
|
|
81
105
|
```
|
|
82
106
|
|
|
83
|
-
By the 5th call in a session:
|
|
107
|
+
By the 5th call in a session: 86% fewer tokens than JSON from dedup alone, 99% stacked with delta encoding.
|
|
84
108
|
|
|
85
109
|
## Streaming Encode
|
|
86
110
|
|
|
@@ -94,7 +118,7 @@ enc = StreamEncoder(sys.stdout, "context_for_task", token_budget=5000)
|
|
|
94
118
|
enc.write_symbol(Symbol(qualified_name="pkg.Auth", kind="function", score=0.95, provenance="lsp", distance=0))
|
|
95
119
|
enc.write_symbol(Symbol(qualified_name="pkg.Server", kind="function", score=0.60, provenance="lsp", distance=1))
|
|
96
120
|
enc.write_edge(Edge(source="pkg.Server", target="pkg.Auth", edge_type="calls"))
|
|
97
|
-
enc.close() # emits
|
|
121
|
+
enc.close() # emits ##! summary trailer
|
|
98
122
|
```
|
|
99
123
|
|
|
100
124
|
Output:
|
|
@@ -106,7 +130,7 @@ GCF tool=context_for_task budget=5000
|
|
|
106
130
|
@1 fn pkg.Server 0.60 lsp
|
|
107
131
|
## edges [?]
|
|
108
132
|
@0<@1 calls
|
|
109
|
-
|
|
133
|
+
##! summary symbols=2 edges=1 counts=1,1,1
|
|
110
134
|
```
|
|
111
135
|
|
|
112
136
|
The writer is any object with a `write(s: str)` method. Thread-safe. Standard `decode()` handles streaming output with no changes.
|
|
@@ -225,7 +249,7 @@ for snapshot in stream: # each turn's current GenericSet
|
|
|
225
249
|
|
|
226
250
|
## Benchmarks
|
|
227
251
|
|
|
228
|
-
2,
|
|
252
|
+
2,500+ LLM evaluations across 11 models, 4 providers, and 50+ independent test runs.
|
|
229
253
|
|
|
230
254
|
| | GCF | TOON | JSON |
|
|
231
255
|
|---|---|---|---|
|
|
@@ -255,11 +279,23 @@ GCF wins 15/16 datasets on the expanded [token efficiency benchmark](https://git
|
|
|
255
279
|
|
|
256
280
|
**Zero runtime dependencies. Permanently.** All six implementations depend only on their language's standard library. No transitive dependencies. No supply chain risk. This is a permanent commitment: GCF will never take on external runtime dependencies. MIT licensed. All implementations support both generic profile (`encodeGeneric`) and graph profile (`encode`). CLI included in all 6 languages.
|
|
257
281
|
|
|
258
|
-
**Specification:** [SPEC v3.
|
|
282
|
+
**Specification:** [SPEC v3.4.1 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 204 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.4.0+ (Go v1.5.0). Cross-language 6x6 matrix verified.
|
|
259
283
|
|
|
260
284
|
## Adopted by
|
|
261
285
|
|
|
262
|
-
|
|
286
|
+
| Project | |
|
|
287
|
+
|---------|--|
|
|
288
|
+
| **[Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp)** | 47K★ · the Google Chrome DevTools team's MCP server; exposes live browser state (DOM, network, console, performance) to AI coding agents |
|
|
289
|
+
| **[Speakeasy](https://speakeasy.com)** | OpenAPI tooling (customers include Google, Verizon, Mistral AI, DocuSign, Vercel); GCF is a native output format in their `oq` CLI |
|
|
290
|
+
| **[OmniRoute](https://omniroute.online)** | 17K★ · AI gateway, registry, and proxy between AI clients and model providers; GCF vendored into its compression engine |
|
|
291
|
+
| **[NetClaw](https://github.com/automateyournetwork/netclaw)** | 610★ · AI-powered network automation (113 skills, 66 MCP integrations); replaced TOON with GCF across every MCP server |
|
|
292
|
+
| **[ctx](https://github.com/stevesolun/ctx)** | 552★ · real-time context selector for Claude Code; surfaces only the relevant tools from a 103K-node knowledge graph |
|
|
293
|
+
| **[Lynkr](https://github.com/Fast-Editor/Lynkr)** | 531★ · local LLM gateway for AI coding clients; GCF as a drop-in tool-result compressor alongside TOON |
|
|
294
|
+
| **[Open Data Products SDK](https://opendataproducts.org/sdk/)** | Linux Foundation · Python toolkit and MCP server for data-product standards; GCF sidecars for agent context |
|
|
295
|
+
| **[NeuroNest](https://neuronest.cc)** | agent-first IDE; first commercial GCF adoption, across four encoding surfaces with session dedup and delta |
|
|
296
|
+
| **[Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter)** | JSON-to-GCF Converter extension in the Raycast Store, for the macOS productivity launcher |
|
|
297
|
+
|
|
298
|
+
[See all adopters →](https://gcformat.com/ecosystem/adopters.html)
|
|
263
299
|
|
|
264
300
|
## License
|
|
265
301
|
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "gcf-python"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.5.0"
|
|
8
8
|
description = "The AI-native wire format for structured data. 50-92% fewer tokens than JSON, with multi-turn delta encoding for agent loops. 100% comprehension on every frontier model. Zero dependencies."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "MIT"}
|
|
@@ -36,7 +36,7 @@ Specification: https://github.com/blackwell-systems/gcf
|
|
|
36
36
|
|
|
37
37
|
from .constants import KIND_ABBREV, KIND_EXPAND
|
|
38
38
|
from .decode import DecodeError, decode
|
|
39
|
-
from .delta import encode_delta
|
|
39
|
+
from .delta import decode_delta, encode_delta, verify_delta
|
|
40
40
|
from .encode import encode
|
|
41
41
|
from .generic import encode_generic, GenericOptions
|
|
42
42
|
from .generic_delta import (
|
|
@@ -56,6 +56,7 @@ from .generic_delta import (
|
|
|
56
56
|
size_guard,
|
|
57
57
|
DEFAULT_REANCHOR_N,
|
|
58
58
|
)
|
|
59
|
+
from .packroot import pack_root
|
|
59
60
|
from .session import Session, encode_with_session
|
|
60
61
|
from .decode_generic import decode_generic
|
|
61
62
|
from .stream import StreamEncoder
|
|
@@ -75,15 +76,18 @@ __all__ = [
|
|
|
75
76
|
"StreamEncoder",
|
|
76
77
|
"Symbol",
|
|
77
78
|
"decode",
|
|
79
|
+
"decode_delta",
|
|
78
80
|
"decode_generic",
|
|
79
81
|
"encode",
|
|
80
82
|
"encode_delta",
|
|
83
|
+
"verify_delta",
|
|
81
84
|
"encode_generic",
|
|
82
85
|
"GenericOptions",
|
|
83
86
|
"encode_with_session",
|
|
84
87
|
"GenericSet",
|
|
85
88
|
"GenericDeltaPayload",
|
|
86
89
|
"generic_pack_root",
|
|
90
|
+
"pack_root",
|
|
87
91
|
"diff_generic_sets",
|
|
88
92
|
"encode_generic_full",
|
|
89
93
|
"encode_generic_delta",
|
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
from typing import Any
|
|
6
6
|
|
|
7
7
|
from .decode import decode
|
|
8
|
+
from .keyed_map import keyed_rows_to_map
|
|
8
9
|
from .scalar import (
|
|
9
10
|
parse_scalar, parse_quoted_string, split_respecting_quotes, split_field_decl,
|
|
10
11
|
is_bare_key, MISSING, ATTACHMENT,
|
|
@@ -72,7 +73,7 @@ def decode_generic(input_text: str) -> Any:
|
|
|
72
73
|
if trimmed.startswith("##! "):
|
|
73
74
|
summary_line = trimmed
|
|
74
75
|
continue
|
|
75
|
-
if trimmed.startswith("## ") and "[?]" in trimmed:
|
|
76
|
+
if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
|
|
76
77
|
deferred_count += 1
|
|
77
78
|
content_lines.append(line)
|
|
78
79
|
|
|
@@ -134,7 +135,7 @@ def _parse_object_body(
|
|
|
134
135
|
|
|
135
136
|
if content.startswith("## "):
|
|
136
137
|
hdr = content[3:]
|
|
137
|
-
bi = hdr
|
|
138
|
+
bi = _find_bracket_start(hdr)
|
|
138
139
|
if bi >= 0:
|
|
139
140
|
name = _parse_key_from_header(hdr[:bi])
|
|
140
141
|
_check_dup(out, name)
|
|
@@ -177,7 +178,14 @@ def _parse_object_body(
|
|
|
177
178
|
i += 1
|
|
178
179
|
continue
|
|
179
180
|
|
|
180
|
-
|
|
181
|
+
# An object-body line that is not a `## ` section, a `key=value` field, or
|
|
182
|
+
# an inline array is not valid content and MUST NOT be silently skipped
|
|
183
|
+
# (that dropped data, a lossless round-trip hole). A pipe-delimited line is
|
|
184
|
+
# a stray positional inline body with no eligible `^` cell (SPEC 16.5,
|
|
185
|
+
# orphan_inline_attachment); any other unrecognized line is likewise rejected.
|
|
186
|
+
if "|" in content:
|
|
187
|
+
raise ValueError(f"orphan_inline_attachment: {content}")
|
|
188
|
+
raise ValueError(f"invalid_line: unexpected content in object body: {content!r}")
|
|
181
189
|
return i - start
|
|
182
190
|
|
|
183
191
|
|
|
@@ -226,10 +234,24 @@ def _parse_array_from_header(
|
|
|
226
234
|
raise ValueError("invalid_count")
|
|
227
235
|
count_str = bp[1:close]
|
|
228
236
|
after = bp[close + 1:]
|
|
237
|
+
|
|
238
|
+
# A keyed map is marked by `:` after the count inside the bracket (`[N:]`).
|
|
239
|
+
# The decoder reconstructs a JSON object, not an array (SPEC 7.2a.2).
|
|
240
|
+
keyed = count_str.endswith(":")
|
|
241
|
+
if keyed:
|
|
242
|
+
count_str = count_str[:-1]
|
|
243
|
+
if not after.startswith("{"):
|
|
244
|
+
raise ValueError("keyed_map: missing field declaration")
|
|
245
|
+
|
|
229
246
|
count = -1
|
|
230
247
|
if count_str != "?":
|
|
231
248
|
count = _parse_count(count_str)
|
|
232
249
|
|
|
250
|
+
# A keyed map has at least one member; an empty object is encoded per
|
|
251
|
+
# Section 7.7, never as [0:] (SPEC 7.2a.4).
|
|
252
|
+
if keyed and count == 0:
|
|
253
|
+
raise ValueError("keyed_map: zero count [0:] is invalid (an empty object uses Section 7.7)")
|
|
254
|
+
|
|
233
255
|
if count == 0 and not after.startswith("{") and not after.startswith(":"):
|
|
234
256
|
return [], 1
|
|
235
257
|
|
|
@@ -252,6 +274,8 @@ def _parse_array_from_header(
|
|
|
252
274
|
rows, consumed = _parse_tabular_body(lines, header_line + 1, depth, fields, count)
|
|
253
275
|
if count >= 0 and len(rows) != count:
|
|
254
276
|
raise ValueError(f"count_mismatch: declared {count}, got {len(rows)}")
|
|
277
|
+
if keyed:
|
|
278
|
+
return keyed_rows_to_map(rows, fields), consumed + 1
|
|
255
279
|
return rows, consumed + 1
|
|
256
280
|
|
|
257
281
|
items, consumed = _parse_expanded_body(lines, header_line + 1, depth)
|
|
@@ -260,6 +284,27 @@ def _parse_array_from_header(
|
|
|
260
284
|
return items, consumed + 1
|
|
261
285
|
|
|
262
286
|
|
|
287
|
+
def _find_bracket_start(s: str) -> int:
|
|
288
|
+
# Find " [" (the named-array count bracket) that is OUTSIDE any quoted name,
|
|
289
|
+
# so a quoted section/key name containing " [" (e.g. `## "a [1] b"`) is not
|
|
290
|
+
# misread as a named-array header. Mirrors _find_closing_brace's quote tracking.
|
|
291
|
+
in_quote = False
|
|
292
|
+
escaped = False
|
|
293
|
+
for i, c in enumerate(s):
|
|
294
|
+
if escaped:
|
|
295
|
+
escaped = False
|
|
296
|
+
continue
|
|
297
|
+
if c == "\\" and in_quote:
|
|
298
|
+
escaped = True
|
|
299
|
+
continue
|
|
300
|
+
if c == '"':
|
|
301
|
+
in_quote = not in_quote
|
|
302
|
+
continue
|
|
303
|
+
if not in_quote and c == " " and i + 1 < len(s) and s[i + 1] == "[":
|
|
304
|
+
return i
|
|
305
|
+
return -1
|
|
306
|
+
|
|
307
|
+
|
|
263
308
|
def _find_closing_brace(s: str) -> int:
|
|
264
309
|
in_quote = False
|
|
265
310
|
escaped = False
|
|
@@ -521,6 +566,11 @@ def _parse_tabular_body(
|
|
|
521
566
|
if row_has_id:
|
|
522
567
|
inline_idx = 0
|
|
523
568
|
|
|
569
|
+
# Columns that carry a `^` marker cell in this row legitimately expect
|
|
570
|
+
# a `.field` body. Any other `.field` is an orphan (Section 16.5) unless
|
|
571
|
+
# its name contains `>` (the flatten-fallback attachment, Section 7.4.6.1.4).
|
|
572
|
+
expected_att = set(traditional_att_fields) | set(inline_att_fields)
|
|
573
|
+
|
|
524
574
|
while i < len(lines):
|
|
525
575
|
a_line = lines[i]
|
|
526
576
|
a_content: str | None = None
|
|
@@ -541,6 +591,14 @@ def _parse_tabular_body(
|
|
|
541
591
|
att_name, after_name = _parse_attachment_name(rest)
|
|
542
592
|
after_name_stripped = after_name.lstrip()
|
|
543
593
|
|
|
594
|
+
# Orphan attachment: a `.field` with no matching `^` cell in this
|
|
595
|
+
# row is only legitimate for a `>`-named field (Section 7.4.6.1.4).
|
|
596
|
+
# Any other unmatched attachment is rejected rather than silently
|
|
597
|
+
# injected as an undeclared extra field, which would decode to a
|
|
598
|
+
# record no encoder produces (Section 16.5, lossless round-trip).
|
|
599
|
+
if att_name not in expected_att and ">" not in att_name:
|
|
600
|
+
raise ValueError(f"orphan_attachment: {att_name}")
|
|
601
|
+
|
|
544
602
|
# Prefixed inline data.
|
|
545
603
|
ifs = inline_schemas.get(att_name)
|
|
546
604
|
if ifs and not after_name_stripped.startswith("{}") and not after_name_stripped.startswith("["):
|
|
@@ -613,8 +671,22 @@ def _parse_tabular_body(
|
|
|
613
671
|
if extra_name in attachment_values:
|
|
614
672
|
raise ValueError(f"duplicate_attachment: {extra_name}")
|
|
615
673
|
|
|
674
|
+
# Reconstruct the row in declared field-union order. A flattened group is
|
|
675
|
+
# emitted at the position of its first path column, so the nested object
|
|
676
|
+
# reappears where the original field was, not appended at the end (SPEC
|
|
677
|
+
# 7.4.6.1 step 7 and the key-order preservation requirement, SPEC 52, 931).
|
|
678
|
+
nested = _unflatten_paths(path_column_map, flat_values, flat_absent) if path_column_map else {}
|
|
679
|
+
emitted_groups: set[str] = set()
|
|
616
680
|
row: dict[str, Any] = {}
|
|
617
681
|
for f in fields:
|
|
682
|
+
if f in path_column_map:
|
|
683
|
+
top = path_column_map[f][0]
|
|
684
|
+
if top in emitted_groups:
|
|
685
|
+
continue
|
|
686
|
+
emitted_groups.add(top)
|
|
687
|
+
if top in nested: # omitted when the whole group is absent
|
|
688
|
+
row[top] = nested[top]
|
|
689
|
+
continue
|
|
618
690
|
if f in missing_fields:
|
|
619
691
|
continue
|
|
620
692
|
if f in cell_values:
|
|
@@ -625,10 +697,6 @@ def _parse_tabular_body(
|
|
|
625
697
|
for k, v in attachment_values.items():
|
|
626
698
|
if k not in row:
|
|
627
699
|
row[k] = v
|
|
628
|
-
# Unflatten path columns into nested objects.
|
|
629
|
-
if path_column_map:
|
|
630
|
-
nested = _unflatten_paths(path_column_map, flat_values, flat_absent)
|
|
631
|
-
row.update(nested)
|
|
632
700
|
|
|
633
701
|
rows.append(row)
|
|
634
702
|
|
|
@@ -725,7 +793,7 @@ def _validate_summary_counts(
|
|
|
725
793
|
current_count = 0
|
|
726
794
|
for line in content_lines:
|
|
727
795
|
trimmed = line.lstrip()
|
|
728
|
-
if trimmed.startswith("## ") and "[?]" in trimmed:
|
|
796
|
+
if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
|
|
729
797
|
if in_deferred:
|
|
730
798
|
actual_counts.append(current_count)
|
|
731
799
|
in_deferred = True
|