gcf-python 2.4.0__tar.gz → 2.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {gcf_python-2.4.0 → gcf_python-2.5.1}/CHANGELOG.md +16 -0
  2. {gcf_python-2.4.0 → gcf_python-2.5.1}/PKG-INFO +42 -10
  3. {gcf_python-2.4.0 → gcf_python-2.5.1}/README.md +41 -9
  4. gcf_python-2.5.1/assets/divider-wave-2.png +0 -0
  5. gcf_python-2.5.1/assets/divider.png +0 -0
  6. gcf_python-2.5.1/assets/gcf-hero-wire-delta.png +0 -0
  7. {gcf_python-2.4.0 → gcf_python-2.5.1}/pyproject.toml +1 -1
  8. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/decode.py +26 -1
  9. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/decode_generic.py +63 -8
  10. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/generic.py +79 -7
  11. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/generic_delta.py +27 -13
  12. gcf_python-2.5.1/src/gcf/keyed_map.py +86 -0
  13. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/scalar.py +11 -3
  14. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/stream_generic.py +54 -3
  15. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_conformance_v2.py +39 -4
  16. gcf_python-2.5.1/tests/test_keyed_map_fuzz.py +281 -0
  17. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_roundtrip_v2.py +68 -8
  18. gcf_python-2.5.1/tests/test_stream_fielddecl.py +110 -0
  19. {gcf_python-2.4.0 → gcf_python-2.5.1}/.github/FUNDING.yml +0 -0
  20. {gcf_python-2.4.0 → gcf_python-2.5.1}/.github/workflows/ci.yml +0 -0
  21. {gcf_python-2.4.0 → gcf_python-2.5.1}/.github/workflows/publish.yml +0 -0
  22. {gcf_python-2.4.0 → gcf_python-2.5.1}/.gitignore +0 -0
  23. {gcf_python-2.4.0 → gcf_python-2.5.1}/LICENSE +0 -0
  24. {gcf_python-2.4.0 → gcf_python-2.5.1}/assets/gcf-python-diagram.png +0 -0
  25. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/__init__.py +0 -0
  26. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/__main__.py +0 -0
  27. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/cli.py +0 -0
  28. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/constants.py +0 -0
  29. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/delta.py +0 -0
  30. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/encode.py +0 -0
  31. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/packroot.py +0 -0
  32. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/session.py +0 -0
  33. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/stream.py +0 -0
  34. {gcf_python-2.4.0 → gcf_python-2.5.1}/src/gcf/types.py +0 -0
  35. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/__init__.py +0 -0
  36. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_decode.py +0 -0
  37. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_delta.py +0 -0
  38. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_encode.py +0 -0
  39. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_generic.py +0 -0
  40. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_generic_delta.py +0 -0
  41. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_generic_delta_fuzz.py +0 -0
  42. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_generic_delta_session.py +0 -0
  43. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_roundtrip.py +0 -0
  44. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_session.py +0 -0
  45. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_stream.py +0 -0
  46. {gcf_python-2.4.0 → gcf_python-2.5.1}/tests/test_stream_generic.py +0 -0
  47. {gcf_python-2.4.0 → gcf_python-2.5.1}/uv.lock +0 -0
@@ -1,5 +1,21 @@
1
1
  # Changelog
2
2
 
3
+ ## v2.5.1 (2026-08-07)
4
+
5
+ - Decoders now reject a declared `[N]` section count that does not match the actual item count, in both directions, per SPEC Section 13 (Count Validation). A declared count smaller than the rows or entries present was previously read as a limit and the surplus was dropped; it is now an error. Covers the generic tabular, keyed-map, and root-array forms, the delta and full-set decoders, and the graph `## edges [N]` section. Valid payloads and encoding are unchanged.
6
+
7
+ ## v2.5.0 (2026-08-07)
8
+
9
+ ### Added
10
+ - Keyed-tabular map encoding (SPEC 7.2a): a JSON object whose values are all objects forming a tabular set is encoded as a keyed table (`## [N:]{key,...}`) - the shared value fields are declared once, with one key-prefixed row per member. Canonical by default, supported in nested and streaming positions, and integrated with generic delta using the map key as the identity.
11
+
12
+ ### Changed
13
+ - Negative zero is canonicalized to `0` for both integer and floating-point values (SPEC 2.3.1).
14
+ - Canonical-output alignment across all six SDKs: object key ordering, graph header fields, and symbol ordering follow the specification and reference implementation exactly.
15
+
16
+ ### Testing
17
+ - Conformance runners assert re-encode idempotence (`encode(decode(x)) == x`) for the generic, graph, and delta profiles; a differential cross-SDK fuzz was added to the verification suite.
18
+
3
19
  ## v2.4.0 (2026-07-12)
4
20
 
5
21
  ### Fixes
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gcf-python
3
- Version: 2.4.0
3
+ Version: 2.5.1
4
4
  Summary: The AI-native wire format for structured data. 50-92% fewer tokens than JSON, with multi-turn delta encoding for agent loops. 100% comprehension on every frontier model. Zero dependencies.
5
5
  Project-URL: Homepage, https://github.com/blackwell-systems/gcf-python
6
6
  Project-URL: Documentation, https://gcformat.com/
@@ -24,19 +24,39 @@ Requires-Python: >=3.9
24
24
  Description-Content-Type: text/markdown
25
25
 
26
26
  <p align="center">
27
- <img src="assets/gcf-python-diagram.png" alt="gcf-python" width="100%">
27
+ <a href="https://gcformat.com/playground.html"><img src="https://img.shields.io/badge/playground-live-2563eb?style=for-the-badge" alt="Playground"></a>
28
+ <a href="https://gcformat.com/guide/benchmarks.html"><img src="https://img.shields.io/badge/benchmarks-2%2C500%2B%20evals-22c55e?style=for-the-badge" alt="Benchmarks"></a>
29
+ <a href="https://pypi.org/project/gcf-python/"><img src="https://img.shields.io/pypi/v/gcf-python?style=for-the-badge&logo=python&logoColor=white&color=3776AB" alt="PyPI"></a>
30
+ <a href="https://github.com/blackwell-systems/gcf-python/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-333?style=for-the-badge" alt="License"></a>
28
31
  </p>
29
32
 
30
33
  <p align="center">
31
- <a href="https://github.com/blackwell-systems"><img src="https://raw.githubusercontent.com/blackwell-systems/blackwell-docs-theme/main/badge-trademark.svg" alt="Blackwell Systems"></a>
32
- <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="License"></a>
34
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-hero-wire-delta.png" alt="gcf-python" width="760">
33
35
  </p>
34
36
 
35
37
  # gcf-python
36
38
 
37
- Python implementation of [GCF](https://gcformat.com/) — the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
39
+ Python implementation of [GCF](https://gcformat.com/), the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
38
40
 
39
- **100% comprehension on every frontier model tested. 29% fewer tokens than TOON, 56% fewer than JSON across 16 datasets. 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%). 2,400+ LLM evaluations. Zero training.**
41
+ <p align="center">
42
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider-wave-2.png" alt="" width="100%">
43
+ </p>
44
+
45
+ <p align="center">
46
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-python-diagram.png" alt="gcf-python" width="80%">
47
+ </p>
48
+
49
+ <p align="center">
50
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider.png" alt="" width="100%">
51
+ </p>
52
+
53
+ **Built for the agentic loop, where the same structured context crosses the model boundary turn after turn.** A single payload is 50-92% smaller than JSON, but GCF also deduplicates repeated structure across turns and sends only deltas when context changes, so by the 5th overlapping call each response costs 99% fewer tokens than JSON, and a 10-call session runs 94.4% cheaper than re-sending JSON every turn. Session dedup and delta both need local IDs and a multi-turn design that neither JSON nor TOON has.
54
+
55
+ - **100% comprehension on every frontier model**, zero training. 29% fewer tokens than TOON and 56% fewer than JSON across 16 datasets; 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%).
56
+ - **Proven lossless** across 43,000,000,000+ round-trips in 5 formats and 6 languages. Zero runtime dependencies.
57
+ - **One format, four properties no other single format holds at once:** schema-free, lossless, token-compact (50-92% vs JSON), and model-readable with zero training. JSON is verbose, Protobuf needs a schema, MessagePack is binary, and TOON isn't reliably lossless.
58
+
59
+ 2,500+ LLM evaluations. [Full benchmarks](https://gcformat.com/guide/benchmarks.html).
40
60
 
41
61
  Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
42
62
 
@@ -109,7 +129,7 @@ out1 = encode_with_session(payload1, sess) # full declarations
109
129
  out2 = encode_with_session(payload2, sess) # reused symbols as "@N # previously transmitted"
110
130
  ```
111
131
 
112
- By the 5th call in a session: 92.7% token savings vs JSON.
132
+ By the 5th call in a session: 86% fewer tokens than JSON from dedup alone, 99% stacked with delta encoding.
113
133
 
114
134
  ## Streaming Encode
115
135
 
@@ -254,7 +274,7 @@ for snapshot in stream: # each turn's current GenericSet
254
274
 
255
275
  ## Benchmarks
256
276
 
257
- 2,400+ LLM evaluations across 10 models, 3 providers, and 51 independent test runs.
277
+ 2,500+ LLM evaluations across 11 models, 4 providers, and 50+ independent test runs.
258
278
 
259
279
  | | GCF | TOON | JSON |
260
280
  |---|---|---|---|
@@ -284,11 +304,23 @@ GCF wins 15/16 datasets on the expanded [token efficiency benchmark](https://git
284
304
 
285
305
  **Zero runtime dependencies. Permanently.** All six implementations depend only on their language's standard library. No transitive dependencies. No supply chain risk. This is a permanent commitment: GCF will never take on external runtime dependencies. MIT licensed. All implementations support both generic profile (`encodeGeneric`) and graph profile (`encode`). CLI included in all 6 languages.
286
306
 
287
- **Specification:** [SPEC v3.2 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 174 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.2.1+ (Go v1.3.1). Cross-language 6x6 matrix verified.
307
+ **Specification:** [SPEC v3.4.1 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 204 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.4.0+ (Go v1.5.0). Cross-language 6x6 matrix verified.
288
308
 
289
309
  ## Adopted by
290
310
 
291
- [Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp) (46K stars, Google Chrome DevTools team) · [Speakeasy](https://speakeasy.com) (API tooling, customers include Google, Verizon, Mistral AI, DocuSign, Vercel) · [OmniRoute](https://omniroute.online) (6.1K stars) · [NetClaw](https://github.com/automateyournetwork/netclaw) (556 stars) · [ctx](https://github.com/stevesolun/ctx) (510 stars) · [NeuroNest](https://neuronest.cc) · [Open Data Products SDK](https://opendataproducts.org/sdk/) (Linux Foundation) · [Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter) · [and more](https://gcformat.com/ecosystem/adopters.html)
311
+ | Project | |
312
+ |---------|--|
313
+ | **[Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp)** | 47K★ · the Google Chrome DevTools team's MCP server; exposes live browser state (DOM, network, console, performance) to AI coding agents |
314
+ | **[Speakeasy](https://speakeasy.com)** | OpenAPI tooling (customers include Google, Verizon, Mistral AI, DocuSign, Vercel); GCF is a native output format in their `oq` CLI |
315
+ | **[OmniRoute](https://omniroute.online)** | 17K★ · AI gateway, registry, and proxy between AI clients and model providers; GCF vendored into its compression engine |
316
+ | **[NetClaw](https://github.com/automateyournetwork/netclaw)** | 610★ · AI-powered network automation (113 skills, 66 MCP integrations); replaced TOON with GCF across every MCP server |
317
+ | **[ctx](https://github.com/stevesolun/ctx)** | 552★ · real-time context selector for Claude Code; surfaces only the relevant tools from a 103K-node knowledge graph |
318
+ | **[Lynkr](https://github.com/Fast-Editor/Lynkr)** | 531★ · local LLM gateway for AI coding clients; GCF as a drop-in tool-result compressor alongside TOON |
319
+ | **[Open Data Products SDK](https://opendataproducts.org/sdk/)** | Linux Foundation · Python toolkit and MCP server for data-product standards; GCF sidecars for agent context |
320
+ | **[NeuroNest](https://neuronest.cc)** | agent-first IDE; first commercial GCF adoption, across four encoding surfaces with session dedup and delta |
321
+ | **[Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter)** | JSON-to-GCF Converter extension in the Raycast Store, for the macOS productivity launcher |
322
+
323
+ [See all adopters →](https://gcformat.com/ecosystem/adopters.html)
292
324
 
293
325
  ## License
294
326
 
@@ -1,17 +1,37 @@
1
1
  <p align="center">
2
- <img src="assets/gcf-python-diagram.png" alt="gcf-python" width="100%">
2
+ <a href="https://gcformat.com/playground.html"><img src="https://img.shields.io/badge/playground-live-2563eb?style=for-the-badge" alt="Playground"></a>
3
+ <a href="https://gcformat.com/guide/benchmarks.html"><img src="https://img.shields.io/badge/benchmarks-2%2C500%2B%20evals-22c55e?style=for-the-badge" alt="Benchmarks"></a>
4
+ <a href="https://pypi.org/project/gcf-python/"><img src="https://img.shields.io/pypi/v/gcf-python?style=for-the-badge&logo=python&logoColor=white&color=3776AB" alt="PyPI"></a>
5
+ <a href="https://github.com/blackwell-systems/gcf-python/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-333?style=for-the-badge" alt="License"></a>
3
6
  </p>
4
7
 
5
8
  <p align="center">
6
- <a href="https://github.com/blackwell-systems"><img src="https://raw.githubusercontent.com/blackwell-systems/blackwell-docs-theme/main/badge-trademark.svg" alt="Blackwell Systems"></a>
7
- <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="License"></a>
9
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-hero-wire-delta.png" alt="gcf-python" width="760">
8
10
  </p>
9
11
 
10
12
  # gcf-python
11
13
 
12
- Python implementation of [GCF](https://gcformat.com/) — the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
14
+ Python implementation of [GCF](https://gcformat.com/), the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
13
15
 
14
- **100% comprehension on every frontier model tested. 29% fewer tokens than TOON, 56% fewer than JSON across 16 datasets. 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%). 2,400+ LLM evaluations. Zero training.**
16
+ <p align="center">
17
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider-wave-2.png" alt="" width="100%">
18
+ </p>
19
+
20
+ <p align="center">
21
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-python-diagram.png" alt="gcf-python" width="80%">
22
+ </p>
23
+
24
+ <p align="center">
25
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider.png" alt="" width="100%">
26
+ </p>
27
+
28
+ **Built for the agentic loop, where the same structured context crosses the model boundary turn after turn.** A single payload is 50-92% smaller than JSON, but GCF also deduplicates repeated structure across turns and sends only deltas when context changes, so by the 5th overlapping call each response costs 99% fewer tokens than JSON, and a 10-call session runs 94.4% cheaper than re-sending JSON every turn. Session dedup and delta both need local IDs and a multi-turn design that neither JSON nor TOON has.
29
+
30
+ - **100% comprehension on every frontier model**, zero training. 29% fewer tokens than TOON and 56% fewer than JSON across 16 datasets; 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%).
31
+ - **Proven lossless** across 43,000,000,000+ round-trips in 5 formats and 6 languages. Zero runtime dependencies.
32
+ - **One format, four properties no other single format holds at once:** schema-free, lossless, token-compact (50-92% vs JSON), and model-readable with zero training. JSON is verbose, Protobuf needs a schema, MessagePack is binary, and TOON isn't reliably lossless.
33
+
34
+ 2,500+ LLM evaluations. [Full benchmarks](https://gcformat.com/guide/benchmarks.html).
15
35
 
16
36
  Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
17
37
 
@@ -84,7 +104,7 @@ out1 = encode_with_session(payload1, sess) # full declarations
84
104
  out2 = encode_with_session(payload2, sess) # reused symbols as "@N # previously transmitted"
85
105
  ```
86
106
 
87
- By the 5th call in a session: 92.7% token savings vs JSON.
107
+ By the 5th call in a session: 86% fewer tokens than JSON from dedup alone, 99% stacked with delta encoding.
88
108
 
89
109
  ## Streaming Encode
90
110
 
@@ -229,7 +249,7 @@ for snapshot in stream: # each turn's current GenericSet
229
249
 
230
250
  ## Benchmarks
231
251
 
232
- 2,400+ LLM evaluations across 10 models, 3 providers, and 51 independent test runs.
252
+ 2,500+ LLM evaluations across 11 models, 4 providers, and 50+ independent test runs.
233
253
 
234
254
  | | GCF | TOON | JSON |
235
255
  |---|---|---|---|
@@ -259,11 +279,23 @@ GCF wins 15/16 datasets on the expanded [token efficiency benchmark](https://git
259
279
 
260
280
  **Zero runtime dependencies. Permanently.** All six implementations depend only on their language's standard library. No transitive dependencies. No supply chain risk. This is a permanent commitment: GCF will never take on external runtime dependencies. MIT licensed. All implementations support both generic profile (`encodeGeneric`) and graph profile (`encode`). CLI included in all 6 languages.
261
281
 
262
- **Specification:** [SPEC v3.2 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 174 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.2.1+ (Go v1.3.1). Cross-language 6x6 matrix verified.
282
+ **Specification:** [SPEC v3.4.1 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 204 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.4.0+ (Go v1.5.0). Cross-language 6x6 matrix verified.
263
283
 
264
284
  ## Adopted by
265
285
 
266
- [Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp) (46K stars, Google Chrome DevTools team) · [Speakeasy](https://speakeasy.com) (API tooling, customers include Google, Verizon, Mistral AI, DocuSign, Vercel) · [OmniRoute](https://omniroute.online) (6.1K stars) · [NetClaw](https://github.com/automateyournetwork/netclaw) (556 stars) · [ctx](https://github.com/stevesolun/ctx) (510 stars) · [NeuroNest](https://neuronest.cc) · [Open Data Products SDK](https://opendataproducts.org/sdk/) (Linux Foundation) · [Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter) · [and more](https://gcformat.com/ecosystem/adopters.html)
286
+ | Project | |
287
+ |---------|--|
288
+ | **[Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp)** | 47K★ · the Google Chrome DevTools team's MCP server; exposes live browser state (DOM, network, console, performance) to AI coding agents |
289
+ | **[Speakeasy](https://speakeasy.com)** | OpenAPI tooling (customers include Google, Verizon, Mistral AI, DocuSign, Vercel); GCF is a native output format in their `oq` CLI |
290
+ | **[OmniRoute](https://omniroute.online)** | 17K★ · AI gateway, registry, and proxy between AI clients and model providers; GCF vendored into its compression engine |
291
+ | **[NetClaw](https://github.com/automateyournetwork/netclaw)** | 610★ · AI-powered network automation (113 skills, 66 MCP integrations); replaced TOON with GCF across every MCP server |
292
+ | **[ctx](https://github.com/stevesolun/ctx)** | 552★ · real-time context selector for Claude Code; surfaces only the relevant tools from a 103K-node knowledge graph |
293
+ | **[Lynkr](https://github.com/Fast-Editor/Lynkr)** | 531★ · local LLM gateway for AI coding clients; GCF as a drop-in tool-result compressor alongside TOON |
294
+ | **[Open Data Products SDK](https://opendataproducts.org/sdk/)** | Linux Foundation · Python toolkit and MCP server for data-product standards; GCF sidecars for agent context |
295
+ | **[NeuroNest](https://neuronest.cc)** | agent-first IDE; first commercial GCF adoption, across four encoding surfaces with session dedup and delta |
296
+ | **[Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter)** | JSON-to-GCF Converter extension in the Raycast Store, for the macOS productivity launcher |
297
+
298
+ [See all adopters →](https://gcformat.com/ecosystem/adopters.html)
267
299
 
268
300
  ## License
269
301
 
Binary file
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "gcf-python"
7
- version = "2.4.0"
7
+ version = "2.5.1"
8
8
  description = "The AI-native wire format for structured data. 50-92% fewer tokens than JSON, with multi-turn delta encoding for agent loops. 100% comprehension on every frontier model. Zero dependencies."
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
@@ -45,6 +45,8 @@ def decode(input_text: str) -> Payload:
45
45
  sym_by_id: dict[int, Symbol] = {}
46
46
  current_distance = 0
47
47
  in_edges = False
48
+ declared_edges = -1
49
+ edges_declared = False
48
50
 
49
51
  for line in lines[1:]:
50
52
  line = line.rstrip("\r")
@@ -58,13 +60,29 @@ def decode(input_text: str) -> Payload:
58
60
  # Group header.
59
61
  if line.startswith("## "):
60
62
  group = line[3:]
61
- # Strip bracket suffix: "edges [200]" -> "edges"
63
+ # Strip bracket suffix: "edges [200]" -> "edges", capturing the
64
+ # declared count so it can be enforced per Section 13.
65
+ declared_count = -1
62
66
  bracket_idx = group.find(" [")
63
67
  if bracket_idx >= 0:
68
+ bracket = group[bracket_idx + 2:]
64
69
  group = group[:bracket_idx]
70
+ end = bracket.find("]")
71
+ if end >= 0:
72
+ cnt_str = bracket[:end]
73
+ if cnt_str != "?": # "[?]" is a streaming deferred count (Section 8)
74
+ try:
75
+ declared_count = int(cnt_str)
76
+ except ValueError:
77
+ raise DecodeError(
78
+ f"count_mismatch: invalid section count {cnt_str!r}"
79
+ )
65
80
  if is_delta and group not in valid_delta_sections:
66
81
  raise DecodeError(f"malformed_delta: invalid delta section {group!r}")
67
82
  in_edges = group == "edges"
83
+ if in_edges and declared_count >= 0:
84
+ declared_edges = declared_count
85
+ edges_declared = True
68
86
  if not in_edges:
69
87
  if group == "targets":
70
88
  current_distance = 0
@@ -91,6 +109,13 @@ def decode(input_text: str) -> Payload:
91
109
  symbols.append(sym)
92
110
  sym_by_id[sym_id] = sym
93
111
 
112
+ # Section 13: a declared [N] section count MUST match the actual item count.
113
+ # The graph edges section is the graph profile's only [N]-bearing section.
114
+ if edges_declared and len(p.edges) != declared_edges:
115
+ raise DecodeError(
116
+ f"count_mismatch: declared {declared_edges} edges, got {len(p.edges)}"
117
+ )
118
+
94
119
  p.symbols = symbols
95
120
  return p
96
121
 
@@ -5,6 +5,7 @@ from __future__ import annotations
5
5
  from typing import Any
6
6
 
7
7
  from .decode import decode
8
+ from .keyed_map import keyed_rows_to_map
8
9
  from .scalar import (
9
10
  parse_scalar, parse_quoted_string, split_respecting_quotes, split_field_decl,
10
11
  is_bare_key, MISSING, ATTACHMENT,
@@ -72,7 +73,7 @@ def decode_generic(input_text: str) -> Any:
72
73
  if trimmed.startswith("##! "):
73
74
  summary_line = trimmed
74
75
  continue
75
- if trimmed.startswith("## ") and "[?]" in trimmed:
76
+ if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
76
77
  deferred_count += 1
77
78
  content_lines.append(line)
78
79
 
@@ -90,7 +91,14 @@ def decode_generic(input_text: str) -> Any:
90
91
  return parse_scalar(first[1:])
91
92
 
92
93
  if first.startswith("## ["):
93
- arr, _ = _parse_array_from_header(content_lines, 0, 0, first[3:])
94
+ arr, consumed = _parse_array_from_header(content_lines, 0, 0, first[3:])
95
+ # A root array or keyed map spans the whole document, so any structural line
96
+ # past the consumed rows is a surplus item, not sibling content. The row loop
97
+ # stops at the declared count, so the count assert only catches the deficit
98
+ # case; surplus is caught here (SPEC Section 13: a mismatch, fewer OR more
99
+ # items than declared, is an error).
100
+ if consumed < len(content_lines):
101
+ raise ValueError("count_mismatch: declared count is fewer than the rows present")
94
102
  return arr
95
103
 
96
104
  result: dict[str, Any] = {}
@@ -134,7 +142,7 @@ def _parse_object_body(
134
142
 
135
143
  if content.startswith("## "):
136
144
  hdr = content[3:]
137
- bi = hdr.find(" [")
145
+ bi = _find_bracket_start(hdr)
138
146
  if bi >= 0:
139
147
  name = _parse_key_from_header(hdr[:bi])
140
148
  _check_dup(out, name)
@@ -233,10 +241,24 @@ def _parse_array_from_header(
233
241
  raise ValueError("invalid_count")
234
242
  count_str = bp[1:close]
235
243
  after = bp[close + 1:]
244
+
245
+ # A keyed map is marked by `:` after the count inside the bracket (`[N:]`).
246
+ # The decoder reconstructs a JSON object, not an array (SPEC 7.2a.2).
247
+ keyed = count_str.endswith(":")
248
+ if keyed:
249
+ count_str = count_str[:-1]
250
+ if not after.startswith("{"):
251
+ raise ValueError("keyed_map: missing field declaration")
252
+
236
253
  count = -1
237
254
  if count_str != "?":
238
255
  count = _parse_count(count_str)
239
256
 
257
+ # A keyed map has at least one member; an empty object is encoded per
258
+ # Section 7.7, never as [0:] (SPEC 7.2a.4).
259
+ if keyed and count == 0:
260
+ raise ValueError("keyed_map: zero count [0:] is invalid (an empty object uses Section 7.7)")
261
+
240
262
  if count == 0 and not after.startswith("{") and not after.startswith(":"):
241
263
  return [], 1
242
264
 
@@ -259,6 +281,8 @@ def _parse_array_from_header(
259
281
  rows, consumed = _parse_tabular_body(lines, header_line + 1, depth, fields, count)
260
282
  if count >= 0 and len(rows) != count:
261
283
  raise ValueError(f"count_mismatch: declared {count}, got {len(rows)}")
284
+ if keyed:
285
+ return keyed_rows_to_map(rows, fields), consumed + 1
262
286
  return rows, consumed + 1
263
287
 
264
288
  items, consumed = _parse_expanded_body(lines, header_line + 1, depth)
@@ -267,6 +291,27 @@ def _parse_array_from_header(
267
291
  return items, consumed + 1
268
292
 
269
293
 
294
+ def _find_bracket_start(s: str) -> int:
295
+ # Find " [" (the named-array count bracket) that is OUTSIDE any quoted name,
296
+ # so a quoted section/key name containing " [" (e.g. `## "a [1] b"`) is not
297
+ # misread as a named-array header. Mirrors _find_closing_brace's quote tracking.
298
+ in_quote = False
299
+ escaped = False
300
+ for i, c in enumerate(s):
301
+ if escaped:
302
+ escaped = False
303
+ continue
304
+ if c == "\\" and in_quote:
305
+ escaped = True
306
+ continue
307
+ if c == '"':
308
+ in_quote = not in_quote
309
+ continue
310
+ if not in_quote and c == " " and i + 1 < len(s) and s[i + 1] == "[":
311
+ return i
312
+ return -1
313
+
314
+
270
315
  def _find_closing_brace(s: str) -> int:
271
316
  in_quote = False
272
317
  escaped = False
@@ -633,8 +678,22 @@ def _parse_tabular_body(
633
678
  if extra_name in attachment_values:
634
679
  raise ValueError(f"duplicate_attachment: {extra_name}")
635
680
 
681
+ # Reconstruct the row in declared field-union order. A flattened group is
682
+ # emitted at the position of its first path column, so the nested object
683
+ # reappears where the original field was, not appended at the end (SPEC
684
+ # 7.4.6.1 step 7 and the key-order preservation requirement, SPEC 52, 931).
685
+ nested = _unflatten_paths(path_column_map, flat_values, flat_absent) if path_column_map else {}
686
+ emitted_groups: set[str] = set()
636
687
  row: dict[str, Any] = {}
637
688
  for f in fields:
689
+ if f in path_column_map:
690
+ top = path_column_map[f][0]
691
+ if top in emitted_groups:
692
+ continue
693
+ emitted_groups.add(top)
694
+ if top in nested: # omitted when the whole group is absent
695
+ row[top] = nested[top]
696
+ continue
638
697
  if f in missing_fields:
639
698
  continue
640
699
  if f in cell_values:
@@ -645,10 +704,6 @@ def _parse_tabular_body(
645
704
  for k, v in attachment_values.items():
646
705
  if k not in row:
647
706
  row[k] = v
648
- # Unflatten path columns into nested objects.
649
- if path_column_map:
650
- nested = _unflatten_paths(path_column_map, flat_values, flat_absent)
651
- row.update(nested)
652
707
 
653
708
  rows.append(row)
654
709
 
@@ -745,7 +800,7 @@ def _validate_summary_counts(
745
800
  current_count = 0
746
801
  for line in content_lines:
747
802
  trimmed = line.lstrip()
748
- if trimmed.startswith("## ") and "[?]" in trimmed:
803
+ if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
749
804
  if in_deferred:
750
805
  actual_counts.append(current_count)
751
806
  in_deferred = True
@@ -6,6 +6,7 @@ from dataclasses import dataclass
6
6
  from typing import Any
7
7
 
8
8
  from .scalar import format_scalar, format_key
9
+ from .keyed_map import keyed_map_eligible
9
10
 
10
11
 
11
12
  @dataclass
@@ -30,6 +31,11 @@ def _encode_root_value(v: Any, out: list[str], opts: GenericOptions) -> None:
30
31
  if v is None:
31
32
  out.append("=-")
32
33
  elif isinstance(v, dict):
34
+ km = keyed_map_eligible(v)
35
+ if km is not None:
36
+ keys, values, value_fields, key_label = km
37
+ _encode_keyed_map("", False, keys, values, value_fields, key_label, out, 0, opts)
38
+ return
33
39
  _encode_object(v, out, 0, opts)
34
40
  elif isinstance(v, list):
35
41
  _encode_root_array(v, out, opts)
@@ -42,6 +48,11 @@ def _encode_object(d: dict, out: list[str], depth: int, opts: GenericOptions) ->
42
48
  for key, value in d.items():
43
49
  fk = format_key(key)
44
50
  if isinstance(value, dict):
51
+ km = keyed_map_eligible(value)
52
+ if km is not None:
53
+ keys, values, value_fields, key_label = km
54
+ _encode_keyed_map(key, True, keys, values, value_fields, key_label, out, depth, opts)
55
+ continue
45
56
  out.append(f"{prefix}## {fk}")
46
57
  _encode_object(value, out, depth + 1, opts)
47
58
  elif isinstance(value, list):
@@ -165,8 +176,9 @@ def _analyze_flattenable(
165
176
  arr: list[dict], field_name: str, parent_path: str
166
177
  ) -> list[dict] | None:
167
178
  """Analyze whether a field can be flattened. Returns list of leaf descriptors or None."""
168
- # Field names containing ">" cannot be flattened (would create ambiguous paths).
169
- if ">" in field_name:
179
+ # A field name that is empty or contains ">" cannot be flattened: it would create an
180
+ # ambiguous path column the decoder treats as literal (SPEC 7.4.6.1.3).
181
+ if field_name == "" or ">" in field_name:
170
182
  return None
171
183
  canonical_shape: dict[str, str] | None = None # key -> "scalar" | "nested"
172
184
 
@@ -192,7 +204,7 @@ def _analyze_flattenable(
192
204
  if canonical_shape is None:
193
205
  canonical_shape = {}
194
206
  for k in keys:
195
- if ">" in k:
207
+ if k == "" or ">" in k: # empty/">" -> ambiguous path (SPEC 7.4.6.1.3)
196
208
  return None
197
209
  val = v[k]
198
210
  if isinstance(val, list):
@@ -272,8 +284,53 @@ def _resolve_key_chain(item: Any, keys: list[str]) -> tuple[Any, bool]:
272
284
  return current, True
273
285
 
274
286
 
287
+ # ── Keyed map encoding (SPEC 7.2a) ───────────────────────────────────────
288
+
289
+
290
+ def _keyed_header_prefix(name: str, named: bool, depth: int) -> str:
291
+ """Build the keyed-table header prefix up to the count bracket. named
292
+ distinguishes an anonymous root keyed map (`## `) from a named member whose
293
+ name may itself be the empty string (`## ""`), which format_key quotes so it
294
+ round-trips as a distinct level rather than collapsing into the anonymous
295
+ root form (SPEC 7.2a.1)."""
296
+ prefix = _indent(depth)
297
+ if not named:
298
+ return f"{prefix}## "
299
+ return f"{prefix}## {format_key(name)} "
300
+
301
+
302
+ def _encode_keyed_map(
303
+ name: str, named: bool, keys: list[str], values: list[Any],
304
+ value_fields: list[str], key_label: str, out: list[str], depth: int, opts: GenericOptions
305
+ ) -> None:
306
+ """Emit a keyed table for a map of objects. Routes through _encode_tabular
307
+ with the keyed bracket so nested-value handling (flatten/inline/attachment/
308
+ null/absent) is inherited unchanged. name is empty for a root/anonymous map."""
309
+ _encode_keyed_map_with_prefix(
310
+ _keyed_header_prefix(name, named, depth), keys, values,
311
+ value_fields, key_label, out, depth, opts,
312
+ )
313
+
314
+
315
+ def _encode_keyed_map_with_prefix(
316
+ header_prefix: str, keys: list[str], values: list[Any],
317
+ value_fields: list[str], key_label: str, out: list[str], depth: int, opts: GenericOptions
318
+ ) -> None:
319
+ """Emit `<header_prefix>[N:]{...}` and the keyed rows, reusing _encode_tabular.
320
+ Each value object is augmented with the key column and encoded as a tabular
321
+ row; the key column is declared first."""
322
+ fields = [key_label] + value_fields
323
+ arr: list[dict] = []
324
+ for k, v in zip(keys, values):
325
+ aug = dict(v)
326
+ aug[key_label] = k
327
+ arr.append(aug)
328
+ _encode_tabular(header_prefix, arr, fields, out, depth, opts, keyed=True)
329
+
330
+
275
331
  def _encode_tabular(
276
- header_prefix: str, arr: list[dict], fields: list[str], out: list[str], depth: int, opts: GenericOptions
332
+ header_prefix: str, arr: list[dict], fields: list[str], out: list[str], depth: int,
333
+ opts: GenericOptions, keyed: bool = False
277
334
  ) -> None:
278
335
  prefix = _indent(depth)
279
336
 
@@ -320,7 +377,8 @@ def _encode_tabular(
320
377
  shared_arr_schemas[f] = sas
321
378
 
322
379
  header_fields = ",".join(col["header"] for col in columns)
323
- out.append(f"{header_prefix}[{len(arr)}]{{{header_fields}}}")
380
+ br = ":]" if keyed else "]"
381
+ out.append(f"{header_prefix}[{len(arr)}{br}{{{header_fields}}}")
324
382
 
325
383
  for i, item in enumerate(arr):
326
384
  cells: list[str] = []
@@ -401,8 +459,15 @@ def _encode_tabular(
401
459
  else:
402
460
  _encode_attachment_array(prefix, fk, att_val, out, depth + 2, opts)
403
461
  elif isinstance(att_val, dict):
404
- out.append(f"{prefix}.{fk} {{}}")
405
- _encode_object(att_val, out, depth + 2, opts)
462
+ km = keyed_map_eligible(att_val)
463
+ if km is not None:
464
+ keys, values, value_fields, key_label = km
465
+ _encode_keyed_map_with_prefix(
466
+ f"{prefix}.{fk} ", keys, values, value_fields, key_label, out, depth + 2, opts,
467
+ )
468
+ else:
469
+ out.append(f"{prefix}.{fk} {{}}")
470
+ _encode_object(att_val, out, depth + 2, opts)
406
471
  else:
407
472
  # Scalar attachment (e.g. field names containing ">").
408
473
  if att_val is None:
@@ -467,6 +532,13 @@ def _encode_expanded(header_prefix: str, arr: list, out: list[str], depth: int,
467
532
  out.append(f"{header_prefix}[{len(arr)}]")
468
533
  for i, item in enumerate(arr):
469
534
  if isinstance(item, dict):
535
+ km = keyed_map_eligible(item)
536
+ if km is not None:
537
+ keys, values, value_fields, key_label = km
538
+ _encode_keyed_map_with_prefix(
539
+ f"{prefix}@{i} ", keys, values, value_fields, key_label, out, depth + 1, opts,
540
+ )
541
+ continue
470
542
  out.append(f"{prefix}@{i} {{}}")
471
543
  _encode_object(item, out, depth + 1, opts)
472
544
  elif isinstance(item, list):
@@ -299,16 +299,23 @@ def decode_generic_full(text: str) -> tuple[GenericSet, str]:
299
299
  while i < len(lines):
300
300
  line = lines[i]
301
301
  if not line.startswith("## "):
302
- i += 1
303
- continue
302
+ # Only blank lines, comments, and the ##! summary trailer are valid
303
+ # outside a section; any other line is a surplus row past a declared
304
+ # section count (Section 13).
305
+ if line == "" or line.startswith("# ") or line.startswith("##! "):
306
+ i += 1
307
+ continue
308
+ raise ValueError(
309
+ f"count_mismatch: unexpected content after declared section rows: {line!r}"
310
+ )
304
311
  name, count, fields, key_field = _parse_section_header(line[3:])
305
312
  s.name, s.fields = name, fields
306
313
  if not s.key:
307
314
  s.key = key_field
308
315
  i += 1
309
- for _ in range(count):
310
- if i >= len(lines):
311
- raise ValueError("delta_invalid: fewer rows than declared count")
316
+ for j in range(count):
317
+ if i >= len(lines) or lines[i].startswith("## "):
318
+ raise ValueError(f"count_mismatch: declared {count} rows, got {j}")
312
319
  s.rows.append(_parse_row(lines[i], fields))
313
320
  i += 1
314
321
  return s, hdr.get("pack_root", "")
@@ -335,8 +342,15 @@ def decode_generic_delta(text: str) -> GenericDeltaPayload:
335
342
  while i < len(lines):
336
343
  line = lines[i]
337
344
  if not line.startswith("## "):
338
- i += 1
339
- continue
345
+ # Only blank lines, comments, and the ##! summary trailer are valid
346
+ # outside a section; any other line is a surplus row past a declared
347
+ # section count (Section 13).
348
+ if line == "" or line.startswith("# ") or line.startswith("##! "):
349
+ i += 1
350
+ continue
351
+ raise ValueError(
352
+ f"count_mismatch: unexpected content after declared section rows: {line!r}"
353
+ )
340
354
  name, count, fields, key_field = _parse_section_header(line[3:])
341
355
  if not d.key and key_field:
342
356
  d.key = key_field
@@ -345,9 +359,9 @@ def decode_generic_delta(text: str) -> GenericDeltaPayload:
345
359
  i += 1
346
360
  if name in ("added", "changed"):
347
361
  rows = []
348
- for _ in range(count):
349
- if i >= len(lines):
350
- raise ValueError(f"delta_invalid: fewer rows than declared count in ## {name}")
362
+ for j in range(count):
363
+ if i >= len(lines) or lines[i].startswith("## "):
364
+ raise ValueError(f"count_mismatch: declared {count} rows in ## {name}, got {j}")
351
365
  rows.append(_parse_row(lines[i], fields))
352
366
  i += 1
353
367
  if name == "added":
@@ -355,9 +369,9 @@ def decode_generic_delta(text: str) -> GenericDeltaPayload:
355
369
  else:
356
370
  d.changed = rows
357
371
  elif name == "removed":
358
- for _ in range(count):
359
- if i >= len(lines):
360
- raise ValueError("delta_invalid: fewer identities than declared count in ## removed")
372
+ for j in range(count):
373
+ if i >= len(lines) or lines[i].startswith("## "):
374
+ raise ValueError(f"count_mismatch: declared {count} identities in ## removed, got {j}")
361
375
  d.removed.append(parse_scalar(lines[i], True))
362
376
  i += 1
363
377
  else: