gcf-python 2.4.0__py3-none-any.whl → 2.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
gcf/decode.py CHANGED
@@ -45,6 +45,8 @@ def decode(input_text: str) -> Payload:
45
45
  sym_by_id: dict[int, Symbol] = {}
46
46
  current_distance = 0
47
47
  in_edges = False
48
+ declared_edges = -1
49
+ edges_declared = False
48
50
 
49
51
  for line in lines[1:]:
50
52
  line = line.rstrip("\r")
@@ -58,13 +60,29 @@ def decode(input_text: str) -> Payload:
58
60
  # Group header.
59
61
  if line.startswith("## "):
60
62
  group = line[3:]
61
- # Strip bracket suffix: "edges [200]" -> "edges"
63
+ # Strip bracket suffix: "edges [200]" -> "edges", capturing the
64
+ # declared count so it can be enforced per Section 13.
65
+ declared_count = -1
62
66
  bracket_idx = group.find(" [")
63
67
  if bracket_idx >= 0:
68
+ bracket = group[bracket_idx + 2:]
64
69
  group = group[:bracket_idx]
70
+ end = bracket.find("]")
71
+ if end >= 0:
72
+ cnt_str = bracket[:end]
73
+ if cnt_str != "?": # "[?]" is a streaming deferred count (Section 8)
74
+ try:
75
+ declared_count = int(cnt_str)
76
+ except ValueError:
77
+ raise DecodeError(
78
+ f"count_mismatch: invalid section count {cnt_str!r}"
79
+ )
65
80
  if is_delta and group not in valid_delta_sections:
66
81
  raise DecodeError(f"malformed_delta: invalid delta section {group!r}")
67
82
  in_edges = group == "edges"
83
+ if in_edges and declared_count >= 0:
84
+ declared_edges = declared_count
85
+ edges_declared = True
68
86
  if not in_edges:
69
87
  if group == "targets":
70
88
  current_distance = 0
@@ -91,6 +109,13 @@ def decode(input_text: str) -> Payload:
91
109
  symbols.append(sym)
92
110
  sym_by_id[sym_id] = sym
93
111
 
112
+ # Section 13: a declared [N] section count MUST match the actual item count.
113
+ # The graph edges section is the graph profile's only [N]-bearing section.
114
+ if edges_declared and len(p.edges) != declared_edges:
115
+ raise DecodeError(
116
+ f"count_mismatch: declared {declared_edges} edges, got {len(p.edges)}"
117
+ )
118
+
94
119
  p.symbols = symbols
95
120
  return p
96
121
 
gcf/decode_generic.py CHANGED
@@ -5,6 +5,7 @@ from __future__ import annotations
5
5
  from typing import Any
6
6
 
7
7
  from .decode import decode
8
+ from .keyed_map import keyed_rows_to_map
8
9
  from .scalar import (
9
10
  parse_scalar, parse_quoted_string, split_respecting_quotes, split_field_decl,
10
11
  is_bare_key, MISSING, ATTACHMENT,
@@ -72,7 +73,7 @@ def decode_generic(input_text: str) -> Any:
72
73
  if trimmed.startswith("##! "):
73
74
  summary_line = trimmed
74
75
  continue
75
- if trimmed.startswith("## ") and "[?]" in trimmed:
76
+ if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
76
77
  deferred_count += 1
77
78
  content_lines.append(line)
78
79
 
@@ -90,7 +91,14 @@ def decode_generic(input_text: str) -> Any:
90
91
  return parse_scalar(first[1:])
91
92
 
92
93
  if first.startswith("## ["):
93
- arr, _ = _parse_array_from_header(content_lines, 0, 0, first[3:])
94
+ arr, consumed = _parse_array_from_header(content_lines, 0, 0, first[3:])
95
+ # A root array or keyed map spans the whole document, so any structural line
96
+ # past the consumed rows is a surplus item, not sibling content. The row loop
97
+ # stops at the declared count, so the count assert only catches the deficit
98
+ # case; surplus is caught here (SPEC Section 13: a mismatch, fewer OR more
99
+ # items than declared, is an error).
100
+ if consumed < len(content_lines):
101
+ raise ValueError("count_mismatch: declared count is fewer than the rows present")
94
102
  return arr
95
103
 
96
104
  result: dict[str, Any] = {}
@@ -134,7 +142,7 @@ def _parse_object_body(
134
142
 
135
143
  if content.startswith("## "):
136
144
  hdr = content[3:]
137
- bi = hdr.find(" [")
145
+ bi = _find_bracket_start(hdr)
138
146
  if bi >= 0:
139
147
  name = _parse_key_from_header(hdr[:bi])
140
148
  _check_dup(out, name)
@@ -233,10 +241,24 @@ def _parse_array_from_header(
233
241
  raise ValueError("invalid_count")
234
242
  count_str = bp[1:close]
235
243
  after = bp[close + 1:]
244
+
245
+ # A keyed map is marked by `:` after the count inside the bracket (`[N:]`).
246
+ # The decoder reconstructs a JSON object, not an array (SPEC 7.2a.2).
247
+ keyed = count_str.endswith(":")
248
+ if keyed:
249
+ count_str = count_str[:-1]
250
+ if not after.startswith("{"):
251
+ raise ValueError("keyed_map: missing field declaration")
252
+
236
253
  count = -1
237
254
  if count_str != "?":
238
255
  count = _parse_count(count_str)
239
256
 
257
+ # A keyed map has at least one member; an empty object is encoded per
258
+ # Section 7.7, never as [0:] (SPEC 7.2a.4).
259
+ if keyed and count == 0:
260
+ raise ValueError("keyed_map: zero count [0:] is invalid (an empty object uses Section 7.7)")
261
+
240
262
  if count == 0 and not after.startswith("{") and not after.startswith(":"):
241
263
  return [], 1
242
264
 
@@ -259,6 +281,8 @@ def _parse_array_from_header(
259
281
  rows, consumed = _parse_tabular_body(lines, header_line + 1, depth, fields, count)
260
282
  if count >= 0 and len(rows) != count:
261
283
  raise ValueError(f"count_mismatch: declared {count}, got {len(rows)}")
284
+ if keyed:
285
+ return keyed_rows_to_map(rows, fields), consumed + 1
262
286
  return rows, consumed + 1
263
287
 
264
288
  items, consumed = _parse_expanded_body(lines, header_line + 1, depth)
@@ -267,6 +291,27 @@ def _parse_array_from_header(
267
291
  return items, consumed + 1
268
292
 
269
293
 
294
+ def _find_bracket_start(s: str) -> int:
295
+ # Find " [" (the named-array count bracket) that is OUTSIDE any quoted name,
296
+ # so a quoted section/key name containing " [" (e.g. `## "a [1] b"`) is not
297
+ # misread as a named-array header. Mirrors _find_closing_brace's quote tracking.
298
+ in_quote = False
299
+ escaped = False
300
+ for i, c in enumerate(s):
301
+ if escaped:
302
+ escaped = False
303
+ continue
304
+ if c == "\\" and in_quote:
305
+ escaped = True
306
+ continue
307
+ if c == '"':
308
+ in_quote = not in_quote
309
+ continue
310
+ if not in_quote and c == " " and i + 1 < len(s) and s[i + 1] == "[":
311
+ return i
312
+ return -1
313
+
314
+
270
315
  def _find_closing_brace(s: str) -> int:
271
316
  in_quote = False
272
317
  escaped = False
@@ -633,8 +678,22 @@ def _parse_tabular_body(
633
678
  if extra_name in attachment_values:
634
679
  raise ValueError(f"duplicate_attachment: {extra_name}")
635
680
 
681
+ # Reconstruct the row in declared field-union order. A flattened group is
682
+ # emitted at the position of its first path column, so the nested object
683
+ # reappears where the original field was, not appended at the end (SPEC
684
+ # 7.4.6.1 step 7 and the key-order preservation requirement, SPEC 52, 931).
685
+ nested = _unflatten_paths(path_column_map, flat_values, flat_absent) if path_column_map else {}
686
+ emitted_groups: set[str] = set()
636
687
  row: dict[str, Any] = {}
637
688
  for f in fields:
689
+ if f in path_column_map:
690
+ top = path_column_map[f][0]
691
+ if top in emitted_groups:
692
+ continue
693
+ emitted_groups.add(top)
694
+ if top in nested: # omitted when the whole group is absent
695
+ row[top] = nested[top]
696
+ continue
638
697
  if f in missing_fields:
639
698
  continue
640
699
  if f in cell_values:
@@ -645,10 +704,6 @@ def _parse_tabular_body(
645
704
  for k, v in attachment_values.items():
646
705
  if k not in row:
647
706
  row[k] = v
648
- # Unflatten path columns into nested objects.
649
- if path_column_map:
650
- nested = _unflatten_paths(path_column_map, flat_values, flat_absent)
651
- row.update(nested)
652
707
 
653
708
  rows.append(row)
654
709
 
@@ -745,7 +800,7 @@ def _validate_summary_counts(
745
800
  current_count = 0
746
801
  for line in content_lines:
747
802
  trimmed = line.lstrip()
748
- if trimmed.startswith("## ") and "[?]" in trimmed:
803
+ if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
749
804
  if in_deferred:
750
805
  actual_counts.append(current_count)
751
806
  in_deferred = True
gcf/generic.py CHANGED
@@ -6,6 +6,7 @@ from dataclasses import dataclass
6
6
  from typing import Any
7
7
 
8
8
  from .scalar import format_scalar, format_key
9
+ from .keyed_map import keyed_map_eligible
9
10
 
10
11
 
11
12
  @dataclass
@@ -30,6 +31,11 @@ def _encode_root_value(v: Any, out: list[str], opts: GenericOptions) -> None:
30
31
  if v is None:
31
32
  out.append("=-")
32
33
  elif isinstance(v, dict):
34
+ km = keyed_map_eligible(v)
35
+ if km is not None:
36
+ keys, values, value_fields, key_label = km
37
+ _encode_keyed_map("", False, keys, values, value_fields, key_label, out, 0, opts)
38
+ return
33
39
  _encode_object(v, out, 0, opts)
34
40
  elif isinstance(v, list):
35
41
  _encode_root_array(v, out, opts)
@@ -42,6 +48,11 @@ def _encode_object(d: dict, out: list[str], depth: int, opts: GenericOptions) ->
42
48
  for key, value in d.items():
43
49
  fk = format_key(key)
44
50
  if isinstance(value, dict):
51
+ km = keyed_map_eligible(value)
52
+ if km is not None:
53
+ keys, values, value_fields, key_label = km
54
+ _encode_keyed_map(key, True, keys, values, value_fields, key_label, out, depth, opts)
55
+ continue
45
56
  out.append(f"{prefix}## {fk}")
46
57
  _encode_object(value, out, depth + 1, opts)
47
58
  elif isinstance(value, list):
@@ -165,8 +176,9 @@ def _analyze_flattenable(
165
176
  arr: list[dict], field_name: str, parent_path: str
166
177
  ) -> list[dict] | None:
167
178
  """Analyze whether a field can be flattened. Returns list of leaf descriptors or None."""
168
- # Field names containing ">" cannot be flattened (would create ambiguous paths).
169
- if ">" in field_name:
179
+ # A field name that is empty or contains ">" cannot be flattened: it would create an
180
+ # ambiguous path column the decoder treats as literal (SPEC 7.4.6.1.3).
181
+ if field_name == "" or ">" in field_name:
170
182
  return None
171
183
  canonical_shape: dict[str, str] | None = None # key -> "scalar" | "nested"
172
184
 
@@ -192,7 +204,7 @@ def _analyze_flattenable(
192
204
  if canonical_shape is None:
193
205
  canonical_shape = {}
194
206
  for k in keys:
195
- if ">" in k:
207
+ if k == "" or ">" in k: # empty/">" -> ambiguous path (SPEC 7.4.6.1.3)
196
208
  return None
197
209
  val = v[k]
198
210
  if isinstance(val, list):
@@ -272,8 +284,53 @@ def _resolve_key_chain(item: Any, keys: list[str]) -> tuple[Any, bool]:
272
284
  return current, True
273
285
 
274
286
 
287
+ # ── Keyed map encoding (SPEC 7.2a) ───────────────────────────────────────
288
+
289
+
290
+ def _keyed_header_prefix(name: str, named: bool, depth: int) -> str:
291
+ """Build the keyed-table header prefix up to the count bracket. named
292
+ distinguishes an anonymous root keyed map (`## `) from a named member whose
293
+ name may itself be the empty string (`## ""`), which format_key quotes so it
294
+ round-trips as a distinct level rather than collapsing into the anonymous
295
+ root form (SPEC 7.2a.1)."""
296
+ prefix = _indent(depth)
297
+ if not named:
298
+ return f"{prefix}## "
299
+ return f"{prefix}## {format_key(name)} "
300
+
301
+
302
+ def _encode_keyed_map(
303
+ name: str, named: bool, keys: list[str], values: list[Any],
304
+ value_fields: list[str], key_label: str, out: list[str], depth: int, opts: GenericOptions
305
+ ) -> None:
306
+ """Emit a keyed table for a map of objects. Routes through _encode_tabular
307
+ with the keyed bracket so nested-value handling (flatten/inline/attachment/
308
+ null/absent) is inherited unchanged. name is empty for a root/anonymous map."""
309
+ _encode_keyed_map_with_prefix(
310
+ _keyed_header_prefix(name, named, depth), keys, values,
311
+ value_fields, key_label, out, depth, opts,
312
+ )
313
+
314
+
315
+ def _encode_keyed_map_with_prefix(
316
+ header_prefix: str, keys: list[str], values: list[Any],
317
+ value_fields: list[str], key_label: str, out: list[str], depth: int, opts: GenericOptions
318
+ ) -> None:
319
+ """Emit `<header_prefix>[N:]{...}` and the keyed rows, reusing _encode_tabular.
320
+ Each value object is augmented with the key column and encoded as a tabular
321
+ row; the key column is declared first."""
322
+ fields = [key_label] + value_fields
323
+ arr: list[dict] = []
324
+ for k, v in zip(keys, values):
325
+ aug = dict(v)
326
+ aug[key_label] = k
327
+ arr.append(aug)
328
+ _encode_tabular(header_prefix, arr, fields, out, depth, opts, keyed=True)
329
+
330
+
275
331
  def _encode_tabular(
276
- header_prefix: str, arr: list[dict], fields: list[str], out: list[str], depth: int, opts: GenericOptions
332
+ header_prefix: str, arr: list[dict], fields: list[str], out: list[str], depth: int,
333
+ opts: GenericOptions, keyed: bool = False
277
334
  ) -> None:
278
335
  prefix = _indent(depth)
279
336
 
@@ -320,7 +377,8 @@ def _encode_tabular(
320
377
  shared_arr_schemas[f] = sas
321
378
 
322
379
  header_fields = ",".join(col["header"] for col in columns)
323
- out.append(f"{header_prefix}[{len(arr)}]{{{header_fields}}}")
380
+ br = ":]" if keyed else "]"
381
+ out.append(f"{header_prefix}[{len(arr)}{br}{{{header_fields}}}")
324
382
 
325
383
  for i, item in enumerate(arr):
326
384
  cells: list[str] = []
@@ -401,8 +459,15 @@ def _encode_tabular(
401
459
  else:
402
460
  _encode_attachment_array(prefix, fk, att_val, out, depth + 2, opts)
403
461
  elif isinstance(att_val, dict):
404
- out.append(f"{prefix}.{fk} {{}}")
405
- _encode_object(att_val, out, depth + 2, opts)
462
+ km = keyed_map_eligible(att_val)
463
+ if km is not None:
464
+ keys, values, value_fields, key_label = km
465
+ _encode_keyed_map_with_prefix(
466
+ f"{prefix}.{fk} ", keys, values, value_fields, key_label, out, depth + 2, opts,
467
+ )
468
+ else:
469
+ out.append(f"{prefix}.{fk} {{}}")
470
+ _encode_object(att_val, out, depth + 2, opts)
406
471
  else:
407
472
  # Scalar attachment (e.g. field names containing ">").
408
473
  if att_val is None:
@@ -467,6 +532,13 @@ def _encode_expanded(header_prefix: str, arr: list, out: list[str], depth: int,
467
532
  out.append(f"{header_prefix}[{len(arr)}]")
468
533
  for i, item in enumerate(arr):
469
534
  if isinstance(item, dict):
535
+ km = keyed_map_eligible(item)
536
+ if km is not None:
537
+ keys, values, value_fields, key_label = km
538
+ _encode_keyed_map_with_prefix(
539
+ f"{prefix}@{i} ", keys, values, value_fields, key_label, out, depth + 1, opts,
540
+ )
541
+ continue
470
542
  out.append(f"{prefix}@{i} {{}}")
471
543
  _encode_object(item, out, depth + 1, opts)
472
544
  elif isinstance(item, list):
gcf/generic_delta.py CHANGED
@@ -299,16 +299,23 @@ def decode_generic_full(text: str) -> tuple[GenericSet, str]:
299
299
  while i < len(lines):
300
300
  line = lines[i]
301
301
  if not line.startswith("## "):
302
- i += 1
303
- continue
302
+ # Only blank lines, comments, and the ##! summary trailer are valid
303
+ # outside a section; any other line is a surplus row past a declared
304
+ # section count (Section 13).
305
+ if line == "" or line.startswith("# ") or line.startswith("##! "):
306
+ i += 1
307
+ continue
308
+ raise ValueError(
309
+ f"count_mismatch: unexpected content after declared section rows: {line!r}"
310
+ )
304
311
  name, count, fields, key_field = _parse_section_header(line[3:])
305
312
  s.name, s.fields = name, fields
306
313
  if not s.key:
307
314
  s.key = key_field
308
315
  i += 1
309
- for _ in range(count):
310
- if i >= len(lines):
311
- raise ValueError("delta_invalid: fewer rows than declared count")
316
+ for j in range(count):
317
+ if i >= len(lines) or lines[i].startswith("## "):
318
+ raise ValueError(f"count_mismatch: declared {count} rows, got {j}")
312
319
  s.rows.append(_parse_row(lines[i], fields))
313
320
  i += 1
314
321
  return s, hdr.get("pack_root", "")
@@ -335,8 +342,15 @@ def decode_generic_delta(text: str) -> GenericDeltaPayload:
335
342
  while i < len(lines):
336
343
  line = lines[i]
337
344
  if not line.startswith("## "):
338
- i += 1
339
- continue
345
+ # Only blank lines, comments, and the ##! summary trailer are valid
346
+ # outside a section; any other line is a surplus row past a declared
347
+ # section count (Section 13).
348
+ if line == "" or line.startswith("# ") or line.startswith("##! "):
349
+ i += 1
350
+ continue
351
+ raise ValueError(
352
+ f"count_mismatch: unexpected content after declared section rows: {line!r}"
353
+ )
340
354
  name, count, fields, key_field = _parse_section_header(line[3:])
341
355
  if not d.key and key_field:
342
356
  d.key = key_field
@@ -345,9 +359,9 @@ def decode_generic_delta(text: str) -> GenericDeltaPayload:
345
359
  i += 1
346
360
  if name in ("added", "changed"):
347
361
  rows = []
348
- for _ in range(count):
349
- if i >= len(lines):
350
- raise ValueError(f"delta_invalid: fewer rows than declared count in ## {name}")
362
+ for j in range(count):
363
+ if i >= len(lines) or lines[i].startswith("## "):
364
+ raise ValueError(f"count_mismatch: declared {count} rows in ## {name}, got {j}")
351
365
  rows.append(_parse_row(lines[i], fields))
352
366
  i += 1
353
367
  if name == "added":
@@ -355,9 +369,9 @@ def decode_generic_delta(text: str) -> GenericDeltaPayload:
355
369
  else:
356
370
  d.changed = rows
357
371
  elif name == "removed":
358
- for _ in range(count):
359
- if i >= len(lines):
360
- raise ValueError("delta_invalid: fewer identities than declared count in ## removed")
372
+ for j in range(count):
373
+ if i >= len(lines) or lines[i].startswith("## "):
374
+ raise ValueError(f"count_mismatch: declared {count} identities in ## removed, got {j}")
361
375
  d.removed.append(parse_scalar(lines[i], True))
362
376
  i += 1
363
377
  else:
gcf/keyed_map.py ADDED
@@ -0,0 +1,86 @@
1
+ """GCF keyed-tabular map encoding (SPEC 7.2a).
2
+
3
+ A JSON object whose values are all objects forming a losslessly-tabular set is
4
+ encoded as a keyed table `## [N:]{key,...}`: the shared value fields are declared
5
+ once in a header and each member is one positional row prefixed by its key. This
6
+ is the object-valued analogue of Section 7.4 tabular array encoding. It is
7
+ canonical (default-on): eligible maps always encode as keyed tables, with no
8
+ option to disable.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from typing import Any
14
+
15
+
16
+ def keyed_map_eligible(m: Any) -> tuple[list[str], list[Any], list[str], str] | None:
17
+ """Report whether an object is a keyed map of objects that should render as a
18
+ keyed table `## [N:]{key,...}` (SPEC 7.2a.1). Returns the ordered member keys,
19
+ the corresponding value objects, the ordered value-field union, and the
20
+ key-column label, or None when the object is not eligible.
21
+ """
22
+ if not isinstance(m, dict):
23
+ return None
24
+
25
+ keys = list(m.keys())
26
+ values = [m[k] for k in keys]
27
+
28
+ # A keyed map requires at least two members: the form factors the shared value
29
+ # fields into one header, which only pays off across multiple members. A
30
+ # single-member map yields a one-row table the same size as a section, so keying
31
+ # it would change canonical output for every nested single-member object (e.g.
32
+ # `{"data": {...}}` wrappers) with no benefit. Single-member objects use ordinary
33
+ # encoding; a single-key wrapper of a multi-member map therefore defers, and the
34
+ # inner map is keyed at its own level (SPEC 7.2a.1).
35
+ if len(keys) < 2:
36
+ return None
37
+
38
+ # Every value must be an object; build the ordered field union.
39
+ seen: set[str] = set()
40
+ value_fields: list[str] = []
41
+ for v in values:
42
+ if not isinstance(v, dict):
43
+ return None # non-object value
44
+ for f in v:
45
+ if f not in seen:
46
+ seen.add(f)
47
+ value_fields.append(f)
48
+ if not value_fields:
49
+ return None # all-empty value objects
50
+
51
+ # A keyed header needs at least one value field that can be a tabular column.
52
+ # A field name containing ">" cannot be a column (SPEC 7.4.6.1.4); if every
53
+ # value field contains ">", the keyed form would have only the key column,
54
+ # which is invalid. Such a map uses Section 7.2 section encoding instead, the
55
+ # object analogue of an array falling back to expanded form.
56
+ if not any(">" not in f for f in value_fields):
57
+ return None
58
+
59
+ # Key-column label: "key", made unique by prepending "_" on collision.
60
+ key_label = "key"
61
+ while key_label in seen:
62
+ key_label = "_" + key_label
63
+
64
+ return keys, values, value_fields, key_label
65
+
66
+
67
+ def keyed_rows_to_map(rows: list[Any], fields: list[str]) -> dict[str, Any]:
68
+ """Reconstruct the map from decoded keyed-table rows: the first declared field
69
+ is the member key; the remaining fields form the value object (SPEC 7.2a.4).
70
+ """
71
+ if len(fields) < 2:
72
+ raise ValueError("keyed_map: header must declare at least two fields")
73
+ key_label = fields[0]
74
+ out: dict[str, Any] = {}
75
+ for r in rows:
76
+ if not isinstance(r, dict):
77
+ raise ValueError("keyed_map: row is not an object")
78
+ if key_label not in r:
79
+ raise ValueError(f"keyed_map: row missing key column {key_label!r}")
80
+ kv = r[key_label]
81
+ ks = kv if isinstance(kv, str) else str(kv)
82
+ if ks in out:
83
+ raise ValueError(f"keyed_map: duplicate member key {ks!r}")
84
+ value = {k: v for k, v in r.items() if k != key_label}
85
+ out[ks] = value
86
+ return out
gcf/scalar.py CHANGED
@@ -6,10 +6,13 @@ import math
6
6
  import re
7
7
  from typing import Any
8
8
 
9
- _JSON_NUMBER_RE = re.compile(r"^-?(?:0|[1-9]\d*)(?:\.\d+)?(?:[eE][+-]?\d+)?$")
9
+ # \Z (end of string), not $, so a trailing newline does not count as the end:
10
+ # in Python $ also matches just before a final \n, which would misclassify a
11
+ # string like "5\n" as a number or "W\n" as a bare key and break round-trip.
12
+ _JSON_NUMBER_RE = re.compile(r"^-?(?:0|[1-9]\d*)(?:\.\d+)?(?:[eE][+-]?\d+)?\Z")
10
13
  _NUMERIC_LIKE_RE = re.compile(r"^[+-]\.?\d|^\.\d|^0\d")
11
14
  _INLINE_ARRAY_RE = re.compile(r"\[[^\]]*\]\s*:")
12
- _BARE_KEY_RE = re.compile(r"^[a-zA-Z_][a-zA-Z0-9_]*$")
15
+ _BARE_KEY_RE = re.compile(r"^[a-zA-Z_][a-zA-Z0-9_]*\Z")
13
16
 
14
17
 
15
18
  class _Missing:
@@ -29,6 +32,10 @@ def needs_quote(s: str) -> bool:
29
32
  return True
30
33
  if s in ("-", "~", "^", "true", "false"):
31
34
  return True
35
+ # A value shaped like an inline-schema attachment marker (^{...}) would decode
36
+ # as an attachment and lose the string, so it must be quoted (SPEC 2.4).
37
+ if len(s) >= 3 and s[0] == "^" and s[1] == "{" and s[-1] == "}":
38
+ return True
32
39
  if _JSON_NUMBER_RE.match(s):
33
40
  return True
34
41
  if _NUMERIC_LIKE_RE.match(s):
@@ -99,7 +106,8 @@ def format_number(f: float) -> str:
99
106
  if math.isinf(f):
100
107
  return "0"
101
108
  if f == 0.0:
102
- return "-0" if math.copysign(1.0, f) < 0 else "0"
109
+ # Negative zero canonicalizes to 0 (SPEC 2.3.1): -0.0 equals 0.0 by value.
110
+ return "0"
103
111
  a = abs(f)
104
112
  if 1e-6 <= a < 1e21:
105
113
  # Use repr for shortest round-trippable form.
gcf/stream_generic.py CHANGED
@@ -4,7 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  import threading
6
6
 
7
- from .scalar import format_scalar
7
+ from .scalar import format_scalar, format_key
8
8
  from typing import Any, Sequence
9
9
 
10
10
 
@@ -29,15 +29,52 @@ class GenericStreamEncoder:
29
29
  self._lock = threading.Lock()
30
30
  self._sections: list[tuple[str, int]] = []
31
31
  self._current: dict[str, Any] | None = None
32
+ self._err: Exception | None = None
33
+ self._w.write("GCF profile=generic\n")
32
34
 
33
35
  def begin_array(self, name: str, fields: Sequence[str]) -> None:
34
36
  """Start a tabular array section with deferred count [?]."""
35
37
  with self._lock:
38
+ if self._err is not None:
39
+ return
36
40
  if self._current is not None:
37
41
  self._end_array_locked()
38
- self._w.write(f"## {name} [?]{{{','.join(fields)}}}\n")
42
+ # A streaming tabular row has only flat columns; a field name containing
43
+ # ">" is a flattened path the stream cannot represent (SPEC 8.3, 7.4.6).
44
+ # Record the error and surface it at close().
45
+ for f in fields:
46
+ if ">" in f:
47
+ self._err = ValueError(
48
+ f"streaming field name {f!r} contains '>' "
49
+ "(a flattened path is not representable in a streaming row)"
50
+ )
51
+ return
52
+ self._w.write(f"## {format_key(name)} [?]{{{_format_field_decl(fields)}}}\n")
39
53
  self._current = {"name": name, "fields": list(fields), "count": 0}
40
54
 
55
+ def begin_keyed_map(self, name: str, key_label: str, value_fields: Sequence[str]) -> None:
56
+ """Start a keyed-map section with deferred count [?:] (SPEC 7.2a).
57
+
58
+ key_label is the key column; value_fields are the value-object fields.
59
+ Each write_row value slice is [key_value, *value_fields]."""
60
+ with self._lock:
61
+ if self._err is not None:
62
+ return
63
+ if self._current is not None:
64
+ self._end_array_locked()
65
+ # A streaming value field name containing ">" is a flattened path a
66
+ # stream cannot represent (SPEC 8.3, 7.4.6). Record and surface at close().
67
+ for f in value_fields:
68
+ if ">" in f:
69
+ self._err = ValueError(
70
+ f"streaming field name {f!r} contains '>' "
71
+ "(a flattened path is not representable in a streaming row)"
72
+ )
73
+ return
74
+ fields = [key_label] + list(value_fields)
75
+ self._w.write(f"## {format_key(name)} [?:]{{{_format_field_decl(fields)}}}\n")
76
+ self._current = {"name": name, "fields": fields, "count": 0}
77
+
41
78
  def write_row(self, values: Sequence[Any]) -> None:
42
79
  """Emit a single pipe-separated row immediately."""
43
80
  with self._lock:
@@ -71,8 +108,14 @@ class GenericStreamEncoder:
71
108
  self._w.write(f"{name}[{len(values)}]: {','.join(parts)}\n")
72
109
 
73
110
  def close(self) -> None:
74
- """Emit the ##! summary trailer with final counts."""
111
+ """Emit the ##! summary trailer with final counts.
112
+
113
+ Raises any error recorded during encoding (e.g. a field name containing
114
+ ">", which is not representable in a flat streaming row per SPEC 8.3).
115
+ """
75
116
  with self._lock:
117
+ if self._err is not None:
118
+ raise self._err
76
119
  if self._current is not None:
77
120
  self._end_array_locked()
78
121
  if not self._sections:
@@ -87,5 +130,13 @@ class GenericStreamEncoder:
87
130
  self._current = None
88
131
 
89
132
 
133
+ def _format_field_decl(fields: Sequence[str]) -> str:
134
+ """Quote each field name per Section 2.4 (via format_key), matching the
135
+ buffered tabular header. The streaming header previously joined field names
136
+ raw, so a name containing a delimiter or quote produced an invalid or
137
+ ambiguous field declaration (SPEC 8.3)."""
138
+ return ",".join(format_key(f) for f in fields)
139
+
140
+
90
141
  def _format_value(v: Any) -> str:
91
142
  return format_scalar(v, "|")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gcf-python
3
- Version: 2.4.0
3
+ Version: 2.5.1
4
4
  Summary: The AI-native wire format for structured data. 50-92% fewer tokens than JSON, with multi-turn delta encoding for agent loops. 100% comprehension on every frontier model. Zero dependencies.
5
5
  Project-URL: Homepage, https://github.com/blackwell-systems/gcf-python
6
6
  Project-URL: Documentation, https://gcformat.com/
@@ -24,19 +24,39 @@ Requires-Python: >=3.9
24
24
  Description-Content-Type: text/markdown
25
25
 
26
26
  <p align="center">
27
- <img src="assets/gcf-python-diagram.png" alt="gcf-python" width="100%">
27
+ <a href="https://gcformat.com/playground.html"><img src="https://img.shields.io/badge/playground-live-2563eb?style=for-the-badge" alt="Playground"></a>
28
+ <a href="https://gcformat.com/guide/benchmarks.html"><img src="https://img.shields.io/badge/benchmarks-2%2C500%2B%20evals-22c55e?style=for-the-badge" alt="Benchmarks"></a>
29
+ <a href="https://pypi.org/project/gcf-python/"><img src="https://img.shields.io/pypi/v/gcf-python?style=for-the-badge&logo=python&logoColor=white&color=3776AB" alt="PyPI"></a>
30
+ <a href="https://github.com/blackwell-systems/gcf-python/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-333?style=for-the-badge" alt="License"></a>
28
31
  </p>
29
32
 
30
33
  <p align="center">
31
- <a href="https://github.com/blackwell-systems"><img src="https://raw.githubusercontent.com/blackwell-systems/blackwell-docs-theme/main/badge-trademark.svg" alt="Blackwell Systems"></a>
32
- <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="License"></a>
34
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-hero-wire-delta.png" alt="gcf-python" width="760">
33
35
  </p>
34
36
 
35
37
  # gcf-python
36
38
 
37
- Python implementation of [GCF](https://gcformat.com/) — the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
39
+ Python implementation of [GCF](https://gcformat.com/), the most token-efficient wire format for LLMs. A drop-in alternative to JSON and TOON for any structured data.
38
40
 
39
- **100% comprehension on every frontier model tested. 29% fewer tokens than TOON, 56% fewer than JSON across 16 datasets. 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%). 2,400+ LLM evaluations. Zero training.**
41
+ <p align="center">
42
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider-wave-2.png" alt="" width="100%">
43
+ </p>
44
+
45
+ <p align="center">
46
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/gcf-python-diagram.png" alt="gcf-python" width="80%">
47
+ </p>
48
+
49
+ <p align="center">
50
+ <img src="https://raw.githubusercontent.com/blackwell-systems/gcf-python/main/assets/divider.png" alt="" width="100%">
51
+ </p>
52
+
53
+ **Built for the agentic loop, where the same structured context crosses the model boundary turn after turn.** A single payload is 50-92% smaller than JSON, but GCF also deduplicates repeated structure across turns and sends only deltas when context changes, so by the 5th overlapping call each response costs 99% fewer tokens than JSON, and a 10-call session runs 94.4% cheaper than re-sending JSON every turn. Session dedup and delta both need local IDs and a multi-turn design that neither JSON nor TOON has.
54
+
55
+ - **100% comprehension on every frontier model**, zero training. 29% fewer tokens than TOON and 56% fewer than JSON across 16 datasets; 91.2% on structurally complex code graphs (vs TOON 68.8%, JSON 54.1%).
56
+ - **Proven lossless** across 43,000,000,000+ round-trips in 5 formats and 6 languages. Zero runtime dependencies.
57
+ - **One format, four properties no other single format holds at once:** schema-free, lossless, token-compact (50-92% vs JSON), and model-readable with zero training. JSON is verbose, Protobuf needs a schema, MessagePack is binary, and TOON isn't reliably lossless.
58
+
59
+ 2,500+ LLM evaluations. [Full benchmarks](https://gcformat.com/guide/benchmarks.html).
40
60
 
41
61
  Docs: [gcformat.com](https://gcformat.com/) · [Playground](https://gcformat.com/playground.html) · [GCF vs TOON](https://gcformat.com/guide/vs-toon.html)
42
62
 
@@ -109,7 +129,7 @@ out1 = encode_with_session(payload1, sess) # full declarations
109
129
  out2 = encode_with_session(payload2, sess) # reused symbols as "@N # previously transmitted"
110
130
  ```
111
131
 
112
- By the 5th call in a session: 92.7% token savings vs JSON.
132
+ By the 5th call in a session: 86% fewer tokens than JSON from dedup alone, 99% stacked with delta encoding.
113
133
 
114
134
  ## Streaming Encode
115
135
 
@@ -254,7 +274,7 @@ for snapshot in stream: # each turn's current GenericSet
254
274
 
255
275
  ## Benchmarks
256
276
 
257
- 2,400+ LLM evaluations across 10 models, 3 providers, and 51 independent test runs.
277
+ 2,500+ LLM evaluations across 11 models, 4 providers, and 50+ independent test runs.
258
278
 
259
279
  | | GCF | TOON | JSON |
260
280
  |---|---|---|---|
@@ -284,11 +304,23 @@ GCF wins 15/16 datasets on the expanded [token efficiency benchmark](https://git
284
304
 
285
305
  **Zero runtime dependencies. Permanently.** All six implementations depend only on their language's standard library. No transitive dependencies. No supply chain risk. This is a permanent commitment: GCF will never take on external runtime dependencies. MIT licensed. All implementations support both generic profile (`encodeGeneric`) and graph profile (`encode`). CLI included in all 6 languages.
286
306
 
287
- **Specification:** [SPEC v3.2 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 174 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.2.1+ (Go v1.3.1). Cross-language 6x6 matrix verified.
307
+ **Specification:** [SPEC v3.4.1 Stable](https://github.com/blackwell-systems/gcf/blob/main/SPEC.md) with 204 conformance fixtures, 43,000,000,000+ lossless round-trips verified across 5 formats and 6 languages. All implementations at v2.4.0+ (Go v1.5.0). Cross-language 6x6 matrix verified.
288
308
 
289
309
  ## Adopted by
290
310
 
291
- [Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp) (46K stars, Google Chrome DevTools team) · [Speakeasy](https://speakeasy.com) (API tooling, customers include Google, Verizon, Mistral AI, DocuSign, Vercel) · [OmniRoute](https://omniroute.online) (6.1K stars) · [NetClaw](https://github.com/automateyournetwork/netclaw) (556 stars) · [ctx](https://github.com/stevesolun/ctx) (510 stars) · [NeuroNest](https://neuronest.cc) · [Open Data Products SDK](https://opendataproducts.org/sdk/) (Linux Foundation) · [Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter) · [and more](https://gcformat.com/ecosystem/adopters.html)
311
+ | Project | |
312
+ |---------|--|
313
+ | **[Chrome DevTools MCP](https://github.com/ChromeDevTools/chrome-devtools-mcp)** | 47K★ · the Google Chrome DevTools team's MCP server; exposes live browser state (DOM, network, console, performance) to AI coding agents |
314
+ | **[Speakeasy](https://speakeasy.com)** | OpenAPI tooling (customers include Google, Verizon, Mistral AI, DocuSign, Vercel); GCF is a native output format in their `oq` CLI |
315
+ | **[OmniRoute](https://omniroute.online)** | 17K★ · AI gateway, registry, and proxy between AI clients and model providers; GCF vendored into its compression engine |
316
+ | **[NetClaw](https://github.com/automateyournetwork/netclaw)** | 610★ · AI-powered network automation (113 skills, 66 MCP integrations); replaced TOON with GCF across every MCP server |
317
+ | **[ctx](https://github.com/stevesolun/ctx)** | 552★ · real-time context selector for Claude Code; surfaces only the relevant tools from a 103K-node knowledge graph |
318
+ | **[Lynkr](https://github.com/Fast-Editor/Lynkr)** | 531★ · local LLM gateway for AI coding clients; GCF as a drop-in tool-result compressor alongside TOON |
319
+ | **[Open Data Products SDK](https://opendataproducts.org/sdk/)** | Linux Foundation · Python toolkit and MCP server for data-product standards; GCF sidecars for agent context |
320
+ | **[NeuroNest](https://neuronest.cc)** | agent-first IDE; first commercial GCF adoption, across four encoding surfaces with session dedup and delta |
321
+ | **[Raycast](https://raycast.com/blackwell-systems/json-to-gcf-converter)** | JSON-to-GCF Converter extension in the Raycast Store, for the macOS productivity launcher |
322
+
323
+ [See all adopters →](https://gcformat.com/ecosystem/adopters.html)
292
324
 
293
325
  ## License
294
326
 
@@ -0,0 +1,22 @@
1
+ gcf/__init__.py,sha256=dZkdcjbG-dcolg_7TBROjRq3opftDGU423ankMIOhyA,2616
2
+ gcf/__main__.py,sha256=EpvBz1yc8H0D5OJ1zy2tYke-kRzvudKa4DEbfeW14ao,71
3
+ gcf/cli.py,sha256=UEe1CAZn-rKGNIo_ap8-oez3ucl6DSRbsdv6RDnzygY,5256
4
+ gcf/constants.py,sha256=cmZ8YJSOB0im_eyfN8v4UvrLpBC6Fuf4cfcKZGbutxY,638
5
+ gcf/decode.py,sha256=TP58_7UBhfeI0o9zpNgtQvAnLQGhNt0RaUjku7Ke75A,7162
6
+ gcf/decode_generic.py,sha256=OTNhryeAFfE5izpP6dD0DtBzcC_Jht1-vLg51gjnZ28,30510
7
+ gcf/delta.py,sha256=oviQ9WsDRYXPXE8bw6SFutmzMRIvu-U_Zzw-39Nd5Ic,8053
8
+ gcf/encode.py,sha256=OYGDyF3oP2I8Y2PrPL_5yDVURqB0mghtqlOGyDEd-Fs,4081
9
+ gcf/generic.py,sha256=1c53utF2GcD8lS_wks2opQw08hMz4KGL-7C8g7JB5bA,21577
10
+ gcf/generic_delta.py,sha256=5KSUT_QUeH7OILLWfu9JveaXRSrqRIi3c6E8WjlgC4Q,18785
11
+ gcf/keyed_map.py,sha256=qCwhehXB1FQ2iYCPhcvXOqGwrVqdXhGYZGxv56CSUgc,3644
12
+ gcf/packroot.py,sha256=0rZY7TEVcLzA8XhoIOc6_g9lbRCn-MUr54YTRVhzsO4,2019
13
+ gcf/scalar.py,sha256=NMf-ZJzTOGk1JLvydqYPRxUH80u2Sr_bonwRHXTYd5U,9856
14
+ gcf/session.py,sha256=rPqR4xsHKpi_G37t1ODwkTUdHhVAGa3Uig8F60h34Ms,5335
15
+ gcf/stream.py,sha256=3eyzmLMZPyEGC1MS9pujxio2x_t7rCL-eNuJ4DjYkaA,5938
16
+ gcf/stream_generic.py,sha256=h2mJ-jI-c1H2L0uoQVhrvDc7uJEcOwS1MB51k4SCPNM,5775
17
+ gcf/types.py,sha256=AWm-LQoSqLHAYtEjcAxWQZqJ4JXqNreLUKO2mJFgNMA,1465
18
+ gcf_python-2.5.1.dist-info/METADATA,sha256=1qFB-VBVxd0CX6s0k5WprCNE2BoYE4cFu_T5PJ5ptZQ,16068
19
+ gcf_python-2.5.1.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
20
+ gcf_python-2.5.1.dist-info/entry_points.txt,sha256=aFT6gqlkh8iGfM8cblE-LUMxHH08_v71IIoZtDdRIVA,37
21
+ gcf_python-2.5.1.dist-info/licenses/LICENSE,sha256=2Fit9wnaIe--RMSAgyQqxC5hfZTyZqn4fIdBtp9qPDw,1072
22
+ gcf_python-2.5.1.dist-info/RECORD,,
@@ -1,21 +0,0 @@
1
- gcf/__init__.py,sha256=dZkdcjbG-dcolg_7TBROjRq3opftDGU423ankMIOhyA,2616
2
- gcf/__main__.py,sha256=EpvBz1yc8H0D5OJ1zy2tYke-kRzvudKa4DEbfeW14ao,71
3
- gcf/cli.py,sha256=UEe1CAZn-rKGNIo_ap8-oez3ucl6DSRbsdv6RDnzygY,5256
4
- gcf/constants.py,sha256=cmZ8YJSOB0im_eyfN8v4UvrLpBC6Fuf4cfcKZGbutxY,638
5
- gcf/decode.py,sha256=nD8bXYhoeHQQ3LCeAJQOAgFuob-V_6us4mcBYtL_bBc,5978
6
- gcf/decode_generic.py,sha256=ozYmGiZu9S1zTGsN1PuMTF8soV53eq1v5PXMx2YNgWQ,27888
7
- gcf/delta.py,sha256=oviQ9WsDRYXPXE8bw6SFutmzMRIvu-U_Zzw-39Nd5Ic,8053
8
- gcf/encode.py,sha256=OYGDyF3oP2I8Y2PrPL_5yDVURqB0mghtqlOGyDEd-Fs,4081
9
- gcf/generic.py,sha256=t0Px1fAoWsmFFeICbL65LL8UTGFfnbh4PCanvMMKqYs,18198
10
- gcf/generic_delta.py,sha256=GU7CNumz3PQhC0Xs5pO30Nb3ACvSorHEQMu-4kazpFM,17838
11
- gcf/packroot.py,sha256=0rZY7TEVcLzA8XhoIOc6_g9lbRCn-MUr54YTRVhzsO4,2019
12
- gcf/scalar.py,sha256=MZay-KIROvaFHet-g2-pBghahT1bf_5bxZjG4yTZkSo,9329
13
- gcf/session.py,sha256=rPqR4xsHKpi_G37t1ODwkTUdHhVAGa3Uig8F60h34Ms,5335
14
- gcf/stream.py,sha256=3eyzmLMZPyEGC1MS9pujxio2x_t7rCL-eNuJ4DjYkaA,5938
15
- gcf/stream_generic.py,sha256=RnqqiPSu5joJa-7e58QbbzSGfvxBICA587A3aArjZvE,3250
16
- gcf/types.py,sha256=AWm-LQoSqLHAYtEjcAxWQZqJ4JXqNreLUKO2mJFgNMA,1465
17
- gcf_python-2.4.0.dist-info/METADATA,sha256=xDum3Ytd_7KX6SP8940MQDWL08VAwmbdkQuc-nkzN6M,13115
18
- gcf_python-2.4.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
19
- gcf_python-2.4.0.dist-info/entry_points.txt,sha256=aFT6gqlkh8iGfM8cblE-LUMxHH08_v71IIoZtDdRIVA,37
20
- gcf_python-2.4.0.dist-info/licenses/LICENSE,sha256=2Fit9wnaIe--RMSAgyQqxC5hfZTyZqn4fIdBtp9qPDw,1072
21
- gcf_python-2.4.0.dist-info/RECORD,,