gcf-python 2.2.2__py3-none-any.whl → 2.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gcf/__init__.py +38 -2
- gcf/decode_generic.py +76 -8
- gcf/delta.py +186 -3
- gcf/encode.py +36 -14
- gcf/generic.py +79 -7
- gcf/generic_delta.py +487 -0
- gcf/keyed_map.py +86 -0
- gcf/packroot.py +53 -0
- gcf/scalar.py +11 -3
- gcf/session.py +26 -16
- gcf/stream.py +22 -13
- gcf/stream_generic.py +54 -3
- {gcf_python-2.2.2.dist-info → gcf_python-2.5.0.dist-info}/METADATA +97 -17
- gcf_python-2.5.0.dist-info/RECORD +22 -0
- gcf_python-2.2.2.dist-info/RECORD +0 -19
- {gcf_python-2.2.2.dist-info → gcf_python-2.5.0.dist-info}/WHEEL +0 -0
- {gcf_python-2.2.2.dist-info → gcf_python-2.5.0.dist-info}/entry_points.txt +0 -0
- {gcf_python-2.2.2.dist-info → gcf_python-2.5.0.dist-info}/licenses/LICENSE +0 -0
gcf/__init__.py
CHANGED
|
@@ -36,9 +36,27 @@ Specification: https://github.com/blackwell-systems/gcf
|
|
|
36
36
|
|
|
37
37
|
from .constants import KIND_ABBREV, KIND_EXPAND
|
|
38
38
|
from .decode import DecodeError, decode
|
|
39
|
-
from .delta import encode_delta
|
|
39
|
+
from .delta import decode_delta, encode_delta, verify_delta
|
|
40
40
|
from .encode import encode
|
|
41
41
|
from .generic import encode_generic, GenericOptions
|
|
42
|
+
from .generic_delta import (
|
|
43
|
+
GenericSet,
|
|
44
|
+
GenericDeltaPayload,
|
|
45
|
+
generic_pack_root,
|
|
46
|
+
diff_generic_sets,
|
|
47
|
+
encode_generic_full,
|
|
48
|
+
encode_generic_delta,
|
|
49
|
+
decode_generic_full,
|
|
50
|
+
decode_generic_delta,
|
|
51
|
+
verify_generic_delta,
|
|
52
|
+
GenericDeltaSession,
|
|
53
|
+
ReanchorPolicy,
|
|
54
|
+
ReanchorMode,
|
|
55
|
+
fixed_n,
|
|
56
|
+
size_guard,
|
|
57
|
+
DEFAULT_REANCHOR_N,
|
|
58
|
+
)
|
|
59
|
+
from .packroot import pack_root
|
|
42
60
|
from .session import Session, encode_with_session
|
|
43
61
|
from .decode_generic import decode_generic
|
|
44
62
|
from .stream import StreamEncoder
|
|
@@ -58,12 +76,30 @@ __all__ = [
|
|
|
58
76
|
"StreamEncoder",
|
|
59
77
|
"Symbol",
|
|
60
78
|
"decode",
|
|
79
|
+
"decode_delta",
|
|
61
80
|
"decode_generic",
|
|
62
81
|
"encode",
|
|
63
82
|
"encode_delta",
|
|
83
|
+
"verify_delta",
|
|
64
84
|
"encode_generic",
|
|
65
85
|
"GenericOptions",
|
|
66
86
|
"encode_with_session",
|
|
87
|
+
"GenericSet",
|
|
88
|
+
"GenericDeltaPayload",
|
|
89
|
+
"generic_pack_root",
|
|
90
|
+
"pack_root",
|
|
91
|
+
"diff_generic_sets",
|
|
92
|
+
"encode_generic_full",
|
|
93
|
+
"encode_generic_delta",
|
|
94
|
+
"decode_generic_full",
|
|
95
|
+
"decode_generic_delta",
|
|
96
|
+
"verify_generic_delta",
|
|
97
|
+
"GenericDeltaSession",
|
|
98
|
+
"ReanchorPolicy",
|
|
99
|
+
"ReanchorMode",
|
|
100
|
+
"fixed_n",
|
|
101
|
+
"size_guard",
|
|
102
|
+
"DEFAULT_REANCHOR_N",
|
|
67
103
|
]
|
|
68
104
|
|
|
69
|
-
__version__ = "2.
|
|
105
|
+
__version__ = "2.3.0"
|
gcf/decode_generic.py
CHANGED
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
from typing import Any
|
|
6
6
|
|
|
7
7
|
from .decode import decode
|
|
8
|
+
from .keyed_map import keyed_rows_to_map
|
|
8
9
|
from .scalar import (
|
|
9
10
|
parse_scalar, parse_quoted_string, split_respecting_quotes, split_field_decl,
|
|
10
11
|
is_bare_key, MISSING, ATTACHMENT,
|
|
@@ -72,7 +73,7 @@ def decode_generic(input_text: str) -> Any:
|
|
|
72
73
|
if trimmed.startswith("##! "):
|
|
73
74
|
summary_line = trimmed
|
|
74
75
|
continue
|
|
75
|
-
if trimmed.startswith("## ") and "[?]" in trimmed:
|
|
76
|
+
if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
|
|
76
77
|
deferred_count += 1
|
|
77
78
|
content_lines.append(line)
|
|
78
79
|
|
|
@@ -134,7 +135,7 @@ def _parse_object_body(
|
|
|
134
135
|
|
|
135
136
|
if content.startswith("## "):
|
|
136
137
|
hdr = content[3:]
|
|
137
|
-
bi = hdr
|
|
138
|
+
bi = _find_bracket_start(hdr)
|
|
138
139
|
if bi >= 0:
|
|
139
140
|
name = _parse_key_from_header(hdr[:bi])
|
|
140
141
|
_check_dup(out, name)
|
|
@@ -177,7 +178,14 @@ def _parse_object_body(
|
|
|
177
178
|
i += 1
|
|
178
179
|
continue
|
|
179
180
|
|
|
180
|
-
|
|
181
|
+
# An object-body line that is not a `## ` section, a `key=value` field, or
|
|
182
|
+
# an inline array is not valid content and MUST NOT be silently skipped
|
|
183
|
+
# (that dropped data, a lossless round-trip hole). A pipe-delimited line is
|
|
184
|
+
# a stray positional inline body with no eligible `^` cell (SPEC 16.5,
|
|
185
|
+
# orphan_inline_attachment); any other unrecognized line is likewise rejected.
|
|
186
|
+
if "|" in content:
|
|
187
|
+
raise ValueError(f"orphan_inline_attachment: {content}")
|
|
188
|
+
raise ValueError(f"invalid_line: unexpected content in object body: {content!r}")
|
|
181
189
|
return i - start
|
|
182
190
|
|
|
183
191
|
|
|
@@ -226,10 +234,24 @@ def _parse_array_from_header(
|
|
|
226
234
|
raise ValueError("invalid_count")
|
|
227
235
|
count_str = bp[1:close]
|
|
228
236
|
after = bp[close + 1:]
|
|
237
|
+
|
|
238
|
+
# A keyed map is marked by `:` after the count inside the bracket (`[N:]`).
|
|
239
|
+
# The decoder reconstructs a JSON object, not an array (SPEC 7.2a.2).
|
|
240
|
+
keyed = count_str.endswith(":")
|
|
241
|
+
if keyed:
|
|
242
|
+
count_str = count_str[:-1]
|
|
243
|
+
if not after.startswith("{"):
|
|
244
|
+
raise ValueError("keyed_map: missing field declaration")
|
|
245
|
+
|
|
229
246
|
count = -1
|
|
230
247
|
if count_str != "?":
|
|
231
248
|
count = _parse_count(count_str)
|
|
232
249
|
|
|
250
|
+
# A keyed map has at least one member; an empty object is encoded per
|
|
251
|
+
# Section 7.7, never as [0:] (SPEC 7.2a.4).
|
|
252
|
+
if keyed and count == 0:
|
|
253
|
+
raise ValueError("keyed_map: zero count [0:] is invalid (an empty object uses Section 7.7)")
|
|
254
|
+
|
|
233
255
|
if count == 0 and not after.startswith("{") and not after.startswith(":"):
|
|
234
256
|
return [], 1
|
|
235
257
|
|
|
@@ -252,6 +274,8 @@ def _parse_array_from_header(
|
|
|
252
274
|
rows, consumed = _parse_tabular_body(lines, header_line + 1, depth, fields, count)
|
|
253
275
|
if count >= 0 and len(rows) != count:
|
|
254
276
|
raise ValueError(f"count_mismatch: declared {count}, got {len(rows)}")
|
|
277
|
+
if keyed:
|
|
278
|
+
return keyed_rows_to_map(rows, fields), consumed + 1
|
|
255
279
|
return rows, consumed + 1
|
|
256
280
|
|
|
257
281
|
items, consumed = _parse_expanded_body(lines, header_line + 1, depth)
|
|
@@ -260,6 +284,27 @@ def _parse_array_from_header(
|
|
|
260
284
|
return items, consumed + 1
|
|
261
285
|
|
|
262
286
|
|
|
287
|
+
def _find_bracket_start(s: str) -> int:
|
|
288
|
+
# Find " [" (the named-array count bracket) that is OUTSIDE any quoted name,
|
|
289
|
+
# so a quoted section/key name containing " [" (e.g. `## "a [1] b"`) is not
|
|
290
|
+
# misread as a named-array header. Mirrors _find_closing_brace's quote tracking.
|
|
291
|
+
in_quote = False
|
|
292
|
+
escaped = False
|
|
293
|
+
for i, c in enumerate(s):
|
|
294
|
+
if escaped:
|
|
295
|
+
escaped = False
|
|
296
|
+
continue
|
|
297
|
+
if c == "\\" and in_quote:
|
|
298
|
+
escaped = True
|
|
299
|
+
continue
|
|
300
|
+
if c == '"':
|
|
301
|
+
in_quote = not in_quote
|
|
302
|
+
continue
|
|
303
|
+
if not in_quote and c == " " and i + 1 < len(s) and s[i + 1] == "[":
|
|
304
|
+
return i
|
|
305
|
+
return -1
|
|
306
|
+
|
|
307
|
+
|
|
263
308
|
def _find_closing_brace(s: str) -> int:
|
|
264
309
|
in_quote = False
|
|
265
310
|
escaped = False
|
|
@@ -521,6 +566,11 @@ def _parse_tabular_body(
|
|
|
521
566
|
if row_has_id:
|
|
522
567
|
inline_idx = 0
|
|
523
568
|
|
|
569
|
+
# Columns that carry a `^` marker cell in this row legitimately expect
|
|
570
|
+
# a `.field` body. Any other `.field` is an orphan (Section 16.5) unless
|
|
571
|
+
# its name contains `>` (the flatten-fallback attachment, Section 7.4.6.1.4).
|
|
572
|
+
expected_att = set(traditional_att_fields) | set(inline_att_fields)
|
|
573
|
+
|
|
524
574
|
while i < len(lines):
|
|
525
575
|
a_line = lines[i]
|
|
526
576
|
a_content: str | None = None
|
|
@@ -541,6 +591,14 @@ def _parse_tabular_body(
|
|
|
541
591
|
att_name, after_name = _parse_attachment_name(rest)
|
|
542
592
|
after_name_stripped = after_name.lstrip()
|
|
543
593
|
|
|
594
|
+
# Orphan attachment: a `.field` with no matching `^` cell in this
|
|
595
|
+
# row is only legitimate for a `>`-named field (Section 7.4.6.1.4).
|
|
596
|
+
# Any other unmatched attachment is rejected rather than silently
|
|
597
|
+
# injected as an undeclared extra field, which would decode to a
|
|
598
|
+
# record no encoder produces (Section 16.5, lossless round-trip).
|
|
599
|
+
if att_name not in expected_att and ">" not in att_name:
|
|
600
|
+
raise ValueError(f"orphan_attachment: {att_name}")
|
|
601
|
+
|
|
544
602
|
# Prefixed inline data.
|
|
545
603
|
ifs = inline_schemas.get(att_name)
|
|
546
604
|
if ifs and not after_name_stripped.startswith("{}") and not after_name_stripped.startswith("["):
|
|
@@ -613,8 +671,22 @@ def _parse_tabular_body(
|
|
|
613
671
|
if extra_name in attachment_values:
|
|
614
672
|
raise ValueError(f"duplicate_attachment: {extra_name}")
|
|
615
673
|
|
|
674
|
+
# Reconstruct the row in declared field-union order. A flattened group is
|
|
675
|
+
# emitted at the position of its first path column, so the nested object
|
|
676
|
+
# reappears where the original field was, not appended at the end (SPEC
|
|
677
|
+
# 7.4.6.1 step 7 and the key-order preservation requirement, SPEC 52, 931).
|
|
678
|
+
nested = _unflatten_paths(path_column_map, flat_values, flat_absent) if path_column_map else {}
|
|
679
|
+
emitted_groups: set[str] = set()
|
|
616
680
|
row: dict[str, Any] = {}
|
|
617
681
|
for f in fields:
|
|
682
|
+
if f in path_column_map:
|
|
683
|
+
top = path_column_map[f][0]
|
|
684
|
+
if top in emitted_groups:
|
|
685
|
+
continue
|
|
686
|
+
emitted_groups.add(top)
|
|
687
|
+
if top in nested: # omitted when the whole group is absent
|
|
688
|
+
row[top] = nested[top]
|
|
689
|
+
continue
|
|
618
690
|
if f in missing_fields:
|
|
619
691
|
continue
|
|
620
692
|
if f in cell_values:
|
|
@@ -625,10 +697,6 @@ def _parse_tabular_body(
|
|
|
625
697
|
for k, v in attachment_values.items():
|
|
626
698
|
if k not in row:
|
|
627
699
|
row[k] = v
|
|
628
|
-
# Unflatten path columns into nested objects.
|
|
629
|
-
if path_column_map:
|
|
630
|
-
nested = _unflatten_paths(path_column_map, flat_values, flat_absent)
|
|
631
|
-
row.update(nested)
|
|
632
700
|
|
|
633
701
|
rows.append(row)
|
|
634
702
|
|
|
@@ -725,7 +793,7 @@ def _validate_summary_counts(
|
|
|
725
793
|
current_count = 0
|
|
726
794
|
for line in content_lines:
|
|
727
795
|
trimmed = line.lstrip()
|
|
728
|
-
if trimmed.startswith("## ") and "[?]" in trimmed:
|
|
796
|
+
if trimmed.startswith("## ") and ("[?]" in trimmed or "[?:]" in trimmed):
|
|
729
797
|
if in_deferred:
|
|
730
798
|
actual_counts.append(current_count)
|
|
731
799
|
in_deferred = True
|
gcf/delta.py
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
"""GCF delta encoding: only added/removed symbols for incremental delivery."""
|
|
2
2
|
|
|
3
|
-
from .constants import KIND_ABBREV
|
|
4
|
-
from .
|
|
3
|
+
from .constants import KIND_ABBREV, KIND_EXPAND
|
|
4
|
+
from .packroot import pack_root
|
|
5
|
+
from .types import DeltaPayload, Edge, Symbol
|
|
5
6
|
|
|
6
7
|
|
|
7
8
|
def encode_delta(d: DeltaPayload) -> str:
|
|
@@ -37,7 +38,9 @@ def encode_delta(d: DeltaPayload) -> str:
|
|
|
37
38
|
parts.append("## added")
|
|
38
39
|
for i, s in enumerate(d.added):
|
|
39
40
|
kind = KIND_ABBREV.get(s.kind, s.kind)
|
|
40
|
-
parts.append(
|
|
41
|
+
parts.append(
|
|
42
|
+
f"@{i} {kind} {s.qualified_name} {s.score:.2f} {s.provenance} {s.distance}"
|
|
43
|
+
)
|
|
41
44
|
|
|
42
45
|
# Removed edges.
|
|
43
46
|
if d.removed_edges:
|
|
@@ -52,3 +55,183 @@ def encode_delta(d: DeltaPayload) -> str:
|
|
|
52
55
|
parts.append(f"{e.source} -> {e.target} {e.edge_type}")
|
|
53
56
|
|
|
54
57
|
return "\n".join(parts) + "\n"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _expand_kind(k: str) -> str:
|
|
61
|
+
"""Reverse a kind abbreviation to its full form (identity if unknown)."""
|
|
62
|
+
return KIND_EXPAND.get(k, k)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _parse_delta_edge(line: str) -> Edge:
|
|
66
|
+
"""Parse a `source -> target type` delta edge line."""
|
|
67
|
+
idx = line.find(" -> ")
|
|
68
|
+
if idx < 0:
|
|
69
|
+
raise ValueError(f"malformed_delta: edge line missing ' -> ': {line!r}")
|
|
70
|
+
source = line[:idx]
|
|
71
|
+
rest = line[idx + 4 :].split()
|
|
72
|
+
if len(rest) != 2:
|
|
73
|
+
raise ValueError(
|
|
74
|
+
f"malformed_delta: edge line {line!r} must be 'source -> target type'"
|
|
75
|
+
)
|
|
76
|
+
return Edge(source=source, target=rest[0], edge_type=rest[1])
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def decode_delta(wire: str) -> DeltaPayload:
|
|
80
|
+
"""Parse a GCF graph delta wire payload back into a DeltaPayload.
|
|
81
|
+
|
|
82
|
+
Kind abbreviations on removed/added lines are expanded to their full form so the
|
|
83
|
+
result matches a base snapshot's symbol identities. Raises ValueError containing
|
|
84
|
+
``malformed_delta`` on bad lines or unknown sections.
|
|
85
|
+
"""
|
|
86
|
+
lines = wire.rstrip("\n").split("\n")
|
|
87
|
+
if not lines or lines[0] == "":
|
|
88
|
+
raise ValueError("missing_header: empty delta payload")
|
|
89
|
+
header = lines[0].rstrip("\r")
|
|
90
|
+
if not header.startswith("GCF profile=graph"):
|
|
91
|
+
raise ValueError(
|
|
92
|
+
"missing_profile: delta header must begin with 'GCF profile=graph'"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
d = DeltaPayload()
|
|
96
|
+
for field in header.split():
|
|
97
|
+
kv = field.split("=", 1)
|
|
98
|
+
if len(kv) != 2:
|
|
99
|
+
continue
|
|
100
|
+
key, value = kv
|
|
101
|
+
if key == "tool":
|
|
102
|
+
d.tool = value
|
|
103
|
+
elif key == "base_root":
|
|
104
|
+
d.base_root = value
|
|
105
|
+
elif key == "new_root":
|
|
106
|
+
d.new_root = value
|
|
107
|
+
|
|
108
|
+
section = ""
|
|
109
|
+
for raw in lines[1:]:
|
|
110
|
+
line = raw.rstrip("\r")
|
|
111
|
+
if line == "":
|
|
112
|
+
continue
|
|
113
|
+
if line.startswith("## "):
|
|
114
|
+
section = line[3:].strip()
|
|
115
|
+
if section not in ("removed", "added", "edges_removed", "edges_added"):
|
|
116
|
+
raise ValueError(f"malformed_delta: unknown section {section!r}")
|
|
117
|
+
continue
|
|
118
|
+
if section == "removed":
|
|
119
|
+
parts = line.split()
|
|
120
|
+
if len(parts) != 2:
|
|
121
|
+
raise ValueError(
|
|
122
|
+
f"malformed_delta: removed line {line!r} must be 'kind qname'"
|
|
123
|
+
)
|
|
124
|
+
d.removed.append(
|
|
125
|
+
Symbol(kind=_expand_kind(parts[0]), qualified_name=parts[1])
|
|
126
|
+
)
|
|
127
|
+
elif section == "added":
|
|
128
|
+
parts = line.split()
|
|
129
|
+
if len(parts) != 6:
|
|
130
|
+
raise ValueError(
|
|
131
|
+
f"malformed_delta: added line {line!r} must be "
|
|
132
|
+
"'@id kind qname score provenance distance'"
|
|
133
|
+
)
|
|
134
|
+
try:
|
|
135
|
+
score = float(parts[3])
|
|
136
|
+
except ValueError:
|
|
137
|
+
raise ValueError(f"malformed_delta: invalid added score {parts[3]!r}")
|
|
138
|
+
try:
|
|
139
|
+
dist = int(parts[5])
|
|
140
|
+
except ValueError:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
f"malformed_delta: invalid added distance {parts[5]!r}"
|
|
143
|
+
)
|
|
144
|
+
d.added.append(
|
|
145
|
+
Symbol(
|
|
146
|
+
kind=_expand_kind(parts[1]),
|
|
147
|
+
qualified_name=parts[2],
|
|
148
|
+
score=score,
|
|
149
|
+
provenance=parts[4],
|
|
150
|
+
distance=dist,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
elif section in ("edges_removed", "edges_added"):
|
|
154
|
+
e = _parse_delta_edge(line)
|
|
155
|
+
if section == "edges_removed":
|
|
156
|
+
d.removed_edges.append(e)
|
|
157
|
+
else:
|
|
158
|
+
d.added_edges.append(e)
|
|
159
|
+
else:
|
|
160
|
+
raise ValueError(
|
|
161
|
+
f"malformed_delta: data line {line!r} before any section header"
|
|
162
|
+
)
|
|
163
|
+
return d
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def verify_delta(
|
|
167
|
+
base_symbols: list[Symbol],
|
|
168
|
+
base_edges: list[Edge],
|
|
169
|
+
removed: list[Symbol],
|
|
170
|
+
added: list[Symbol],
|
|
171
|
+
removed_edges: list[Edge],
|
|
172
|
+
added_edges: list[Edge],
|
|
173
|
+
expected_new_root: str,
|
|
174
|
+
) -> tuple[list[Symbol], list[Edge]]:
|
|
175
|
+
"""Apply a delta to a base snapshot and verify the resulting pack root.
|
|
176
|
+
|
|
177
|
+
Symbols are matched by identity ``(kind, qualified_name)``; edges by
|
|
178
|
+
``(source, target, edge_type)``. Raises ValueError containing ``delta_invalid``
|
|
179
|
+
when removing a symbol/edge that does not exist or adding one that already exists,
|
|
180
|
+
and ``root_mismatch`` when the recomputed pack root differs from
|
|
181
|
+
``expected_new_root``. On success returns the applied ``(symbols, edges)``.
|
|
182
|
+
"""
|
|
183
|
+
sym_map: dict[tuple[str, str], Symbol] = {}
|
|
184
|
+
for s in base_symbols:
|
|
185
|
+
sym_map[(s.kind, s.qualified_name)] = s
|
|
186
|
+
|
|
187
|
+
for s in removed:
|
|
188
|
+
key = (s.kind, s.qualified_name)
|
|
189
|
+
if key not in sym_map:
|
|
190
|
+
raise ValueError(
|
|
191
|
+
f"delta_invalid: removing symbol {s.kind} {s.qualified_name} "
|
|
192
|
+
"that does not exist in base"
|
|
193
|
+
)
|
|
194
|
+
del sym_map[key]
|
|
195
|
+
|
|
196
|
+
for s in added:
|
|
197
|
+
key = (s.kind, s.qualified_name)
|
|
198
|
+
if key in sym_map:
|
|
199
|
+
raise ValueError(
|
|
200
|
+
f"delta_invalid: adding symbol {s.kind} {s.qualified_name} "
|
|
201
|
+
"that already exists"
|
|
202
|
+
)
|
|
203
|
+
sym_map[key] = s
|
|
204
|
+
|
|
205
|
+
result_symbols = list(sym_map.values())
|
|
206
|
+
|
|
207
|
+
edge_map: dict[tuple[str, str, str], Edge] = {}
|
|
208
|
+
for e in base_edges:
|
|
209
|
+
edge_map[(e.source, e.target, e.edge_type)] = e
|
|
210
|
+
|
|
211
|
+
for e in removed_edges:
|
|
212
|
+
key = (e.source, e.target, e.edge_type)
|
|
213
|
+
if key not in edge_map:
|
|
214
|
+
raise ValueError(
|
|
215
|
+
f"delta_invalid: removing edge {e.source} -> {e.target} "
|
|
216
|
+
f"{e.edge_type} that does not exist"
|
|
217
|
+
)
|
|
218
|
+
del edge_map[key]
|
|
219
|
+
|
|
220
|
+
for e in added_edges:
|
|
221
|
+
key = (e.source, e.target, e.edge_type)
|
|
222
|
+
if key in edge_map:
|
|
223
|
+
raise ValueError(
|
|
224
|
+
f"delta_invalid: adding edge {e.source} -> {e.target} "
|
|
225
|
+
f"{e.edge_type} that already exists"
|
|
226
|
+
)
|
|
227
|
+
edge_map[key] = e
|
|
228
|
+
|
|
229
|
+
result_edges = list(edge_map.values())
|
|
230
|
+
|
|
231
|
+
computed_root = pack_root(result_symbols, result_edges)
|
|
232
|
+
if computed_root != expected_new_root:
|
|
233
|
+
raise ValueError(
|
|
234
|
+
f"root_mismatch: computed {computed_root}, expected {expected_new_root}"
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
return result_symbols, result_edges
|
gcf/encode.py
CHANGED
|
@@ -17,10 +17,16 @@ def encode(p: Payload) -> str:
|
|
|
17
17
|
"""
|
|
18
18
|
parts: list[str] = []
|
|
19
19
|
|
|
20
|
-
#
|
|
20
|
+
# Group symbols by distance (sorted by score descending within each group),
|
|
21
|
+
# then assign local IDs in output order so they are sequential in the wire
|
|
22
|
+
# (SPEC 16.1).
|
|
23
|
+
groups = _group_by_distance(p.symbols)
|
|
21
24
|
sym_index: dict[str, int] = {}
|
|
22
|
-
|
|
23
|
-
|
|
25
|
+
next_id = 0
|
|
26
|
+
for _distance, g_symbols in groups:
|
|
27
|
+
for s in g_symbols:
|
|
28
|
+
sym_index[s.qualified_name] = next_id
|
|
29
|
+
next_id += 1
|
|
24
30
|
|
|
25
31
|
# Count valid edges (both endpoints in symbol index).
|
|
26
32
|
valid_edges = sum(
|
|
@@ -28,14 +34,20 @@ def encode(p: Payload) -> str:
|
|
|
28
34
|
if e.source in sym_index and e.target in sym_index
|
|
29
35
|
)
|
|
30
36
|
|
|
31
|
-
# Header line.
|
|
32
|
-
|
|
37
|
+
# Header line (SPEC 16.1): omit budget/tokens/edges when zero, matching the
|
|
38
|
+
# reference encoder.
|
|
39
|
+
header = f"GCF profile=graph tool={p.tool}"
|
|
40
|
+
if p.token_budget > 0:
|
|
41
|
+
header += f" budget={p.token_budget}"
|
|
42
|
+
if p.tokens_used > 0:
|
|
43
|
+
header += f" tokens={p.tokens_used}"
|
|
44
|
+
header += f" symbols={len(p.symbols)}"
|
|
45
|
+
if valid_edges > 0:
|
|
46
|
+
header += f" edges={valid_edges}"
|
|
33
47
|
if p.pack_root:
|
|
34
48
|
header += f" pack_root={p.pack_root}"
|
|
35
49
|
parts.append(header)
|
|
36
50
|
|
|
37
|
-
# Group symbols by distance.
|
|
38
|
-
groups = _group_by_distance(p.symbols)
|
|
39
51
|
group_names = ["targets", "related", "extended"]
|
|
40
52
|
|
|
41
53
|
for g_distance, g_symbols in groups:
|
|
@@ -52,17 +64,25 @@ def encode(p: Payload) -> str:
|
|
|
52
64
|
kind = KIND_ABBREV.get(s.kind, s.kind)
|
|
53
65
|
parts.append(f"@{idx} {kind} {s.qualified_name} {s.score:.2f} {s.provenance}")
|
|
54
66
|
|
|
55
|
-
# Edges section.
|
|
67
|
+
# Edges section. Order edges by source ID then target ID (then edge type
|
|
68
|
+
# for parallel edges) so the wire is canonical regardless of the order
|
|
69
|
+
# edges were provided (SPEC 16.1). Edge reordering is decode-invariant
|
|
70
|
+
# (edges are a set) and does not affect pack_root, which sorts edge records
|
|
71
|
+
# independently.
|
|
56
72
|
if p.edges:
|
|
57
|
-
|
|
73
|
+
resolved: list[tuple[int, int, str, str]] = []
|
|
58
74
|
for e in p.edges:
|
|
59
75
|
src_idx = sym_index.get(e.source)
|
|
60
76
|
tgt_idx = sym_index.get(e.target)
|
|
61
77
|
if src_idx is None or tgt_idx is None:
|
|
62
78
|
continue
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
79
|
+
resolved.append((src_idx, tgt_idx, e.edge_type, e.status))
|
|
80
|
+
resolved.sort(key=lambda r: (r[0], r[1], r[2]))
|
|
81
|
+
edge_lines: list[str] = []
|
|
82
|
+
for src_idx, tgt_idx, edge_type, status in resolved:
|
|
83
|
+
line = f"@{tgt_idx}<@{src_idx} {edge_type}"
|
|
84
|
+
if status and status != "unchanged":
|
|
85
|
+
line += f" {status}"
|
|
66
86
|
edge_lines.append(line)
|
|
67
87
|
parts.append(f"## edges [{len(edge_lines)}]")
|
|
68
88
|
parts.extend(edge_lines)
|
|
@@ -71,15 +91,17 @@ def encode(p: Payload) -> str:
|
|
|
71
91
|
|
|
72
92
|
|
|
73
93
|
def _group_by_distance(symbols: list[Symbol]) -> list[tuple[int, list[Symbol]]]:
|
|
74
|
-
"""Group symbols by distance,
|
|
94
|
+
"""Group symbols by distance ascending, sorted by score descending within each
|
|
95
|
+
group (stable), matching the reference encoder so IDs are assigned canonically."""
|
|
75
96
|
if not symbols:
|
|
76
97
|
return []
|
|
77
98
|
|
|
99
|
+
ordered = sorted(symbols, key=lambda s: (s.distance, -s.score))
|
|
78
100
|
groups: list[tuple[int, list[Symbol]]] = []
|
|
79
101
|
current_distance: int | None = None
|
|
80
102
|
current_symbols: list[Symbol] = []
|
|
81
103
|
|
|
82
|
-
for s in
|
|
104
|
+
for s in ordered:
|
|
83
105
|
if current_distance is None or current_distance != s.distance:
|
|
84
106
|
if current_symbols:
|
|
85
107
|
groups.append((current_distance, current_symbols)) # type: ignore[arg-type]
|
gcf/generic.py
CHANGED
|
@@ -6,6 +6,7 @@ from dataclasses import dataclass
|
|
|
6
6
|
from typing import Any
|
|
7
7
|
|
|
8
8
|
from .scalar import format_scalar, format_key
|
|
9
|
+
from .keyed_map import keyed_map_eligible
|
|
9
10
|
|
|
10
11
|
|
|
11
12
|
@dataclass
|
|
@@ -30,6 +31,11 @@ def _encode_root_value(v: Any, out: list[str], opts: GenericOptions) -> None:
|
|
|
30
31
|
if v is None:
|
|
31
32
|
out.append("=-")
|
|
32
33
|
elif isinstance(v, dict):
|
|
34
|
+
km = keyed_map_eligible(v)
|
|
35
|
+
if km is not None:
|
|
36
|
+
keys, values, value_fields, key_label = km
|
|
37
|
+
_encode_keyed_map("", False, keys, values, value_fields, key_label, out, 0, opts)
|
|
38
|
+
return
|
|
33
39
|
_encode_object(v, out, 0, opts)
|
|
34
40
|
elif isinstance(v, list):
|
|
35
41
|
_encode_root_array(v, out, opts)
|
|
@@ -42,6 +48,11 @@ def _encode_object(d: dict, out: list[str], depth: int, opts: GenericOptions) ->
|
|
|
42
48
|
for key, value in d.items():
|
|
43
49
|
fk = format_key(key)
|
|
44
50
|
if isinstance(value, dict):
|
|
51
|
+
km = keyed_map_eligible(value)
|
|
52
|
+
if km is not None:
|
|
53
|
+
keys, values, value_fields, key_label = km
|
|
54
|
+
_encode_keyed_map(key, True, keys, values, value_fields, key_label, out, depth, opts)
|
|
55
|
+
continue
|
|
45
56
|
out.append(f"{prefix}## {fk}")
|
|
46
57
|
_encode_object(value, out, depth + 1, opts)
|
|
47
58
|
elif isinstance(value, list):
|
|
@@ -165,8 +176,9 @@ def _analyze_flattenable(
|
|
|
165
176
|
arr: list[dict], field_name: str, parent_path: str
|
|
166
177
|
) -> list[dict] | None:
|
|
167
178
|
"""Analyze whether a field can be flattened. Returns list of leaf descriptors or None."""
|
|
168
|
-
#
|
|
169
|
-
|
|
179
|
+
# A field name that is empty or contains ">" cannot be flattened: it would create an
|
|
180
|
+
# ambiguous path column the decoder treats as literal (SPEC 7.4.6.1.3).
|
|
181
|
+
if field_name == "" or ">" in field_name:
|
|
170
182
|
return None
|
|
171
183
|
canonical_shape: dict[str, str] | None = None # key -> "scalar" | "nested"
|
|
172
184
|
|
|
@@ -192,7 +204,7 @@ def _analyze_flattenable(
|
|
|
192
204
|
if canonical_shape is None:
|
|
193
205
|
canonical_shape = {}
|
|
194
206
|
for k in keys:
|
|
195
|
-
if ">" in k:
|
|
207
|
+
if k == "" or ">" in k: # empty/">" -> ambiguous path (SPEC 7.4.6.1.3)
|
|
196
208
|
return None
|
|
197
209
|
val = v[k]
|
|
198
210
|
if isinstance(val, list):
|
|
@@ -272,8 +284,53 @@ def _resolve_key_chain(item: Any, keys: list[str]) -> tuple[Any, bool]:
|
|
|
272
284
|
return current, True
|
|
273
285
|
|
|
274
286
|
|
|
287
|
+
# ── Keyed map encoding (SPEC 7.2a) ───────────────────────────────────────
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _keyed_header_prefix(name: str, named: bool, depth: int) -> str:
|
|
291
|
+
"""Build the keyed-table header prefix up to the count bracket. named
|
|
292
|
+
distinguishes an anonymous root keyed map (`## `) from a named member whose
|
|
293
|
+
name may itself be the empty string (`## ""`), which format_key quotes so it
|
|
294
|
+
round-trips as a distinct level rather than collapsing into the anonymous
|
|
295
|
+
root form (SPEC 7.2a.1)."""
|
|
296
|
+
prefix = _indent(depth)
|
|
297
|
+
if not named:
|
|
298
|
+
return f"{prefix}## "
|
|
299
|
+
return f"{prefix}## {format_key(name)} "
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _encode_keyed_map(
|
|
303
|
+
name: str, named: bool, keys: list[str], values: list[Any],
|
|
304
|
+
value_fields: list[str], key_label: str, out: list[str], depth: int, opts: GenericOptions
|
|
305
|
+
) -> None:
|
|
306
|
+
"""Emit a keyed table for a map of objects. Routes through _encode_tabular
|
|
307
|
+
with the keyed bracket so nested-value handling (flatten/inline/attachment/
|
|
308
|
+
null/absent) is inherited unchanged. name is empty for a root/anonymous map."""
|
|
309
|
+
_encode_keyed_map_with_prefix(
|
|
310
|
+
_keyed_header_prefix(name, named, depth), keys, values,
|
|
311
|
+
value_fields, key_label, out, depth, opts,
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _encode_keyed_map_with_prefix(
|
|
316
|
+
header_prefix: str, keys: list[str], values: list[Any],
|
|
317
|
+
value_fields: list[str], key_label: str, out: list[str], depth: int, opts: GenericOptions
|
|
318
|
+
) -> None:
|
|
319
|
+
"""Emit `<header_prefix>[N:]{...}` and the keyed rows, reusing _encode_tabular.
|
|
320
|
+
Each value object is augmented with the key column and encoded as a tabular
|
|
321
|
+
row; the key column is declared first."""
|
|
322
|
+
fields = [key_label] + value_fields
|
|
323
|
+
arr: list[dict] = []
|
|
324
|
+
for k, v in zip(keys, values):
|
|
325
|
+
aug = dict(v)
|
|
326
|
+
aug[key_label] = k
|
|
327
|
+
arr.append(aug)
|
|
328
|
+
_encode_tabular(header_prefix, arr, fields, out, depth, opts, keyed=True)
|
|
329
|
+
|
|
330
|
+
|
|
275
331
|
def _encode_tabular(
|
|
276
|
-
header_prefix: str, arr: list[dict], fields: list[str], out: list[str], depth: int,
|
|
332
|
+
header_prefix: str, arr: list[dict], fields: list[str], out: list[str], depth: int,
|
|
333
|
+
opts: GenericOptions, keyed: bool = False
|
|
277
334
|
) -> None:
|
|
278
335
|
prefix = _indent(depth)
|
|
279
336
|
|
|
@@ -320,7 +377,8 @@ def _encode_tabular(
|
|
|
320
377
|
shared_arr_schemas[f] = sas
|
|
321
378
|
|
|
322
379
|
header_fields = ",".join(col["header"] for col in columns)
|
|
323
|
-
|
|
380
|
+
br = ":]" if keyed else "]"
|
|
381
|
+
out.append(f"{header_prefix}[{len(arr)}{br}{{{header_fields}}}")
|
|
324
382
|
|
|
325
383
|
for i, item in enumerate(arr):
|
|
326
384
|
cells: list[str] = []
|
|
@@ -401,8 +459,15 @@ def _encode_tabular(
|
|
|
401
459
|
else:
|
|
402
460
|
_encode_attachment_array(prefix, fk, att_val, out, depth + 2, opts)
|
|
403
461
|
elif isinstance(att_val, dict):
|
|
404
|
-
|
|
405
|
-
|
|
462
|
+
km = keyed_map_eligible(att_val)
|
|
463
|
+
if km is not None:
|
|
464
|
+
keys, values, value_fields, key_label = km
|
|
465
|
+
_encode_keyed_map_with_prefix(
|
|
466
|
+
f"{prefix}.{fk} ", keys, values, value_fields, key_label, out, depth + 2, opts,
|
|
467
|
+
)
|
|
468
|
+
else:
|
|
469
|
+
out.append(f"{prefix}.{fk} {{}}")
|
|
470
|
+
_encode_object(att_val, out, depth + 2, opts)
|
|
406
471
|
else:
|
|
407
472
|
# Scalar attachment (e.g. field names containing ">").
|
|
408
473
|
if att_val is None:
|
|
@@ -467,6 +532,13 @@ def _encode_expanded(header_prefix: str, arr: list, out: list[str], depth: int,
|
|
|
467
532
|
out.append(f"{header_prefix}[{len(arr)}]")
|
|
468
533
|
for i, item in enumerate(arr):
|
|
469
534
|
if isinstance(item, dict):
|
|
535
|
+
km = keyed_map_eligible(item)
|
|
536
|
+
if km is not None:
|
|
537
|
+
keys, values, value_fields, key_label = km
|
|
538
|
+
_encode_keyed_map_with_prefix(
|
|
539
|
+
f"{prefix}@{i} ", keys, values, value_fields, key_label, out, depth + 1, opts,
|
|
540
|
+
)
|
|
541
|
+
continue
|
|
470
542
|
out.append(f"{prefix}@{i} {{}}")
|
|
471
543
|
_encode_object(item, out, depth + 1, opts)
|
|
472
544
|
elif isinstance(item, list):
|