dotted-notation 0.44.11__tar.gz → 0.45.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/CHANGELOG.md +52 -0
  2. {dotted_notation-0.44.11/dotted_notation.egg-info → dotted_notation-0.45.0}/PKG-INFO +8 -1
  3. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/README.md +7 -0
  4. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/__init__.py +2 -2
  5. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/access.py +63 -76
  6. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/api.py +88 -32
  7. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/base.py +31 -3
  8. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/engine.py +16 -6
  9. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/filters.py +1 -2
  10. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/groups.py +73 -13
  11. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/matchers.py +45 -7
  12. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/recursive.py +48 -20
  13. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/results.py +59 -7
  14. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/utils.py +64 -0
  15. {dotted_notation-0.44.11 → dotted_notation-0.45.0/dotted_notation.egg-info}/PKG-INFO +8 -1
  16. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/pyproject.toml +1 -1
  17. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/LICENSE +0 -0
  18. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/MANIFEST.in +0 -0
  19. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/__main__.py +0 -0
  20. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/cli/__init__.py +0 -0
  21. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/cli/_compat.py +0 -0
  22. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/cli/formats.py +0 -0
  23. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/cli/main.py +0 -0
  24. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/containers.py +0 -0
  25. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/grammar.py +0 -0
  26. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/predicates.py +0 -0
  27. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/sql/__init__.py +0 -0
  28. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/sql/core.py +0 -0
  29. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/sql/pg.py +0 -0
  30. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/transforms.py +0 -0
  31. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/utypes.py +0 -0
  32. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted/wrappers.py +0 -0
  33. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted_notation.egg-info/SOURCES.txt +0 -0
  34. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted_notation.egg-info/dependency_links.txt +0 -0
  35. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted_notation.egg-info/entry_points.txt +0 -0
  36. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted_notation.egg-info/requires.txt +0 -0
  37. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/dotted_notation.egg-info/top_level.txt +0 -0
  38. {dotted_notation-0.44.11 → dotted_notation-0.45.0}/setup.cfg +0 -0
@@ -3,6 +3,58 @@
3
3
  All notable changes to `dotted` are recorded here. Versions prior to
4
4
  the ones listed are omitted — browse git history for earlier entries.
5
5
 
6
+ ## [0.45.0]
7
+
8
+ A performance release. Figures compare against 0.44.11 on a document of
9
+ 1000 records, using `python -m benchmarks`.
10
+
11
+ ### Performance
12
+ - `unpack()` no longer grows with the square of the leaf count. A soft
13
+ cut compared every new path against every path already yielded; those
14
+ paths are now indexed by their leading keys. A recursive op with a
15
+ negative depth also recomputed each node's depth-to-leaf for every
16
+ ancestor; a traversal now remembers it. 16,000 leaves: 382s to 0.21s.
17
+ - `unpack(project=...)` matches projections against the paths the walk
18
+ already has instead of parsing each leaf path it generated.
19
+ - A concrete key is no longer found by comparing it against every key
20
+ of the node: an absent key matches without a scan and a present one
21
+ stops at the first match. `update()` looks each level up once, not
22
+ twice. Concrete `update` and `remove` into a 1000-key dict take about
23
+ 70% less time.
24
+ - Writing paths out is cheaper: quoting no longer imports per call,
25
+ tests for an integer by catching an exception, or scans characters in
26
+ a Python loop. `pluck()` and `expand()` take about 40% less time.
27
+ - `setdefault`, `update` and `remove` with `AUTO`, `pluck`, `translate`,
28
+ `match_multi` and `unpack` compile each path once per call.
29
+ - Pattern reads shed per-match layers that did nothing without filters
30
+ or transforms: wildcard and recursive `get` take 15-21% less time,
31
+ pattern `update` 30-40%.
32
+ - `build()` takes about 60% less time, `match_multi` 40%.
33
+ - Imports that ran inside hot functions are now at module level.
34
+
35
+ ### Added
36
+ - `compile()`, an alias for `parse()`. Every API that takes a path also
37
+ takes the compiled result.
38
+ - Benchmark cases for `unpack`, `translate`, `match_multi`,
39
+ `setdefault`, filters, a late key and an absent key.
40
+
41
+ ### Fixed
42
+ - A top-level key starting with `-` was written out bare, so it read
43
+ back as an inverted path and `pack(unpack(obj))` lost it. Such a key
44
+ is now quoted when it leads a path: `'-n'` with quotes, `#'-1'` for a
45
+ negative number.
46
+
47
+ ### Changed
48
+ - Paths returned by `pluck`, `expand`, `unpack` and `walk` quote a
49
+ leading `-` key as above, and so do `match()` captures that begin
50
+ with one. Keys elsewhere in a path are unchanged.
51
+ - A dict-like whose `in` disagrees with its `keys()` can see a present
52
+ key reported as absent, since a concrete key is now tested for
53
+ membership before the keys are scanned.
54
+ - `match()` without `groups` no longer assembles the unmatched tail of
55
+ a partial match, so a hand-built path containing an op that cannot be
56
+ written as text no longer raises there.
57
+
6
58
  ## [0.44.11]
7
59
 
8
60
  ### Changed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dotted_notation
3
- Version: 0.44.11
3
+ Version: 0.45.0
4
4
  Summary: Dotted notation for safe nested data traversal with optional chaining, pattern matching, and transforms
5
5
  Author-email: Frey Waid <logophage1@gmail.com>
6
6
  License: MIT
@@ -1456,6 +1456,13 @@ When `partial=True` (the default for `parse()`), unresolved substitutions are
1456
1456
  allowed through. When `partial=False` (the default for traversal APIs),
1457
1457
  unresolved templates raise `TypeError`.
1458
1458
 
1459
+ `compile()` is an alias for `parse()`. Every API that takes a path also takes
1460
+ the parsed result, so a path used many times can be compiled once:
1461
+
1462
+ >>> ops = dotted.compile('a.b')
1463
+ >>> dotted.get({'a': {'b': 1}}, ops)
1464
+ 1
1465
+
1459
1466
  <a id="references"></a>
1460
1467
  ### References
1461
1468
 
@@ -1410,6 +1410,13 @@ When `partial=True` (the default for `parse()`), unresolved substitutions are
1410
1410
  allowed through. When `partial=False` (the default for traversal APIs),
1411
1411
  unresolved templates raise `TypeError`.
1412
1412
 
1413
+ `compile()` is an alias for `parse()`. Every API that takes a path also takes
1414
+ the parsed result, so a path used many times can be compiled once:
1415
+
1416
+ >>> ops = dotted.compile('a.b')
1417
+ >>> dotted.get({'a': {'b': 1}}, ops)
1418
+ 1
1419
+
1413
1420
  <a id="references"></a>
1414
1421
  ### References
1415
1422
 
@@ -56,7 +56,7 @@ For full documentation including all options and flags:
56
56
  or see README.md
57
57
  """
58
58
  from .api import \
59
- parse, is_pattern, is_template, is_reference, is_indeterminate, is_simple, \
59
+ parse, compile, is_pattern, is_template, is_reference, is_indeterminate, is_simple, \
60
60
  is_inverted, is_mutable, mutable, quote, ANY, AUTO, Attrs, GroupMode, \
61
61
  set_simple_fastpath, set_parse_cache, \
62
62
  register, transform, \
@@ -84,7 +84,7 @@ __all__ = [
84
84
  # Transform
85
85
  'apply', 'apply_multi', 'register', 'transform',
86
86
  # Utility
87
- 'parse', 'assemble', 'assemble_multi', 'quote',
87
+ 'parse', 'compile', 'assemble', 'assemble_multi', 'quote',
88
88
  'is_pattern', 'is_template', 'is_reference',
89
89
  'is_indeterminate', 'is_simple',
90
90
  'is_inverted', 'is_mutable', 'mutable',
@@ -2,68 +2,18 @@
2
2
  """
3
3
  import functools
4
4
  import itertools
5
- import re
5
+ import numbers
6
6
  import types
7
7
 
8
8
  from . import base
9
9
  from . import matchers
10
+ from . import utils
10
11
 
11
12
 
12
13
  def itemof(node, val):
13
14
  return val if isinstance(node, (str, bytes)) else node.__class__([val])
14
15
 
15
16
 
16
- # ---- quoting utilities ----
17
-
18
- _RESERVED = frozenset('.[]*:|+?/=,@&()!~#{}$<>')
19
- _NEEDS_QUOTE = _RESERVED | frozenset(' \t\n\r')
20
-
21
- _NUMERIC_RE = re.compile(
22
- r'[-]?0[xX][0-9a-fA-F]+$' # hex
23
- r'|[-]?0[oO][0-7]+$' # octal
24
- r'|[-]?0[bB][01]+$' # binary
25
- r'|[-]?[0-9][0-9_]*[eE][+-]?[0-9]+$' # scientific notation
26
- r'|[-]?[0-9]+(?:_[0-9]+)+$' # underscore separators
27
- r'|[-]?[0-9]+$' # plain integers
28
- )
29
-
30
-
31
- def _needs_quoting(s):
32
- """
33
- Return True if a string key must be quoted in dotted notation.
34
- """
35
- if not s:
36
- return True
37
- # matchers.Numeric forms (integers, scientific notation, underscore separators)
38
- # are handled by the grammar and don't need quoting, even if they
39
- # contain reserved characters like '+' in '1e+10'.
40
- if s[0].isdigit() or (len(s) > 1 and s[0] == '-' and s[1].isdigit()):
41
- return not _NUMERIC_RE.match(s)
42
- if any(c in _NEEDS_QUOTE for c in s):
43
- return True
44
- return False
45
-
46
-
47
- def _is_numeric_str(s):
48
- """
49
- Return True if s is a string that parses as an integer.
50
- """
51
- try:
52
- int(s)
53
- return True
54
- except (ValueError, TypeError):
55
- return False
56
-
57
-
58
- def _quote_str(s):
59
- """
60
- Wrap a string in single quotes, escaping backslashes and single quotes.
61
- """
62
- s = s.replace('\\', '\\\\').replace("'", "\\'")
63
- return f"'{s}'"
64
-
65
-
66
-
67
17
  class BaseOp(base.TraversalOp):
68
18
  def __init__(self, *args, **kwargs):
69
19
  super().__init__(*args, **kwargs)
@@ -92,32 +42,32 @@ class BaseOp(base.TraversalOp):
92
42
  return (v for _, v in self.items(node, **kwargs))
93
43
 
94
44
  def do_update(self, ops, node, val, has_defaults, _path, nop, nop_from_unwrap=False, **kwargs):
95
- from . import engine
96
- from . import wrappers
97
45
  if not ops:
98
46
  if nop:
99
47
  return node
100
48
  if kwargs.get('strict') and not any(True for _ in self.items(node, **kwargs)):
101
49
  return node
102
50
  return self.upsert(node, val)
103
- if self.is_empty(node) and not has_defaults:
104
- if nop or isinstance(ops[0], wrappers.NopWrap):
105
- return node
106
- built = engine.updates(ops, engine.build_default(ops), val, True, _path, nop, **kwargs)
107
- return self.upsert(node, built)
51
+ items = self.update_items(node, **kwargs)
52
+ if items is None:
53
+ if not has_defaults:
54
+ if nop or isinstance(ops[0], wrappers.NopWrap):
55
+ return node
56
+ built = engine.updates(ops, engine.build_default(ops), val, True, _path, nop, **kwargs)
57
+ return self.upsert(node, built)
58
+ items = self.items(node, **kwargs)
108
59
  inner_kwargs = kwargs
109
60
  if kwargs.get('_parents') is not None:
110
61
  inner_kwargs = dict(kwargs)
111
62
  inner_kwargs['_parents'] = (node,) + kwargs['_parents']
112
63
  pass_nop = nop and not nop_from_unwrap
113
- for k, v in self.items(node, **kwargs):
64
+ for k, v in items:
114
65
  if v is None:
115
66
  v = engine.build_default(ops)
116
67
  node = self.update(node, k, engine.updates(ops, v, val, has_defaults, _path + [(self, k)], pass_nop, **inner_kwargs))
117
68
  return node
118
69
 
119
70
  def do_remove(self, ops, node, val, nop, **kwargs):
120
- from . import engine
121
71
  if not ops:
122
72
  if nop:
123
73
  return node
@@ -343,6 +293,21 @@ class AccessOp(SimpleOp):
343
293
  def is_empty(self, node):
344
294
  return not any(True for _ in self.keys(node))
345
295
 
296
+ def update_items(self, node, **kwargs):
297
+ """
298
+ One lookup, not two: is_empty() above asks whether items() yields
299
+ anything, so take the first item and keep the rest. Only when the
300
+ two would see the same items: is_empty() ignores strict, and a
301
+ subclass may define emptiness differently.
302
+ """
303
+ if kwargs.get('strict') or type(self).is_empty is not AccessOp.is_empty:
304
+ return super().update_items(node, **kwargs)
305
+ items = iter(self.items(node, **kwargs))
306
+ first = next(items, base.marker)
307
+ if first is base.marker:
308
+ return None
309
+ return itertools.chain((first,), items)
310
+
346
311
 
347
312
  class Key(AccessOp):
348
313
  """
@@ -361,18 +326,28 @@ class Key(AccessOp):
361
326
  @classmethod
362
327
  @functools.lru_cache()
363
328
  def _concrete_cached(cls, _type, val):
364
- import numbers
365
329
  if isinstance(val, numbers.Number):
366
330
  return cls(matchers.NumericQuoted(val))
367
331
  return cls(matchers.Word(val))
368
332
 
369
333
  def operator(self, top=False):
370
- q = self.op.quote()
371
334
  if top:
372
- return q
373
- return '.' + q
335
+ return self.op.quote_top()
336
+ return '.' + self.op.quote()
374
337
 
375
338
  def _items(self, node, keys, filtered=True):
339
+ if not filtered or not self.filters:
340
+ # nothing to filter: one generator, not two threaded through a filter
341
+ def _unfiltered():
342
+ for k in keys:
343
+ try:
344
+ v = node[k]
345
+ except (TypeError, KeyError, IndexError):
346
+ continue
347
+ yield (k, v)
348
+
349
+ return _unfiltered()
350
+
376
351
  curkey = None
377
352
 
378
353
  def _values():
@@ -399,7 +374,7 @@ class Key(AccessOp):
399
374
  for r in self._resolved(node=node, **kwargs))
400
375
  # Dict-like: use key matching
401
376
  if hasattr(node, 'keys'):
402
- keys = self.op.matches(node.keys()) if filtered else node.keys()
377
+ keys = self.op.match_keys(node) if filtered else node.keys()
403
378
  return self._items(node, keys, filtered)
404
379
  # In strict mode, numeric keys never coerce to list indices
405
380
  if kwargs.get('strict'):
@@ -433,7 +408,6 @@ class Key(AccessOp):
433
408
  return {self.op.value: None}
434
409
 
435
410
  def match(self, op, specials=False):
436
- from . import wrappers
437
411
  if isinstance(op, wrappers.FilterWrap):
438
412
  op = op.inner
439
413
  if not isinstance(op, Key):
@@ -503,6 +477,18 @@ class Attr(Key):
503
477
  return '@' + self.op.quote()
504
478
 
505
479
  def _items(self, node, keys, filtered=True):
480
+ if not filtered or not self.filters:
481
+ # nothing to filter: one generator, not two threaded through a filter
482
+ def _unfiltered():
483
+ for k in keys:
484
+ try:
485
+ v = getattr(node, k)
486
+ except AttributeError:
487
+ continue
488
+ yield (k, v)
489
+
490
+ return _unfiltered()
491
+
506
492
  curkey = None
507
493
 
508
494
  def _values():
@@ -600,9 +586,8 @@ class Attr(Key):
600
586
  if hasattr(node, '_replace'):
601
587
  return node._replace(**{key: val})
602
588
  # Try dataclasses.replace for frozen dataclass (skipped on 3.6)
603
- from .utils import is_dataclass, dataclass_replace
604
- if is_dataclass(node):
605
- return dataclass_replace(node, **{key: val})
589
+ if utils.is_dataclass(node):
590
+ return utils.dataclass_replace(node, **{key: val})
606
591
  raise AttributeError(f"Cannot set attribute '{key}' on {type(node).__name__}")
607
592
  def upsert(self, node, val):
608
593
  if not self.is_pattern():
@@ -622,9 +607,8 @@ class Attr(Key):
622
607
  updates = {k: val for k in keys}
623
608
  if hasattr(node, '_replace'):
624
609
  return node._replace(**updates)
625
- from .utils import is_dataclass, dataclass_replace
626
- if is_dataclass(node):
627
- return dataclass_replace(node, **updates)
610
+ if utils.is_dataclass(node):
611
+ return utils.dataclass_replace(node, **updates)
628
612
  raise AttributeError(f"Cannot set attributes on {type(node).__name__}")
629
613
 
630
614
  def pop(self, node, key):
@@ -652,7 +636,6 @@ class Slot(Key):
652
636
  @classmethod
653
637
  @functools.lru_cache()
654
638
  def _concrete_cached(cls, _type, val):
655
- import numbers
656
639
  if isinstance(val, numbers.Number):
657
640
  return cls(matchers.Numeric(val))
658
641
  return cls(matchers.String(val))
@@ -1092,10 +1075,14 @@ class Invert(SimpleOp):
1092
1075
  yield node
1093
1076
 
1094
1077
  def do_update(self, ops, node, val, has_defaults, _path, nop, nop_from_unwrap=False, **kwargs):
1095
- from . import engine
1096
1078
  return engine.removes(ops, node, val, **kwargs)
1097
1079
 
1098
1080
  def do_remove(self, ops, node, val, nop, **kwargs):
1099
- from . import engine
1100
1081
  assert val is not base.ANY, 'Value required'
1101
1082
  return engine.updates(ops, node, val, **kwargs)
1083
+
1084
+
1085
+ # engine and wrappers both import this module, so they can only be bound
1086
+ # once the classes above exist.
1087
+ from . import engine # noqa: E402
1088
+ from . import wrappers # noqa: E402
@@ -131,6 +131,9 @@ def parse(path, bindings=None, partial=True):
131
131
  return ops
132
132
 
133
133
 
134
+ compile = parse
135
+
136
+
134
137
  def quote(key, as_key=True):
135
138
  """
136
139
  Quote a key for use in a dotted path. Idempotent: quote(quote(x)) == quote(x).
@@ -315,8 +318,7 @@ def _is_mutable_container(obj):
315
318
  if hasattr(obj, '_fields') and hasattr(obj, '_replace'):
316
319
  return False
317
320
  # Frozen dataclasses (no-op on Python 3.6 — no dataclasses module)
318
- from .utils import is_dataclass
319
- if is_dataclass(obj) and obj.__dataclass_fields__:
321
+ if utils.is_dataclass(obj) and obj.__dataclass_fields__:
320
322
  # Check if frozen
321
323
  try:
322
324
  # Try to detect frozen - frozen dataclasses raise FrozenInstanceError
@@ -425,7 +427,9 @@ def build_multi(obj, paths, strict=False, bindings=None):
425
427
  for path in paths:
426
428
  ops = parse(path, bindings=bindings, partial=False)
427
429
  built = engine.build(ops, obj, strict=strict)
428
- obj = update_multi(obj, _pluck_parsed(built, ops, strict=strict), strict=strict)
430
+ # walked paths are concrete: nothing for update() to resolve or check
431
+ for found, val in _pluck_parsed(built, ops, strict=strict):
432
+ obj = engine.updates(found, obj, found.apply(val), strict=strict)
429
433
  return obj
430
434
 
431
435
 
@@ -476,8 +480,8 @@ def get(obj, path, default=None, pattern_default=(), apply_transforms=True, stri
476
480
  if apply_transforms and ops.transforms:
477
481
  return ops.apply(val)
478
482
  return val
479
- vals = engine.iter_until_cut(engine.gets(ops, obj, strict=strict))
480
- if apply_transforms:
483
+ vals = engine.values(ops, obj, strict=strict)
484
+ if apply_transforms and ops.transforms:
481
485
  vals = ( ops.apply(v) for v in vals )
482
486
  if ops.guard is not None:
483
487
  vals = (v for v in vals if ops.guard_matches(v))
@@ -522,12 +526,13 @@ def setdefault(obj, path, val, apply_transforms=True, strict=False, bindings=Non
522
526
  >>> setdefault({}, 'a.b.c', 7)
523
527
  7
524
528
  """
529
+ ops = parse(path, bindings=bindings, partial=False)
525
530
  if obj is AUTO:
526
- obj = _auto_root_from_path(path)
527
- if has(obj, path, strict=strict, bindings=bindings):
528
- return get(obj, path, apply_transforms=apply_transforms, strict=strict, bindings=bindings)
529
- obj = update(obj, path, val, apply_transforms=apply_transforms, strict=strict, bindings=bindings)
530
- return get(obj, path, apply_transforms=False, strict=strict, bindings=bindings)
531
+ obj = _auto_root_from_path(ops)
532
+ if has(obj, ops, strict=strict):
533
+ return get(obj, ops, apply_transforms=apply_transforms, strict=strict)
534
+ obj = update(obj, ops, val, apply_transforms=apply_transforms, strict=strict)
535
+ return get(obj, ops, apply_transforms=False, strict=strict)
531
536
 
532
537
 
533
538
  def setdefault_multi(obj, pathvalues, apply_transforms=True, strict=False, bindings=None):
@@ -559,8 +564,10 @@ def update_if(obj, path, val, pred=lambda val: val is not None, mutable=True, ap
559
564
  >>> update_if({}, 'a', '', pred=bool)
560
565
  {}
561
566
  """
567
+ parsed = path
562
568
  if obj is AUTO:
563
- obj = _auto_root_from_path(path)
569
+ parsed = parse(path)
570
+ obj = _auto_root_from_path(parsed)
564
571
  if not mutable and _is_mutable_container(obj):
565
572
  obj = utils.deepcopy(obj)
566
573
  mutable = True
@@ -568,7 +575,7 @@ def update_if(obj, path, val, pred=lambda val: val is not None, mutable=True, ap
568
575
  if pred is not None and not pred(val):
569
576
  return obj
570
577
 
571
- ops = parse(path, bindings=bindings, partial=False)
578
+ ops = parse(parsed, bindings=bindings, partial=False)
572
579
  return engine.updates(ops, obj, ops.apply(val) if apply_transforms else val, strict=strict)
573
580
 
574
581
 
@@ -661,8 +668,10 @@ def remove_if(obj, path, pred=lambda path: path is not None, val=ANY, mutable=Tr
661
668
  >>> remove_if({'a': 1}, None)
662
669
  {'a': 1}
663
670
  """
671
+ parsed = path
664
672
  if obj is AUTO:
665
- obj = _auto_root_from_path(path)
673
+ parsed = parse(path)
674
+ obj = _auto_root_from_path(parsed)
666
675
  if not mutable and _is_mutable_container(obj):
667
676
  obj = utils.deepcopy(obj)
668
677
  mutable = True
@@ -670,7 +679,7 @@ def remove_if(obj, path, pred=lambda path: path is not None, val=ANY, mutable=Tr
670
679
  if pred is not None and not pred(path):
671
680
  return obj
672
681
 
673
- return engine.removes(parse(path, bindings=bindings, partial=False), obj, val, strict=strict)
682
+ return engine.removes(parse(parsed, bindings=bindings, partial=False), obj, val, strict=strict)
674
683
 
675
684
 
676
685
  def remove_if_multi(obj, items, paths_only=True, pred=lambda path: path is not None, mutable=True, strict=False, bindings=None):
@@ -821,6 +830,13 @@ def match(pattern, path, groups=False, partial=True, strict=False):
821
830
  '*b'
822
831
  >>> match('*', '**')
823
832
  """
833
+ return _match_parsed(parse(pattern), parse(path), path, groups, partial)
834
+
835
+
836
+ def _match_parsed(pats, path_ops, path, groups, partial):
837
+ """
838
+ match() on parsed arguments. `path` is what a successful match returns.
839
+ """
824
840
  # groups can be a bool, a GroupMode member, or its string value.
825
841
  _patterns_only = (groups == GroupMode.patterns
826
842
  or groups == GroupMode.patterns.value)
@@ -832,19 +848,17 @@ def match(pattern, path, groups=False, partial=True, strict=False):
832
848
  matches = [m for m, ip in zip(matches, is_pat) if ip]
833
849
  return (r, tuple(matches))
834
850
 
835
- pats = parse(pattern)
836
- path_ops = parse(path)
837
-
838
851
  # Variadic ops (recursive, groups) consume variable-length path
839
852
  # segments — use the recursive matcher. On the path side they make
840
853
  # the path denote multiple expansions, which likewise needs it.
841
- if any(op.is_variadic() for op in pats) or any(op.is_variadic() for op in path_ops):
854
+ if pats.variadic or path_ops.variadic:
842
855
  result = base.match_ops(list(pats), list(path_ops), partial)
843
856
  if result is None:
844
857
  return returns(None, [])
845
858
  return returns(path, [v for v, _ in result], [p for _, p in result])
846
859
 
847
860
  # Original non-recursive match logic
861
+ collect = bool(groups)
848
862
  _matches = []
849
863
  _is_pat = []
850
864
  for idx,(pop,kop) in enumerate(zip(pats, path_ops)):
@@ -856,12 +870,17 @@ def match(pattern, path, groups=False, partial=True, strict=False):
856
870
  m = pop.match(kop, specials=True)
857
871
  if not m:
858
872
  return returns(None, [], [])
873
+ if not collect:
874
+ # captures are only returned with groups
875
+ continue
876
+ is_pat = pop.is_pattern()
859
877
  if isinstance(m, (tuple, list)):
860
- _matches.extend(_m.val for _m in m)
861
- _is_pat.extend(pop.is_pattern() for _ in m)
878
+ for _m in m:
879
+ _matches.append(_m.val)
880
+ _is_pat.append(is_pat)
862
881
  else:
863
882
  _matches.append(m.val)
864
- _is_pat.append(pop.is_pattern())
883
+ _is_pat.append(is_pat)
865
884
 
866
885
  # we've completed matching but the last item in match groups is treated 'greedily'
867
886
  assert kop is not None # sanity
@@ -875,6 +894,8 @@ def match(pattern, path, groups=False, partial=True, strict=False):
875
894
  return returns(None, [], [])
876
895
 
877
896
  # otherwise inexact (partial) match
897
+ if not collect:
898
+ return path
878
899
  # assemble remaining segments
879
900
  rpath = path_ops.assemble(start=idx)
880
901
  if pop is None:
@@ -891,7 +912,14 @@ def match_multi(pattern, iterable, groups=False, partial=True, strict=False):
891
912
  >>> list(match_multi('/h.*/', ['hello', 'there', 'hi']))
892
913
  ['hello', 'hi']
893
914
  """
894
- matches = (match(pattern, p, groups=groups, partial=partial, strict=strict) for p in iterable)
915
+ def _matches():
916
+ pats = None
917
+ for p in iterable:
918
+ if pats is None:
919
+ pats = parse(pattern)
920
+ yield _match_parsed(pats, parse(p), p, groups, partial)
921
+
922
+ matches = _matches()
895
923
  if groups:
896
924
  return (m for m in matches if m[0])
897
925
  return (m for m in matches if m)
@@ -925,8 +953,12 @@ def translate(path, pattern_map):
925
953
  True
926
954
  """
927
955
  map_items = pattern_map.items() if hasattr(pattern_map, 'items') else pattern_map
956
+ path_ops = None
928
957
  for pattern, template in map_items:
929
- (r, groups) = match(pattern, path, groups=GroupMode.patterns, partial=False)
958
+ pats = parse(pattern)
959
+ if path_ops is None:
960
+ path_ops = parse(path)
961
+ (r, groups) = _match_parsed(pats, path_ops, path, GroupMode.patterns, False)
930
962
  if not r:
931
963
  continue
932
964
  try:
@@ -1049,7 +1081,7 @@ def apply_multi(obj, patterns, strict=False, bindings=None):
1049
1081
  if ops in seen:
1050
1082
  continue
1051
1083
  seen[ops] = None
1052
- first = next(engine.iter_until_cut(engine.gets(ops, obj, strict=strict)), _marker)
1084
+ first = next(engine.values(ops, obj, strict=strict), _marker)
1053
1085
  if first is _marker:
1054
1086
  continue
1055
1087
  val = ops.apply(first)
@@ -1075,17 +1107,26 @@ def pluck_multi(obj, patterns, default=None, strict=False, bindings=None):
1075
1107
  >>> list(pluck_multi(d, ('hello', 'a.b')))
1076
1108
  [('hello', 7), ('a.b', 'seven')]
1077
1109
  """
1110
+ return ((field, val) for field, _, val in _pluck_written(obj, patterns, strict=strict, bindings=bindings))
1111
+
1112
+
1113
+ def _pluck_written(obj, patterns, strict=False, bindings=None):
1114
+ """
1115
+ Yield deduped (field, parsed field, value) triples from obj matching any
1116
+ of `patterns`: what pluck_multi yields, plus each field already parsed,
1117
+ so a caller that needs the field as a path does not parse the string.
1118
+ """
1078
1119
  seen = {}
1079
1120
  for pattern in patterns:
1080
1121
  ops = parse(pattern, bindings=bindings, partial=False)
1081
1122
  for path, val in engine.walk(ops, obj, paths=True, strict=strict):
1082
1123
  if path is utypes.CUT_SENTINEL:
1083
1124
  break
1084
- field = results.Dotted({'ops': path, 'transforms': ops.transforms}).assemble()
1125
+ field, parsed = results.Dotted({'ops': path, 'transforms': ops.transforms}).written()
1085
1126
  if field in seen:
1086
1127
  continue
1087
1128
  seen[field] = None
1088
- yield (field, val)
1129
+ yield (field, parsed, val)
1089
1130
 
1090
1131
 
1091
1132
  def pluck(obj, pattern, default=None, strict=False, bindings=None):
@@ -1097,10 +1138,11 @@ def pluck(obj, pattern, default=None, strict=False, bindings=None):
1097
1138
  >>> pluck(d, 'a.b')
1098
1139
  ('a.b', 'seven')
1099
1140
  """
1100
- out = tuple(pluck_multi(obj, (pattern,), default=default, strict=strict, bindings=bindings))
1141
+ parsed = parse(pattern)
1142
+ out = tuple(pluck_multi(obj, (parsed,), default=default, strict=strict, bindings=bindings))
1101
1143
  if not out:
1102
1144
  return ()
1103
- if is_pattern(pattern):
1145
+ if is_pattern(parsed):
1104
1146
  return out
1105
1147
  return out[0]
1106
1148
 
@@ -1196,16 +1238,30 @@ def unpack(obj, attrs=None, project=None, partial=True):
1196
1238
  extra = ', @/(?!__).*/'
1197
1239
  else:
1198
1240
  extra = ', @/__.*/'
1199
- result = dict(pluck(obj, f'*(*#, [*]:!(str, bytes){extra}):-2(.*, []{extra})##, (*, []{extra})'))
1241
+ pattern = f'*(*#, [*]:!(str, bytes){extra}):-2(.*, []{extra})##, (*, []{extra})'
1200
1242
  if project is None:
1201
- return result
1243
+ return dict(pluck(obj, pattern))
1244
+ leaves = list(_pluck_written(obj, (pattern,)))
1202
1245
  if isinstance(project, str):
1203
1246
  project = [project]
1204
1247
  # Normalize each entry to (pattern, partial); bare patterns inherit the
1205
1248
  # global `partial`, (pattern, partial) tuples override it per-field.
1206
1249
  specs = [(p, partial) if isinstance(p, str) else tuple(p) for p in project]
1207
- return {k: v for k, v in result.items()
1208
- if any(match(pat, k, partial=pp) for pat, pp in specs)}
1250
+ compiled = [None] * len(specs)
1251
+
1252
+ def selected(field, path_ops):
1253
+ """
1254
+ True if a leaf matches any projection. Each pattern is compiled when
1255
+ first reached; the leaf's path comes parsed from the walk.
1256
+ """
1257
+ for i, (pat, pp) in enumerate(specs):
1258
+ if compiled[i] is None:
1259
+ compiled[i] = parse(pat)
1260
+ if _match_parsed(compiled[i], path_ops, field, False, pp):
1261
+ return True
1262
+ return False
1263
+
1264
+ return {field: val for field, path_ops, val in leaves if selected(field, path_ops)}
1209
1265
 
1210
1266
 
1211
1267
  def items(obj, attrs=None, project=None, partial=True):