diffcone 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
diffcone/cython.py ADDED
@@ -0,0 +1,757 @@
1
+ """Cython sources at function level (roadmap item 7).
2
+
3
+ A tolerant reader of ``.pyx``, ``.pxd`` and ``.pxi`` files that works from
4
+ indentation alone: diffcone is stdlib only, so it does not use Cython's
5
+ parser, and it does not need to. What evidence planning asks of a Cython file
6
+ is which functions changed and what each function's body mentions.
7
+
8
+ * A *function* is a ``def``, ``cpdef`` or ``cdef`` statement with a parameter
9
+ list and a body: its header (which may span lines) ends with ``:`` at
10
+ bracket depth 0. A ``cdef`` declaration without a body (``.pxd`` files), a
11
+ variable, a struct, enum, union, extern or fused block, and a ``ctypedef``
12
+ are not functions.
13
+ * ``cdef class`` and ``class`` blocks are scopes: the functions in them are
14
+ methods, named ``Class.method``. A name that repeats in one scope (a
15
+ property's setter after its getter) is numbered: ``Class.attr#2``.
16
+ * A function's span starts at its first decorator, where a code object's
17
+ first line is, and ends at its last indented line. A function nested in
18
+ another belongs to it: its text is part of the parent's body, as nested
19
+ functions are part of their parent in the Python index.
20
+ * ``nogil`` and ``cpdef`` are recorded, and so is every name the body
21
+ mentions: a profiled build raises no call event for a ``nogil`` function,
22
+ nor for a ``cpdef`` method's C body under ``skip_dispatch`` (an explicit
23
+ ``Base.method(self, ...)`` call), so those are found through the functions
24
+ that name them, which also covers a function taken as a pointer.
25
+ * Blank lines and comment-only lines are ignored by both hashes, except
26
+ compiler directives (``# cython: boundscheck=False``, ``# distutils:``),
27
+ which change how every function in the file is compiled. Everything
28
+ outside the functions (directives, ``cimport``, ``ctypedef``, structs,
29
+ class headers and attribute declarations, constants, ``include``) is
30
+ hashed together.
31
+ * Outside the functions the file is also read as statements, with Python's
32
+ ``tokenize`` (Cython's lexical syntax is Python's): each statement's
33
+ scope (the module or a class), the names it binds, whether Python code
34
+ can see them, and a hash (roadmap item 8). One statement per item of an
35
+ ``import``/``cimport`` list; the declarator of a ``cdef``, ``ctypedef`` or
36
+ ``DEF`` declaration and of an extern or ``.pxd`` function declaration;
37
+ assignment and ``del`` targets; a struct, union, enum, fused type or
38
+ ``cppclass`` with its members as one statement (their order sets enum
39
+ values and layout); a class header with its bases. A statement that binds
40
+ nothing by name (a bare call, a docstring, ``include``, ``IF``, a
41
+ compound statement, a star import) is *unbounded*. A file ``tokenize``
42
+ cannot read has no statements, and any change outside its functions
43
+ stays file-level.
44
+ """
45
+
46
+ from __future__ import annotations
47
+
48
+ import hashlib
49
+ import io
50
+ import keyword
51
+ import re
52
+ import tokenize
53
+ from collections.abc import Iterable
54
+ from dataclasses import dataclass, field
55
+
56
+ CYTHON_SUFFIXES = (".pyx", ".pxd", ".pxi")
57
+
58
+ _HEAD = re.compile(r"^(\s*)(cpdef|cdef|def)\b")
59
+ _UNPROFILED = re.compile(r"@\s*(?:cython\.)?(?:profile|linetrace)\s*\(\s*False\s*\)")
60
+ _CLASS = re.compile(
61
+ r"^(\s*)(?:cdef\s+(?:(?:public|readonly|final)\s+)*class|class)\s+([A-Za-z_]\w*)"
62
+ )
63
+ _NOT_FUNC = re.compile(
64
+ r"^\s*cdef\s+(?:class|struct|enum|union|extern|packed|fused|cppclass|"
65
+ r"public\s+(?:struct|enum))\b"
66
+ )
67
+ _CALLABLE = re.compile(r"([A-Za-z_]\w*)\s*\(")
68
+ _WORD = re.compile(r"\b[A-Za-z_]\w*\b")
69
+ _DIRECTIVE = re.compile(r"^\s*#\s*(?:cython|distutils)\s*:")
70
+
71
+
72
+ def is_cython(path: str) -> bool:
73
+ return path.endswith(CYTHON_SUFFIXES)
74
+
75
+
76
+ @dataclass(frozen=True)
77
+ class CythonFunction:
78
+ name: str # qualified within the file
79
+ start: int # first decorator line (1-based)
80
+ end: int
81
+ body_hash: str
82
+ nogil: bool
83
+ cpdef: bool
84
+ names: frozenset[str] = field(default_factory=frozenset, compare=False)
85
+ python: bool = False # def or cpdef: Python code can see it
86
+ # ``@cython.profile(False)`` or ``@cython.linetrace(False)``: a profiled
87
+ # build never reports it, like a ``nogil`` function.
88
+ unprofiled: bool = False
89
+
90
+ @property
91
+ def scope(self) -> str:
92
+ """The class holding it, or "" for a module-level function."""
93
+ return self.name.rsplit(".", 1)[0] if "." in self.name else ""
94
+
95
+ @property
96
+ def simple_name(self) -> str:
97
+ return self.name.rsplit(".", 1)[-1].split("#")[0]
98
+
99
+
100
+ # Statement kinds.
101
+ IMPORT = "import" # one item of an import or cimport list
102
+ DECLARATION = "declaration" # cdef, ctypedef, DEF, extern and .pxd declarations
103
+ ASSIGNMENT = "assignment" # Python assignment or del at module or class level
104
+ TYPE = "type" # struct, union, enum, fused type or cppclass, members included
105
+ CLASS = "class" # a class header
106
+ CODE = "code" # binds nothing by name: unbounded
107
+
108
+
109
+ @dataclass(frozen=True)
110
+ class CythonStatement:
111
+ """A statement outside every function."""
112
+
113
+ scope: str # "" for the module, else the class's qualified name
114
+ kind: str
115
+ names: tuple[str, ...] # the names it binds; () when unbounded
116
+ visible: bool # whether Python code can see those names
117
+ hash: str
118
+ line: int = field(compare=False)
119
+ why: str = "" # for CODE: what it is
120
+ bases: tuple[str, ...] = () # for CLASS: the names in its bases
121
+ # For IMPORT: the module named as written, then the name imported from
122
+ # it (``.np_datetime npy_datetimestruct``, ``pandas._libs util``).
123
+ module: str = ""
124
+
125
+
126
+ @dataclass(frozen=True)
127
+ class CythonModule:
128
+ path: str
129
+ functions: tuple[CythonFunction, ...]
130
+ outside_hash: str
131
+ # None when tokenize could not read the file.
132
+ statements: tuple[CythonStatement, ...] | None = None
133
+
134
+ def by_name(self) -> dict[str, CythonFunction]:
135
+ return {f.name: f for f in self.functions}
136
+
137
+ def function_at(self, line: int) -> CythonFunction | None:
138
+ """The function whose span holds ``line`` (spans do not nest)."""
139
+ for f in self.functions:
140
+ if f.start <= line <= f.end:
141
+ return f
142
+ return None
143
+
144
+
145
+ def pxd_stem(path: str) -> str:
146
+ """The name a ``.pxd`` file is cimported by: its stem, or for a
147
+ package's ``__init__.pxd`` the package directory's name."""
148
+ parts = path.rsplit(".", 1)[0].split("/")
149
+ return parts[-2] if parts[-1] == "__init__" and len(parts) > 1 else parts[-1]
150
+
151
+
152
+ def names_module(statement: CythonStatement, stem: str) -> bool:
153
+ """Whether an import statement may name the module ``stem`` (any
154
+ component of what it names, so ``from pandas._libs cimport util`` names
155
+ ``util``, and a relative import is not resolved)."""
156
+ return stem in re.split(r"[.\s]+", statement.module)
157
+
158
+
159
+ def symbol_id(path: str, function: str) -> str:
160
+ """How evidence names a Cython function: ``<path>::<qualified name>``."""
161
+ return f"{path}::{function}"
162
+
163
+
164
+ def _code(line: str) -> str:
165
+ """The line without its comment (a ``#`` inside a string is rare enough
166
+ in a header or a mention to be read as a comment)."""
167
+ return line.split("#", 1)[0]
168
+
169
+
170
+ def _significant(line: str) -> bool:
171
+ stripped = line.strip()
172
+ return bool(stripped) and not stripped.startswith("#")
173
+
174
+
175
+ def _in_strings(lines: list[str]) -> list[bool]:
176
+ """For each line, whether it starts inside a triple-quoted string: such
177
+ a line (a docstring's continuation, often at column 0) says nothing about
178
+ where a block ends."""
179
+ out = []
180
+ quote: str | None = None # the open triple quote
181
+ for line in lines:
182
+ out.append(quote is not None)
183
+ k = 0
184
+ while k < len(line):
185
+ if quote is not None:
186
+ end = line.find(quote, k)
187
+ if end < 0:
188
+ break
189
+ quote, k = None, end + 3
190
+ continue
191
+ ch = line[k]
192
+ if ch == "#":
193
+ break
194
+ if line.startswith(('"""', "'''"), k):
195
+ quote, k = line[k : k + 3], k + 3
196
+ continue
197
+ if ch in "\"'": # a one-line string: skip to its closing quote
198
+ close = k + 1
199
+ while close < len(line) and line[close] != ch:
200
+ close += 2 if line[close] == "\\" else 1
201
+ k = close + 1
202
+ continue
203
+ k += 1
204
+ return out
205
+
206
+
207
+ def _indent(line: str) -> int:
208
+ return len(line) - len(line.lstrip(" \t"))
209
+
210
+
211
+ def _hash(lines: Iterable[str], *, directives: bool = False) -> str:
212
+ digest = hashlib.sha256()
213
+ for line in lines:
214
+ if _significant(line) or (directives and _DIRECTIVE.match(line)):
215
+ digest.update(line.rstrip().encode("utf-8", "surrogateescape") + b"\n")
216
+ return digest.hexdigest()[:24]
217
+
218
+
219
+ def _function_name(header: str) -> str | None:
220
+ """The name a function header defines, or None for anything else (a
221
+ declaration such as ``cdef int64_t **x = <int64_t**>malloc(n)``)."""
222
+ rest = re.sub(r"^\s*(cpdef|cdef|def)\b", "", header)
223
+ rest = rest.lstrip()
224
+ if rest.startswith("("): # a C tuple return type: cdef (int, int) f(...)
225
+ depth = 0
226
+ for k, ch in enumerate(rest):
227
+ depth += {"(": 1, ")": -1}.get(ch, 0)
228
+ if depth == 0:
229
+ rest = rest[k + 1 :]
230
+ break
231
+ m = _CALLABLE.search(rest)
232
+ if m is None or "=" in rest[: m.start()]:
233
+ return None
234
+ return m.group(1)
235
+
236
+
237
+ def read(path: str, text: str) -> CythonModule:
238
+ lines = text.split("\n")
239
+ n = len(lines)
240
+ in_string = _in_strings(lines)
241
+
242
+ def structural(k: int) -> bool:
243
+ return _significant(lines[k]) and not in_string[k]
244
+
245
+ functions: list[CythonFunction] = []
246
+ seen: dict[str, int] = {}
247
+ covered: set[int] = set() # 0-based line indexes inside a function's span
248
+ scopes: list[tuple[int, str]] = [] # (indent, class name)
249
+ i = 0
250
+ while i < n:
251
+ line = lines[i]
252
+ if not structural(i):
253
+ i += 1
254
+ continue
255
+ indent = _indent(line)
256
+ while scopes and indent <= scopes[-1][0]:
257
+ scopes.pop()
258
+ m = _CLASS.match(line)
259
+ if m and _code(line).rstrip().endswith(":"):
260
+ scopes.append((indent, m.group(2)))
261
+ i += 1
262
+ continue
263
+ if not _HEAD.match(line) or _NOT_FUNC.match(line) or "(" not in _code(line):
264
+ i += 1
265
+ continue
266
+ # The header, up to ':' at bracket depth 0.
267
+ header, j, depth, is_function = "", i, 0, False
268
+ while j < n:
269
+ part = _code(lines[j])
270
+ header += " " + part.strip()
271
+ depth += part.count("(") + part.count("[") - part.count(")") - part.count("]")
272
+ stripped = part.rstrip()
273
+ if depth <= 0 and not stripped.endswith("\\"):
274
+ is_function = stripped.endswith(":")
275
+ break
276
+ j += 1
277
+ name = _function_name(header) if is_function else None
278
+ if name is None:
279
+ i += 1
280
+ continue
281
+ last = j
282
+ k = j + 1
283
+ while k < n:
284
+ if structural(k):
285
+ if _indent(lines[k]) <= indent:
286
+ break
287
+ last = k
288
+ elif _significant(lines[k]):
289
+ last = k # inside a string that started in the body
290
+ k += 1
291
+ first = i
292
+ while first > 0 and lines[first - 1].strip().startswith("@"):
293
+ if _indent(lines[first - 1]) != indent:
294
+ break
295
+ first -= 1
296
+ qualified = ".".join([c for _, c in scopes] + [name])
297
+ seen[qualified] = seen.get(qualified, 0) + 1
298
+ if seen[qualified] > 1:
299
+ qualified = f"{qualified}#{seen[qualified]}"
300
+ # What the header (types, defaults, decorators) and body mention; the
301
+ # name the header defines is not a mention of itself.
302
+ names: set[str] = set()
303
+ for b in range(first, j + 1):
304
+ names.update(_WORD.findall(_code(lines[b])))
305
+ names.discard(name)
306
+ for b in range(j + 1, last + 1):
307
+ names.update(_WORD.findall(_code(lines[b])))
308
+ tail = header.rsplit(")", 1)[-1]
309
+ functions.append(
310
+ CythonFunction(
311
+ name=qualified,
312
+ start=first + 1,
313
+ end=last + 1,
314
+ body_hash=_hash(lines[first : last + 1]),
315
+ nogil=bool(re.search(r"\bnogil\b", tail)),
316
+ cpdef=header.lstrip().startswith("cpdef"),
317
+ names=frozenset(names),
318
+ python=header.lstrip().startswith(("def", "cpdef")),
319
+ unprofiled=any(_UNPROFILED.search(_code(lines[b])) for b in range(first, j + 1)),
320
+ )
321
+ )
322
+ covered.update(range(first, last + 1))
323
+ i = last + 1 # what is nested in the function is part of it
324
+ outside = _hash((line for k, line in enumerate(lines) if k not in covered), directives=True)
325
+ return CythonModule(path, tuple(functions), outside, _statements(text, covered))
326
+
327
+
328
+ # --------------------------------------------------------------------------- statements
329
+
330
+ # Words that qualify a C declaration and are never the name it declares.
331
+ _QUALIFIERS = frozenset(
332
+ "cdef cpdef ctypedef public readonly api extern inline const volatile unsigned "
333
+ "signed long short struct enum union packed nogil noexcept fused cppclass except "
334
+ "gil static DEF class".split()
335
+ )
336
+ _AGGREGATES = frozenset({"struct", "union", "enum", "fused", "cppclass"})
337
+ _COMPOUND = frozenset("if elif else try except finally for while with async def match case".split())
338
+ _SKIP_TOKENS = (
339
+ tokenize.NL,
340
+ tokenize.COMMENT,
341
+ tokenize.INDENT,
342
+ tokenize.DEDENT,
343
+ tokenize.ENCODING,
344
+ )
345
+ _STRING_TOKENS = {tokenize.STRING} | {
346
+ getattr(tokenize, n)
347
+ for n in ("FSTRING_START", "FSTRING_MIDDLE", "FSTRING_END")
348
+ if hasattr(tokenize, n)
349
+ }
350
+
351
+
352
+ def _is_dunder(name: str) -> bool:
353
+ return len(name) > 4 and name.startswith("__") and name.endswith("__")
354
+
355
+
356
+ def _depths(words: list[str]) -> list[int]:
357
+ """Bracket depth before each word."""
358
+ out, depth = [], 0
359
+ for w in words:
360
+ if w in (")", "]", "}"):
361
+ depth -= 1
362
+ out.append(depth)
363
+ if w in ("(", "[", "{"):
364
+ depth += 1
365
+ return out
366
+
367
+
368
+ def _split(words: list[str], sep: str) -> list[list[str]]:
369
+ """Split at ``sep`` outside brackets."""
370
+ parts: list[list[str]] = [[]]
371
+ for w, d in zip(words, _depths(words), strict=True):
372
+ if w == sep and d == 0:
373
+ parts.append([])
374
+ else:
375
+ parts[-1].append(w)
376
+ return parts
377
+
378
+
379
+ def _identifier(word: str) -> bool:
380
+ return word.isidentifier() and not keyword.iskeyword(word)
381
+
382
+
383
+ def _declarators(words: list[str]) -> list[str] | None:
384
+ """The names a C declaration declares (``cdef int64_t a, b = 1``,
385
+ ``int f(int x) nogil``, ``ctypedef (int, int) pair_t``, ``ctypedef int
386
+ (*func_t)(int)``), or None when it cannot be read."""
387
+ decl = _split(words, "=")[0]
388
+ depths = _depths(decl)
389
+ if decl and decl[0] == "(": # a C tuple type comes first
390
+ k = 1
391
+ while k < len(decl) and not (decl[k] == ")" and depths[k] == 0):
392
+ k += 1
393
+ decl, depths = decl[k + 1 :], depths[k + 1 :]
394
+ for k, w in enumerate(decl):
395
+ if w == "(" and depths[k] == 0:
396
+ if k + 1 < len(decl) and decl[k + 1] in ("*", "&"): # a function pointer
397
+ inner = [x for x in decl[k + 1 :] if _identifier(x) and x not in _QUALIFIERS]
398
+ return inner[:1] or None
399
+ before = [x for x in decl[:k] if _identifier(x) and x not in _QUALIFIERS]
400
+ return before[-1:] or None
401
+ names = []
402
+ for part in _split(decl, ","):
403
+ top = [w for w, d in zip(part, _depths(part), strict=True) if d == 0]
404
+ found = [w for w in top if _identifier(w) and w not in _QUALIFIERS]
405
+ if not found:
406
+ return None
407
+ names.append(found[-1])
408
+ return names or None
409
+
410
+
411
+ def _import_items(words: list[str]) -> list[tuple[str, str, str]] | None:
412
+ """(bound name, module, item text) per item of an import or cimport
413
+ statement, or None for a star import."""
414
+ if words[0] == "from":
415
+ k = words.index("import") if "import" in words else words.index("cimport")
416
+ module = "".join(words[1:k])
417
+ items = [w for w in words[k + 1 :] if w not in ("(", ")")]
418
+ out = []
419
+ for part in _split(items, ","):
420
+ if not part:
421
+ continue
422
+ if part == ["*"]:
423
+ return None
424
+ # The module and the name imported from it (a submodule, maybe).
425
+ target = f"{module} {part[0]}"
426
+ out.append((part[-1], target, f"{' '.join(words[: k + 1])} {' '.join(part)}"))
427
+ return out
428
+ out = []
429
+ for part in _split(words[1:], ","):
430
+ if not part:
431
+ continue
432
+ module = "".join(part[: part.index("as")] if "as" in part else part)
433
+ # import a.b binds a; import a.b as c binds c.
434
+ out.append((part[-1] if "as" in part else part[0], module, f"{words[0]} {' '.join(part)}"))
435
+ return out
436
+
437
+
438
+ def _assigned(words: list[str]) -> list[str] | None:
439
+ """The names a Python assignment, augmented or annotated, binds or
440
+ changes (``a = b = 1``, ``x[k] += 1`` changes ``x``), or None."""
441
+ depths = _depths(words)
442
+ for k, w in enumerate(words):
443
+ if depths[k] == 0 and w.endswith("=") and w not in ("==", "<=", ">=", "!=") and w != "=":
444
+ targets = [words[:k]] # augmented
445
+ break
446
+ else:
447
+ parts = _split(words, "=")
448
+ if len(parts) > 1:
449
+ targets = parts[:-1]
450
+ elif ":" in words: # x: int
451
+ targets = [words]
452
+ else:
453
+ return None
454
+ names = []
455
+ for target in targets:
456
+ target = _split(target, ":")[0] # an annotation
457
+ top = [w for w, d in zip(target, _depths(target), strict=True) if d == 0]
458
+ names += (
459
+ [w for w in top if _identifier(w)][:1]
460
+ if "." in top or "[" in top
461
+ else [w for w in top if _identifier(w)]
462
+ )
463
+ return names or None
464
+
465
+
466
+ def _statements(text: str, covered: set[int]) -> tuple[CythonStatement, ...] | None:
467
+ try:
468
+ tokens = list(tokenize.generate_tokens(io.StringIO(text).readline))
469
+ except (tokenize.TokenError, SyntaxError):
470
+ return None
471
+ logical: list[list[tokenize.TokenInfo]] = []
472
+ current: list[tokenize.TokenInfo] = []
473
+ directives = [
474
+ (tok.start[0], tok.string.strip())
475
+ for tok in tokens
476
+ if tok.type == tokenize.COMMENT and _DIRECTIVE.match(tok.string)
477
+ ]
478
+ for tok in tokens:
479
+ if tok.type in (tokenize.NEWLINE, tokenize.ENDMARKER):
480
+ if current:
481
+ logical.append(current)
482
+ current = []
483
+ elif tok.type not in _SKIP_TOKENS:
484
+ current.append(tok)
485
+
486
+ out: list[CythonStatement] = []
487
+ scopes: list[tuple[int, str]] = [] # (column, class qualified name)
488
+ # (column, kind, header words, statement row) for extern, declaration
489
+ # and aggregate blocks; an aggregate collects its members' words.
490
+ blocks: list[tuple[int, str, list[str], int]] = []
491
+ members: list[list[str]] = []
492
+
493
+ def scope() -> str:
494
+ return scopes[-1][1] if scopes else ""
495
+
496
+ def digest(*parts: str) -> str:
497
+ return hashlib.sha256("\x00".join(parts).encode("utf-8", "surrogateescape")).hexdigest()[
498
+ :24
499
+ ]
500
+
501
+ def add(
502
+ kind: str, names, visible: bool, row: int, words, why: str = "", bases=(), module=""
503
+ ) -> None:
504
+ names = tuple(sorted(set(names)))
505
+ if kind != CODE and any(_is_dunder(n) for n in names):
506
+ kind, why = (
507
+ CODE,
508
+ f"binds a special name ({', '.join(n for n in names if _is_dunder(n))})",
509
+ )
510
+ if kind == CODE:
511
+ names, visible = (), False
512
+ text = " ".join(words) if isinstance(words, list) else words
513
+ out.append(
514
+ CythonStatement(
515
+ scope(), kind, names, visible, digest(kind, text), row, why, bases, module
516
+ )
517
+ )
518
+
519
+ for row, text in directives:
520
+ add(CODE, (), False, row, text, why="a compiler directive")
521
+
522
+ def close_aggregate() -> None:
523
+ col, kind, header, row = blocks.pop()
524
+ name = [w for w in header if _identifier(w) and w not in _QUALIFIERS and w != ":"]
525
+ names = name[-1:]
526
+ enum = "enum" in header
527
+ for member in members:
528
+ if enum:
529
+ names += [w for w in _split(member, "=")[0] if _identifier(w)]
530
+ elif "fused" not in header:
531
+ names += _declarators([w for w in member if w not in _QUALIFIERS]) or []
532
+ text = " ".join(header) + " | " + " | ".join(" ".join(m) for m in members)
533
+ add(TYPE, names, header[0] == "cpdef", row, text)
534
+ members.clear()
535
+
536
+ for line in logical:
537
+ row, col = line[0].start
538
+ while blocks and col <= blocks[-1][0]:
539
+ if blocks[-1][1] == TYPE:
540
+ close_aggregate()
541
+ else:
542
+ blocks.pop()
543
+ while scopes and col <= scopes[-1][0]:
544
+ scopes.pop()
545
+ if row - 1 in covered:
546
+ continue
547
+ words = [t.string for t in line]
548
+ if blocks and blocks[-1][1] == TYPE:
549
+ members.append(words)
550
+ continue
551
+ block = blocks[-1] if blocks else None
552
+ first = words[0]
553
+ header = words[-1] == ":"
554
+ cdef = first in ("cdef", "cpdef", "ctypedef")
555
+
556
+ if (first == "class" or (cdef and "class" in words[:4])) and header:
557
+ k = words.index("class")
558
+ name = words[k + 1]
559
+ bases = tuple(w for w in words[k + 2 : -1] if _identifier(w))
560
+ add(CLASS, [name], True, row, words, bases=bases)
561
+ scopes.append((col, f"{scope()}.{name}" if scope() else name))
562
+ continue
563
+ if header and cdef and words[1:2] == ["extern"]:
564
+ blocks.append((col, DECLARATION, words[:-1], row))
565
+ continue
566
+ if header and cdef and _AGGREGATES & set(words[:-1]):
567
+ blocks.append((col, TYPE, words[:-1], row))
568
+ continue
569
+ if header and cdef and all(w in _QUALIFIERS for w in words[:-1]): # cdef:, cdef public:
570
+ blocks.append((col, DECLARATION, words[:-1], row))
571
+ continue
572
+ if first in ("IF", "ELIF", "ELSE"):
573
+ add(CODE, (), False, row, words, why="compile-time IF")
574
+ continue
575
+ if header or first in _COMPOUND:
576
+ add(CODE, (), False, row, words, why="code that runs at import")
577
+ continue
578
+ if first == "include":
579
+ add(CODE, (), False, row, words, why="include")
580
+ continue
581
+ if first in ("import", "cimport") or (
582
+ first == "from" and ("import" in words or "cimport" in words)
583
+ ):
584
+ items = _import_items(words)
585
+ if items is None:
586
+ add(CODE, (), False, row, words, why="a star import")
587
+ continue
588
+ python = "cimport" not in words
589
+ for name, module, item in items:
590
+ add(IMPORT, [name], python, row, item, module=module)
591
+ continue
592
+ if first in ("pass", "...") and len(words) == 1:
593
+ continue
594
+ if all(t.type in _STRING_TOKENS for t in line):
595
+ add(CODE, (), False, row, words, why="a docstring")
596
+ continue
597
+ prefix = block[2] if block is not None else []
598
+ if cdef or first == "DEF" or (block is not None and block[1] == DECLARATION):
599
+ body = [w for w in words if w not in _QUALIFIERS] if first != "DEF" else words[1:2]
600
+ if cdef and words[1:2] == ["class"]: # cdef class X (a forward declaration)
601
+ body = words[2:3]
602
+ names = _declarators(body) if first != "DEF" else body
603
+ if not names:
604
+ add(CODE, (), False, row, words, why="a declaration diffcone cannot read")
605
+ continue
606
+ qualifiers = set(prefix) | set(words)
607
+ visible = bool({"public", "readonly", "cpdef"} & qualifiers)
608
+ add(DECLARATION, names, visible, row, prefix + ["|"] + words)
609
+ continue
610
+ if first == "del":
611
+ add(ASSIGNMENT, [w for w in words[1:] if _identifier(w)], True, row, words)
612
+ continue
613
+ names = _assigned(words)
614
+ if names:
615
+ add(ASSIGNMENT, names, True, row, words)
616
+ else:
617
+ add(CODE, (), False, row, words, why="an expression statement that runs at import")
618
+ while blocks:
619
+ if blocks[-1][1] == TYPE:
620
+ close_aggregate()
621
+ else:
622
+ blocks.pop()
623
+ return tuple(out)
624
+
625
+
626
+ @dataclass(frozen=True)
627
+ class CythonName:
628
+ """A name bound outside every function body whose binding changed, or a
629
+ function added or deleted (roadmap item 8)."""
630
+
631
+ path: str
632
+ scope: str # "" for the module, else the class's qualified name
633
+ name: str
634
+ change: str # added, deleted or changed
635
+ visible: bool # Python code can see it
636
+ attribute: bool # a class attribute declared or assigned in the class body
637
+
638
+
639
+ @dataclass(frozen=True)
640
+ class CythonChanges:
641
+ """What differs between two snapshots' Cython sources."""
642
+
643
+ # Functions whose body changed, present on both sides: (path, name).
644
+ functions: tuple[tuple[str, str], ...] = ()
645
+ # Files changed in a way not attributed to a function or name: path -> why.
646
+ files: tuple[tuple[str, str], ...] = ()
647
+ names: tuple[CythonName, ...] = ()
648
+
649
+
650
+ def _empty(path: str) -> CythonModule:
651
+ return CythonModule(path, (), "", ())
652
+
653
+
654
+ def cython_changes(
655
+ before: dict[str, CythonModule], after: dict[str, CythonModule]
656
+ ) -> CythonChanges:
657
+ """Function bodies that changed, names whose binding changed outside the
658
+ functions, and what neither bounds, file by file.
659
+
660
+ An added or deleted ``.pyx`` or ``.pxi`` file is file-level (a new
661
+ extension module, or an ``include`` elsewhere); a ``.pxd`` file's
662
+ declarations are names like any other. A file without statements
663
+ (``tokenize`` could not read it) is file-level for any change outside
664
+ its functions."""
665
+ functions: list[tuple[str, str]] = []
666
+ files: list[tuple[str, str]] = []
667
+ names: list[CythonName] = []
668
+ for path in sorted(before.keys() | after.keys()):
669
+ a, b = before.get(path), after.get(path)
670
+ if a is None or b is None:
671
+ if not path.endswith(".pxd"):
672
+ files.append((path, "added" if a is None else "deleted"))
673
+ continue
674
+ a, b = a or _empty(path), b or _empty(path)
675
+ fa, fb = a.by_name(), b.by_name()
676
+ functions.extend(
677
+ (path, f) for f in sorted(fa.keys() & fb.keys()) if fa[f].body_hash != fb[f].body_hash
678
+ )
679
+ if a.statements is None or b.statements is None:
680
+ if a.outside_hash != b.outside_hash or fa.keys() != fb.keys():
681
+ files.append((path, "changed outside its functions"))
682
+ continue
683
+ why = _outside(path, a, b, names)
684
+ if why is not None:
685
+ files.append((path, why))
686
+ return CythonChanges(tuple(functions), tuple(files), tuple(names))
687
+
688
+
689
+ def _outside(path: str, a: CythonModule, b: CythonModule, out: list[CythonName]) -> str | None:
690
+ """Append the names whose binding changed between ``a`` and ``b``; return
691
+ why the change is file-level instead, or None."""
692
+ sa, sb = a.statements or (), b.statements or ()
693
+ # Code that binds nothing by name: any change is unbounded.
694
+ code_a = [(s.scope, s.hash) for s in sa if s.kind == CODE]
695
+ code_b = [(s.scope, s.hash) for s in sb if s.kind == CODE]
696
+ if code_a != code_b:
697
+ changed = [
698
+ s
699
+ for s in (*sa, *sb)
700
+ if s.kind == CODE and (s.scope, s.hash) not in set(code_a) & set(code_b)
701
+ ]
702
+ what = changed[0].why if changed else "statements that bind no name moved"
703
+ return f"changed outside its functions: {what} (line {changed[0].line if changed else '?'})"
704
+
705
+ def bindings(statements, functions):
706
+ by_key: dict[tuple[str, str], list[str]] = {}
707
+ meta: dict[tuple[str, str], tuple[bool, bool, str]] = {} # visible, attribute, kind
708
+ for s in statements:
709
+ if s.kind == CODE:
710
+ continue
711
+ for n in s.names:
712
+ by_key.setdefault((s.scope, n), []).append(s.hash)
713
+ visible, attribute, _ = meta.get((s.scope, n), (False, False, ""))
714
+ meta[(s.scope, n)] = (
715
+ visible or s.visible,
716
+ attribute or (bool(s.scope) and s.kind != CLASS),
717
+ s.kind,
718
+ )
719
+ for f in functions:
720
+ key = (f.scope, f.simple_name)
721
+ # Whether Python can see it is part of the binding (def -> cdef).
722
+ by_key.setdefault(key, []).append(f"function {f.name} {f.python}")
723
+ visible, attribute, kind = meta.get(key, (False, False, "function"))
724
+ meta[key] = (visible or f.python, attribute, kind)
725
+ return by_key, meta
726
+
727
+ ba, ma = bindings(sa, a.functions)
728
+ bb, mb = bindings(sb, b.functions)
729
+ changed = {k for k in ba.keys() | bb.keys() if ba.get(k) != bb.get(k)}
730
+ for key in sorted(changed):
731
+ scope, name = key
732
+ visible_a, attribute_a, kind_a = ma.get(key, (False, False, ""))
733
+ visible_b, attribute_b, kind_b = mb.get(key, (False, False, ""))
734
+ label = f"{scope}.{name}" if scope else name
735
+ if _is_dunder(name):
736
+ return f"changed outside its functions: special name {label} bound differently"
737
+ if key in ba and key in bb and CLASS in (kind_a, kind_b):
738
+ return f"changed outside its functions: the class statement of {label} changed"
739
+ change = "added" if key not in ba else "deleted" if key not in bb else "changed"
740
+ visible = visible_a or visible_b
741
+ if change == "deleted" and visible and not scope and path.endswith(".pyx"):
742
+ return (
743
+ f"{label} deleted: Python code importing it from the compiled module fails "
744
+ "at import, and imports of compiled modules' names are not indexed"
745
+ )
746
+ out.append(CythonName(path, scope, name, change, visible, attribute_a or attribute_b))
747
+ # The other statements keep their order (a value read at import can
748
+ # depend on it); imports are exempt, their order binds nothing.
749
+ order_a = [s.hash for s in sa if s.kind not in (CODE, IMPORT) and not changed & _keys(s)]
750
+ order_b = [s.hash for s in sb if s.kind not in (CODE, IMPORT) and not changed & _keys(s)]
751
+ if order_a != order_b:
752
+ return "changed outside its functions: statements reordered"
753
+ return None
754
+
755
+
756
+ def _keys(statement: CythonStatement) -> set[tuple[str, str]]:
757
+ return {(statement.scope, n) for n in statement.names}