pragmagraph 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,57 @@
1
+ """Standalone observed-fact graph substrate for code and document structure."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pragmagraph.adapters import index_path
6
+ from pragmagraph.contracts import CAPABILITIES, INDEXER_VERSION, SCHEMA_VERSION
7
+ from pragmagraph.models import (
8
+ GraphEdge,
9
+ GraphNode,
10
+ GraphSnapshot,
11
+ HealthSummary,
12
+ OmittedDiagnostic,
13
+ PathResult,
14
+ PragmaGraphError,
15
+ QueryHit,
16
+ QueryRequest,
17
+ QueryResult,
18
+ SourceRef,
19
+ )
20
+ from pragmagraph.storage import load_snapshot, save_snapshot, stable_dumps
21
+
22
+ __version__ = "0.0.1"
23
+
24
+ PACKAGE_STATUS = "semantic-alpha"
25
+ STABLE_IMPORT_ROOTS = (
26
+ "pragmagraph",
27
+ "pragmagraph.contracts",
28
+ "pragmagraph.models",
29
+ "pragmagraph.query",
30
+ "pragmagraph.storage",
31
+ "pragmagraph.adapters",
32
+ "pragmagraph.portability",
33
+ )
34
+
35
+ __all__ = [
36
+ "CAPABILITIES",
37
+ "GraphEdge",
38
+ "GraphNode",
39
+ "GraphSnapshot",
40
+ "HealthSummary",
41
+ "INDEXER_VERSION",
42
+ "OmittedDiagnostic",
43
+ "PACKAGE_STATUS",
44
+ "PathResult",
45
+ "PragmaGraphError",
46
+ "QueryHit",
47
+ "QueryRequest",
48
+ "QueryResult",
49
+ "SCHEMA_VERSION",
50
+ "STABLE_IMPORT_ROOTS",
51
+ "SourceRef",
52
+ "__version__",
53
+ "index_path",
54
+ "load_snapshot",
55
+ "save_snapshot",
56
+ "stable_dumps",
57
+ ]
@@ -0,0 +1,123 @@
1
+ """CLI entrypoint for the reusable ``pragmagraph`` package."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+
8
+ from pragmagraph import PACKAGE_STATUS, STABLE_IMPORT_ROOTS, __version__
9
+ from pragmagraph.adapters import index_path
10
+ from pragmagraph.models import QueryRequest
11
+ from pragmagraph.query import health, neighborhood, path, query
12
+ from pragmagraph.storage import load_snapshot, save_snapshot
13
+
14
+
15
+ def smoke_payload() -> dict[str, object]:
16
+ """Return a deterministic package smoke payload."""
17
+ return {
18
+ "package": "pragmagraph",
19
+ "version": __version__,
20
+ "status": PACKAGE_STATUS,
21
+ "stable_import_roots": list(STABLE_IMPORT_ROOTS),
22
+ "semantic_contract": True,
23
+ "openminion_imports": False,
24
+ }
25
+
26
+
27
+ def _print_payload(payload: object, *, as_json: bool) -> None:
28
+ if as_json:
29
+ if hasattr(payload, "to_dict"):
30
+ payload = payload.to_dict()
31
+ print(json.dumps(payload, sort_keys=True))
32
+ return
33
+ print(payload)
34
+
35
+
36
+ def main(argv: list[str] | None = None) -> int:
37
+ parser = argparse.ArgumentParser(description="pragmagraph package smoke")
38
+ parser.add_argument("--json", action="store_true", help="emit JSON output")
39
+ subparsers = parser.add_subparsers(dest="command")
40
+
41
+ index_parser = subparsers.add_parser("index", help="index a local root")
42
+ index_parser.add_argument("root")
43
+ index_parser.add_argument("--out", required=True)
44
+ index_parser.add_argument("--namespace", default="default")
45
+ index_parser.add_argument("--json", action="store_true", help="emit JSON output")
46
+
47
+ query_parser = subparsers.add_parser("query", help="query a snapshot")
48
+ query_parser.add_argument("snapshot")
49
+ query_parser.add_argument("query")
50
+ query_parser.add_argument("--max-results", type=int, default=10)
51
+ query_parser.add_argument("--json", action="store_true", help="emit JSON output")
52
+
53
+ neighborhood_parser = subparsers.add_parser(
54
+ "neighborhood", help="show nodes around a snapshot node"
55
+ )
56
+ neighborhood_parser.add_argument("snapshot")
57
+ neighborhood_parser.add_argument("node_id")
58
+ neighborhood_parser.add_argument("--depth", type=int, default=1)
59
+ neighborhood_parser.add_argument("--max-results", type=int, default=10)
60
+ neighborhood_parser.add_argument(
61
+ "--json", action="store_true", help="emit JSON output"
62
+ )
63
+
64
+ path_parser = subparsers.add_parser("path", help="find a bounded graph path")
65
+ path_parser.add_argument("snapshot")
66
+ path_parser.add_argument("source_id")
67
+ path_parser.add_argument("target_id")
68
+ path_parser.add_argument("--max-hops", type=int, default=4)
69
+ path_parser.add_argument("--json", action="store_true", help="emit JSON output")
70
+
71
+ health_parser = subparsers.add_parser("health", help="summarize a snapshot")
72
+ health_parser.add_argument("snapshot")
73
+ health_parser.add_argument("--json", action="store_true", help="emit JSON output")
74
+
75
+ args = parser.parse_args(argv)
76
+
77
+ if args.command == "index":
78
+ snapshot = index_path(args.root, namespace=args.namespace)
79
+ save_snapshot(snapshot, args.out)
80
+ _print_payload(health(snapshot), as_json=args.json)
81
+ elif args.command == "query":
82
+ snapshot = load_snapshot(args.snapshot)
83
+ _print_payload(
84
+ query(
85
+ snapshot,
86
+ QueryRequest(query=args.query, max_results=args.max_results),
87
+ ).to_dict(),
88
+ as_json=True,
89
+ )
90
+ elif args.command == "neighborhood":
91
+ snapshot = load_snapshot(args.snapshot)
92
+ _print_payload(
93
+ neighborhood(
94
+ snapshot,
95
+ args.node_id,
96
+ depth=args.depth,
97
+ max_results=args.max_results,
98
+ ).to_dict(),
99
+ as_json=True,
100
+ )
101
+ elif args.command == "path":
102
+ snapshot = load_snapshot(args.snapshot)
103
+ _print_payload(
104
+ path(
105
+ snapshot,
106
+ args.source_id,
107
+ args.target_id,
108
+ max_hops=args.max_hops,
109
+ ).to_dict(),
110
+ as_json=True,
111
+ )
112
+ elif args.command == "health":
113
+ snapshot = load_snapshot(args.snapshot)
114
+ _print_payload(health(snapshot), as_json=True)
115
+ elif args.json:
116
+ print(json.dumps(smoke_payload(), sort_keys=True))
117
+ else:
118
+ print(f"pragmagraph semantic alpha OK: {smoke_payload()}")
119
+ return 0
120
+
121
+
122
+ if __name__ == "__main__":
123
+ raise SystemExit(main())
@@ -0,0 +1,378 @@
1
+ """Local filesystem indexer adapters for PragmaGraph."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ast
6
+ import hashlib
7
+ import re
8
+ from pathlib import Path
9
+ from typing import Iterable
10
+
11
+ from pragmagraph.contracts import (
12
+ EDGE_CONTAINS,
13
+ EDGE_DEFINES,
14
+ EDGE_IMPORTS,
15
+ EDGE_REFERENCES_SECTION,
16
+ INDEXER_VERSION,
17
+ NODE_DIRECTORY,
18
+ NODE_DOC_SECTION,
19
+ NODE_FILE,
20
+ NODE_PROJECT,
21
+ NODE_PYTHON_SYMBOL,
22
+ SCHEMA_VERSION,
23
+ )
24
+ from pragmagraph.models import (
25
+ GraphEdge,
26
+ GraphNode,
27
+ GraphSnapshot,
28
+ OmittedDiagnostic,
29
+ SourceRef,
30
+ )
31
+ from pragmagraph.portability import edge_id, node_id, normalize_relative_path
32
+
33
+ DEFAULT_IGNORES = frozenset(
34
+ {
35
+ ".git",
36
+ ".mypy_cache",
37
+ ".pytest_cache",
38
+ ".ruff_cache",
39
+ ".venv",
40
+ "__pycache__",
41
+ "build",
42
+ "dist",
43
+ "node_modules",
44
+ }
45
+ )
46
+
47
+ TEXT_SUFFIXES = frozenset({".md", ".py", ".txt", ".rst"})
48
+
49
+
50
+ def _read_text(path: Path) -> str:
51
+ try:
52
+ return path.read_text(encoding="utf-8")
53
+ except UnicodeDecodeError:
54
+ return path.read_text(encoding="utf-8", errors="replace")
55
+
56
+
57
+ def _content_hash(path: Path) -> str:
58
+ return hashlib.sha256(path.read_bytes()).hexdigest()
59
+
60
+
61
+ def _rel(path: Path, root: Path) -> str:
62
+ return normalize_relative_path(path.relative_to(root))
63
+
64
+
65
+ def _snippet(text: str, *, limit: int = 360) -> str:
66
+ collapsed = " ".join(text.strip().split())
67
+ return collapsed[:limit]
68
+
69
+
70
+ def _markdown_slug(text: str) -> str:
71
+ slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
72
+ return slug or "section"
73
+
74
+
75
+ def _add_node(nodes: dict[str, GraphNode], node: GraphNode) -> None:
76
+ nodes.setdefault(node.id, node)
77
+
78
+
79
+ def _add_edge(edges: dict[str, GraphEdge], edge: GraphEdge) -> None:
80
+ edges.setdefault(edge.id, edge)
81
+
82
+
83
+ def _iter_paths(root: Path, ignore_names: frozenset[str]) -> Iterable[Path]:
84
+ for path in sorted(root.rglob("*")):
85
+ if any(part in ignore_names for part in path.relative_to(root).parts):
86
+ continue
87
+ yield path
88
+
89
+
90
+ def index_path(
91
+ root_path: str | Path,
92
+ *,
93
+ namespace: str = "default",
94
+ ignore_names: frozenset[str] = DEFAULT_IGNORES,
95
+ created_at: str = "",
96
+ ) -> GraphSnapshot:
97
+ """Index a local code/docs root into a deterministic snapshot."""
98
+ root = Path(root_path).resolve()
99
+ nodes: dict[str, GraphNode] = {}
100
+ edges: dict[str, GraphEdge] = {}
101
+ omitted: list[OmittedDiagnostic] = []
102
+
103
+ project_id = node_id(namespace, NODE_PROJECT, ".")
104
+ _add_node(
105
+ nodes,
106
+ GraphNode(
107
+ id=project_id,
108
+ kind=NODE_PROJECT,
109
+ label=root.name or namespace,
110
+ source_ref=SourceRef(path="."),
111
+ metadata={"namespace": namespace},
112
+ ),
113
+ )
114
+
115
+ parent_by_path: dict[str, str] = {"": project_id}
116
+ for path in _iter_paths(root, ignore_names):
117
+ rel = _rel(path, root)
118
+ parent_rel = normalize_relative_path(Path(rel).parent)
119
+ parent_id = parent_by_path.get(parent_rel, project_id)
120
+ if path.is_dir():
121
+ current_id = node_id(namespace, NODE_DIRECTORY, rel)
122
+ parent_by_path[rel] = current_id
123
+ _add_node(
124
+ nodes,
125
+ GraphNode(
126
+ id=current_id,
127
+ kind=NODE_DIRECTORY,
128
+ label=path.name,
129
+ source_ref=SourceRef(path=rel),
130
+ ),
131
+ )
132
+ _add_edge(
133
+ edges,
134
+ GraphEdge(
135
+ id=edge_id(namespace, parent_id, EDGE_CONTAINS, current_id),
136
+ kind=EDGE_CONTAINS,
137
+ source_id=parent_id,
138
+ target_id=current_id,
139
+ source_ref=SourceRef(path=rel),
140
+ ),
141
+ )
142
+ continue
143
+ if not path.is_file():
144
+ continue
145
+ file_id = node_id(namespace, NODE_FILE, rel)
146
+ text = _read_text(path) if path.suffix.lower() in TEXT_SUFFIXES else ""
147
+ _add_node(
148
+ nodes,
149
+ GraphNode(
150
+ id=file_id,
151
+ kind=NODE_FILE,
152
+ label=path.name,
153
+ source_ref=SourceRef(path=rel),
154
+ text=_snippet(text),
155
+ metadata={
156
+ "content_hash": _content_hash(path),
157
+ "suffix": path.suffix.lower(),
158
+ },
159
+ ),
160
+ )
161
+ _add_edge(
162
+ edges,
163
+ GraphEdge(
164
+ id=edge_id(namespace, parent_id, EDGE_CONTAINS, file_id),
165
+ kind=EDGE_CONTAINS,
166
+ source_id=parent_id,
167
+ target_id=file_id,
168
+ source_ref=SourceRef(path=rel),
169
+ ),
170
+ )
171
+ if path.suffix.lower() == ".md":
172
+ _index_markdown(
173
+ namespace=namespace,
174
+ rel=rel,
175
+ file_id=file_id,
176
+ text=text,
177
+ nodes=nodes,
178
+ edges=edges,
179
+ )
180
+ elif path.suffix.lower() == ".py":
181
+ _index_python(
182
+ namespace=namespace,
183
+ rel=rel,
184
+ file_id=file_id,
185
+ text=text,
186
+ nodes=nodes,
187
+ edges=edges,
188
+ omitted=omitted,
189
+ )
190
+
191
+ stats = {
192
+ "edge_count": len(edges),
193
+ "node_count": len(nodes),
194
+ "omitted_count": len(omitted),
195
+ "root_exists": root.exists(),
196
+ }
197
+ return GraphSnapshot(
198
+ namespace=namespace,
199
+ root_path=str(root),
200
+ nodes=tuple(sorted(nodes.values(), key=lambda node: node.id)),
201
+ edges=tuple(sorted(edges.values(), key=lambda edge: edge.id)),
202
+ omitted=tuple(omitted),
203
+ stats=stats,
204
+ schema_version=SCHEMA_VERSION,
205
+ indexer_version=INDEXER_VERSION,
206
+ created_at=created_at,
207
+ )
208
+
209
+
210
+ def _index_markdown(
211
+ *,
212
+ namespace: str,
213
+ rel: str,
214
+ file_id: str,
215
+ text: str,
216
+ nodes: dict[str, GraphNode],
217
+ edges: dict[str, GraphEdge],
218
+ ) -> None:
219
+ previous_section_id = ""
220
+ for number, line in enumerate(text.splitlines(), start=1):
221
+ match = re.match(r"^(#{1,6})\s+(.+?)\s*$", line)
222
+ if not match:
223
+ continue
224
+ heading = match.group(2).strip()
225
+ slug = _markdown_slug(heading)
226
+ section_id = node_id(namespace, NODE_DOC_SECTION, f"{rel}#{slug}")
227
+ source_ref = SourceRef(path=rel, line=number, section=heading)
228
+ _add_node(
229
+ nodes,
230
+ GraphNode(
231
+ id=section_id,
232
+ kind=NODE_DOC_SECTION,
233
+ label=heading,
234
+ source_ref=source_ref,
235
+ text=heading,
236
+ metadata={"level": len(match.group(1)), "slug": slug},
237
+ ),
238
+ )
239
+ _add_edge(
240
+ edges,
241
+ GraphEdge(
242
+ id=edge_id(namespace, file_id, EDGE_DEFINES, section_id),
243
+ kind=EDGE_DEFINES,
244
+ source_id=file_id,
245
+ target_id=section_id,
246
+ source_ref=source_ref,
247
+ ),
248
+ )
249
+ if previous_section_id:
250
+ _add_edge(
251
+ edges,
252
+ GraphEdge(
253
+ id=edge_id(
254
+ namespace,
255
+ previous_section_id,
256
+ EDGE_REFERENCES_SECTION,
257
+ section_id,
258
+ ),
259
+ kind=EDGE_REFERENCES_SECTION,
260
+ source_id=previous_section_id,
261
+ target_id=section_id,
262
+ source_ref=source_ref,
263
+ ),
264
+ )
265
+ previous_section_id = section_id
266
+
267
+
268
+ def _index_python(
269
+ *,
270
+ namespace: str,
271
+ rel: str,
272
+ file_id: str,
273
+ text: str,
274
+ nodes: dict[str, GraphNode],
275
+ edges: dict[str, GraphEdge],
276
+ omitted: list[OmittedDiagnostic],
277
+ ) -> None:
278
+ try:
279
+ tree = ast.parse(text)
280
+ except SyntaxError as exc:
281
+ omitted.append(
282
+ OmittedDiagnostic(
283
+ reason="python_syntax_error",
284
+ item_id=rel,
285
+ details={"line": exc.lineno, "message": exc.msg},
286
+ )
287
+ )
288
+ return
289
+
290
+ for item in ast.walk(tree):
291
+ if isinstance(item, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)):
292
+ symbol_key = f"{rel}:{item.name}"
293
+ symbol_id = node_id(namespace, NODE_PYTHON_SYMBOL, symbol_key)
294
+ source_ref = SourceRef(path=rel, line=getattr(item, "lineno", None))
295
+ _add_node(
296
+ nodes,
297
+ GraphNode(
298
+ id=symbol_id,
299
+ kind=NODE_PYTHON_SYMBOL,
300
+ label=item.name,
301
+ source_ref=source_ref,
302
+ text=item.name,
303
+ metadata={"symbol_type": type(item).__name__},
304
+ ),
305
+ )
306
+ _add_edge(
307
+ edges,
308
+ GraphEdge(
309
+ id=edge_id(namespace, file_id, EDGE_DEFINES, symbol_id),
310
+ kind=EDGE_DEFINES,
311
+ source_id=file_id,
312
+ target_id=symbol_id,
313
+ source_ref=source_ref,
314
+ ),
315
+ )
316
+ elif isinstance(item, ast.Import):
317
+ for alias in item.names:
318
+ _add_import_edge(
319
+ namespace=namespace,
320
+ rel=rel,
321
+ file_id=file_id,
322
+ module=alias.name,
323
+ line=getattr(item, "lineno", None),
324
+ nodes=nodes,
325
+ edges=edges,
326
+ )
327
+ elif isinstance(item, ast.ImportFrom) and item.module:
328
+ _add_import_edge(
329
+ namespace=namespace,
330
+ rel=rel,
331
+ file_id=file_id,
332
+ module=item.module,
333
+ line=getattr(item, "lineno", None),
334
+ nodes=nodes,
335
+ edges=edges,
336
+ )
337
+
338
+
339
+ def _add_import_edge(
340
+ *,
341
+ namespace: str,
342
+ rel: str,
343
+ file_id: str,
344
+ module: str,
345
+ line: int | None,
346
+ nodes: dict[str, GraphNode],
347
+ edges: dict[str, GraphEdge],
348
+ ) -> None:
349
+ import_id = node_id(namespace, NODE_PYTHON_SYMBOL, f"import:{module}")
350
+ source_ref = SourceRef(path=rel, line=line)
351
+ _add_node(
352
+ nodes,
353
+ GraphNode(
354
+ id=import_id,
355
+ kind=NODE_PYTHON_SYMBOL,
356
+ label=module,
357
+ source_ref=SourceRef(path=rel, line=line),
358
+ text=module,
359
+ metadata={"external": True, "symbol_type": "import"},
360
+ ),
361
+ )
362
+ _add_edge(
363
+ edges,
364
+ GraphEdge(
365
+ id=edge_id(namespace, file_id, EDGE_IMPORTS, import_id),
366
+ kind=EDGE_IMPORTS,
367
+ source_id=file_id,
368
+ target_id=import_id,
369
+ source_ref=source_ref,
370
+ ),
371
+ )
372
+
373
+
374
+ __all__ = [
375
+ "DEFAULT_IGNORES",
376
+ "TEXT_SUFFIXES",
377
+ "index_path",
378
+ ]
@@ -0,0 +1,83 @@
1
+ """Public constants for the PragmaGraph semantic alpha contract."""
2
+
3
+ from __future__ import annotations
4
+
5
+ SCHEMA_VERSION = "pragmagraph.snapshot.v1alpha1"
6
+ INDEXER_VERSION = "pragmagraph.indexer.v1alpha1"
7
+
8
+ CAPABILITY_QUERY = "query"
9
+ CAPABILITY_NEIGHBORHOOD = "neighborhood"
10
+ CAPABILITY_PATH = "path"
11
+ CAPABILITY_HEALTH = "health"
12
+ CAPABILITY_REFRESH = "refresh"
13
+ CAPABILITY_CITATIONS = "citations"
14
+ CAPABILITY_PROVENANCE = "provenance"
15
+
16
+ CAPABILITIES = frozenset(
17
+ {
18
+ CAPABILITY_QUERY,
19
+ CAPABILITY_NEIGHBORHOOD,
20
+ CAPABILITY_PATH,
21
+ CAPABILITY_HEALTH,
22
+ CAPABILITY_REFRESH,
23
+ CAPABILITY_CITATIONS,
24
+ CAPABILITY_PROVENANCE,
25
+ }
26
+ )
27
+
28
+ NODE_PROJECT = "project"
29
+ NODE_DIRECTORY = "directory"
30
+ NODE_FILE = "file"
31
+ NODE_DOC_SECTION = "doc_section"
32
+ NODE_PYTHON_SYMBOL = "python_symbol"
33
+
34
+ NODE_KINDS = frozenset(
35
+ {
36
+ NODE_PROJECT,
37
+ NODE_DIRECTORY,
38
+ NODE_FILE,
39
+ NODE_DOC_SECTION,
40
+ NODE_PYTHON_SYMBOL,
41
+ }
42
+ )
43
+
44
+ EDGE_CONTAINS = "contains"
45
+ EDGE_DEFINES = "defines"
46
+ EDGE_IMPORTS = "imports"
47
+ EDGE_MENTIONS = "mentions"
48
+ EDGE_REFERENCES_SECTION = "references_section"
49
+
50
+ EDGE_KINDS = frozenset(
51
+ {
52
+ EDGE_CONTAINS,
53
+ EDGE_DEFINES,
54
+ EDGE_IMPORTS,
55
+ EDGE_MENTIONS,
56
+ EDGE_REFERENCES_SECTION,
57
+ }
58
+ )
59
+
60
+ __all__ = [
61
+ "CAPABILITIES",
62
+ "CAPABILITY_CITATIONS",
63
+ "CAPABILITY_HEALTH",
64
+ "CAPABILITY_NEIGHBORHOOD",
65
+ "CAPABILITY_PATH",
66
+ "CAPABILITY_PROVENANCE",
67
+ "CAPABILITY_QUERY",
68
+ "CAPABILITY_REFRESH",
69
+ "EDGE_CONTAINS",
70
+ "EDGE_DEFINES",
71
+ "EDGE_IMPORTS",
72
+ "EDGE_KINDS",
73
+ "EDGE_MENTIONS",
74
+ "EDGE_REFERENCES_SECTION",
75
+ "INDEXER_VERSION",
76
+ "NODE_DIRECTORY",
77
+ "NODE_DOC_SECTION",
78
+ "NODE_FILE",
79
+ "NODE_KINDS",
80
+ "NODE_PROJECT",
81
+ "NODE_PYTHON_SYMBOL",
82
+ "SCHEMA_VERSION",
83
+ ]