graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""Zig extractor (tree-sitter). Moved verbatim from graphify/extract.py."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from graphify.extractors.base import _file_stem, _make_id, _read_text
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def extract_zig(path: Path) -> dict:
|
|
11
|
+
"""Extract functions, structs, enums, unions, and imports from a .zig file."""
|
|
12
|
+
try:
|
|
13
|
+
import tree_sitter_zig as tszig
|
|
14
|
+
from tree_sitter import Language, Parser
|
|
15
|
+
except ImportError:
|
|
16
|
+
return {"nodes": [], "edges": [], "error": "tree_sitter_zig not installed"}
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
language = Language(tszig.language())
|
|
20
|
+
parser = Parser(language)
|
|
21
|
+
source = path.read_bytes()
|
|
22
|
+
tree = parser.parse(source)
|
|
23
|
+
root = tree.root_node
|
|
24
|
+
except Exception as e:
|
|
25
|
+
return {"nodes": [], "edges": [], "error": str(e)}
|
|
26
|
+
|
|
27
|
+
stem = _file_stem(path)
|
|
28
|
+
str_path = str(path)
|
|
29
|
+
nodes: list[dict] = []
|
|
30
|
+
edges: list[dict] = []
|
|
31
|
+
seen_ids: set[str] = set()
|
|
32
|
+
function_bodies: list[tuple[str, Any]] = []
|
|
33
|
+
|
|
34
|
+
def add_node(nid: str, label: str, line: int) -> None:
|
|
35
|
+
if nid not in seen_ids:
|
|
36
|
+
seen_ids.add(nid)
|
|
37
|
+
nodes.append({"id": nid, "label": label, "file_type": "code",
|
|
38
|
+
"source_file": str_path, "source_location": f"L{line}"})
|
|
39
|
+
|
|
40
|
+
def add_edge(src: str, tgt: str, relation: str, line: int,
|
|
41
|
+
confidence: str = "EXTRACTED", weight: float = 1.0,
|
|
42
|
+
context: str | None = None) -> None:
|
|
43
|
+
edge = {"source": src, "target": tgt, "relation": relation,
|
|
44
|
+
"confidence": confidence, "source_file": str_path,
|
|
45
|
+
"source_location": f"L{line}", "weight": weight}
|
|
46
|
+
if context:
|
|
47
|
+
edge["context"] = context
|
|
48
|
+
edges.append(edge)
|
|
49
|
+
|
|
50
|
+
file_nid = _make_id(str(path))
|
|
51
|
+
add_node(file_nid, path.name, 1)
|
|
52
|
+
|
|
53
|
+
def _extract_import(node) -> None:
|
|
54
|
+
for child in node.children:
|
|
55
|
+
if child.type == "builtin_function":
|
|
56
|
+
bi = None
|
|
57
|
+
args = None
|
|
58
|
+
for c in child.children:
|
|
59
|
+
if c.type == "builtin_identifier":
|
|
60
|
+
bi = _read_text(c, source)
|
|
61
|
+
elif c.type == "arguments":
|
|
62
|
+
args = c
|
|
63
|
+
if bi in ("@import", "@cImport") and args:
|
|
64
|
+
for arg in args.children:
|
|
65
|
+
if arg.type in ("string_literal", "string"):
|
|
66
|
+
raw = _read_text(arg, source).strip('"')
|
|
67
|
+
module_name = raw.split("/")[-1].split(".")[0]
|
|
68
|
+
if module_name:
|
|
69
|
+
tgt_nid = _make_id(module_name)
|
|
70
|
+
add_edge(file_nid, tgt_nid, "imports_from",
|
|
71
|
+
node.start_point[0] + 1)
|
|
72
|
+
return
|
|
73
|
+
elif child.type == "field_expression":
|
|
74
|
+
_extract_import(child)
|
|
75
|
+
return
|
|
76
|
+
|
|
77
|
+
def walk(node, parent_struct_nid: str | None = None) -> None:
|
|
78
|
+
t = node.type
|
|
79
|
+
|
|
80
|
+
if t == "function_declaration":
|
|
81
|
+
name_node = node.child_by_field_name("name")
|
|
82
|
+
if name_node:
|
|
83
|
+
func_name = _read_text(name_node, source)
|
|
84
|
+
line = node.start_point[0] + 1
|
|
85
|
+
if parent_struct_nid:
|
|
86
|
+
func_nid = _make_id(parent_struct_nid, func_name)
|
|
87
|
+
add_node(func_nid, f".{func_name}()", line)
|
|
88
|
+
add_edge(parent_struct_nid, func_nid, "method", line)
|
|
89
|
+
else:
|
|
90
|
+
func_nid = _make_id(stem, func_name)
|
|
91
|
+
add_node(func_nid, f"{func_name}()", line)
|
|
92
|
+
add_edge(file_nid, func_nid, "contains", line)
|
|
93
|
+
body = node.child_by_field_name("body")
|
|
94
|
+
if body:
|
|
95
|
+
function_bodies.append((func_nid, body))
|
|
96
|
+
return
|
|
97
|
+
|
|
98
|
+
if t == "variable_declaration":
|
|
99
|
+
name_node = None
|
|
100
|
+
value_node = None
|
|
101
|
+
for child in node.children:
|
|
102
|
+
if child.type == "identifier":
|
|
103
|
+
name_node = child
|
|
104
|
+
elif child.type in ("struct_declaration", "enum_declaration",
|
|
105
|
+
"union_declaration", "builtin_function",
|
|
106
|
+
"field_expression"):
|
|
107
|
+
value_node = child
|
|
108
|
+
|
|
109
|
+
if value_node and value_node.type == "struct_declaration":
|
|
110
|
+
if name_node:
|
|
111
|
+
struct_name = _read_text(name_node, source)
|
|
112
|
+
line = node.start_point[0] + 1
|
|
113
|
+
struct_nid = _make_id(stem, struct_name)
|
|
114
|
+
add_node(struct_nid, struct_name, line)
|
|
115
|
+
add_edge(file_nid, struct_nid, "contains", line)
|
|
116
|
+
for child in value_node.children:
|
|
117
|
+
walk(child, parent_struct_nid=struct_nid)
|
|
118
|
+
return
|
|
119
|
+
|
|
120
|
+
if value_node and value_node.type in ("enum_declaration", "union_declaration"):
|
|
121
|
+
if name_node:
|
|
122
|
+
type_name = _read_text(name_node, source)
|
|
123
|
+
line = node.start_point[0] + 1
|
|
124
|
+
type_nid = _make_id(stem, type_name)
|
|
125
|
+
add_node(type_nid, type_name, line)
|
|
126
|
+
add_edge(file_nid, type_nid, "contains", line)
|
|
127
|
+
# Zig enums and tagged unions can declare methods just like
|
|
128
|
+
# structs (`pub fn ...` inside the container). Recurse so those
|
|
129
|
+
# methods — and the calls made from their bodies — are captured
|
|
130
|
+
# rather than dropped along with the whole method layer.
|
|
131
|
+
for child in value_node.children:
|
|
132
|
+
walk(child, parent_struct_nid=type_nid)
|
|
133
|
+
return
|
|
134
|
+
|
|
135
|
+
if value_node and value_node.type in ("builtin_function", "field_expression"):
|
|
136
|
+
_extract_import(node)
|
|
137
|
+
return
|
|
138
|
+
|
|
139
|
+
for child in node.children:
|
|
140
|
+
walk(child, parent_struct_nid)
|
|
141
|
+
|
|
142
|
+
walk(root)
|
|
143
|
+
|
|
144
|
+
seen_call_pairs: set[tuple[str, str]] = set()
|
|
145
|
+
raw_calls: list[dict] = []
|
|
146
|
+
|
|
147
|
+
def walk_calls(node, caller_nid: str) -> None:
|
|
148
|
+
if node.type == "function_declaration":
|
|
149
|
+
return
|
|
150
|
+
if node.type == "call_expression":
|
|
151
|
+
fn = node.child_by_field_name("function")
|
|
152
|
+
if fn:
|
|
153
|
+
fn_text = _read_text(fn, source)
|
|
154
|
+
callee = fn_text.split(".")[-1]
|
|
155
|
+
is_member_call = "." in fn_text
|
|
156
|
+
tgt_nid = next((n["id"] for n in nodes if n["label"] in
|
|
157
|
+
(f"{callee}()", f".{callee}()")), None)
|
|
158
|
+
if tgt_nid and tgt_nid != caller_nid:
|
|
159
|
+
pair = (caller_nid, tgt_nid)
|
|
160
|
+
if pair not in seen_call_pairs:
|
|
161
|
+
seen_call_pairs.add(pair)
|
|
162
|
+
add_edge(caller_nid, tgt_nid, "calls",
|
|
163
|
+
node.start_point[0] + 1,
|
|
164
|
+
confidence="EXTRACTED", weight=1.0)
|
|
165
|
+
elif callee:
|
|
166
|
+
raw_calls.append({
|
|
167
|
+
"caller_nid": caller_nid,
|
|
168
|
+
"callee": callee,
|
|
169
|
+
"is_member_call": is_member_call,
|
|
170
|
+
"source_file": str_path,
|
|
171
|
+
"source_location": f"L{node.start_point[0] + 1}",
|
|
172
|
+
})
|
|
173
|
+
for child in node.children:
|
|
174
|
+
walk_calls(child, caller_nid)
|
|
175
|
+
|
|
176
|
+
for caller_nid, body_node in function_bodies:
|
|
177
|
+
walk_calls(body_node, caller_nid)
|
|
178
|
+
|
|
179
|
+
clean_edges = [e for e in edges if e["source"] in seen_ids and
|
|
180
|
+
(e["target"] in seen_ids or e["relation"] == "imports_from")]
|
|
181
|
+
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
|
graphify/file_slice.py
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""Intra-file slicing for oversized text documents (#1369).
|
|
2
|
+
|
|
3
|
+
The extraction packer (`_pack_chunks_by_tokens`) treats each file as atomic and
|
|
4
|
+
`_read_files` caps every file at ``_FILE_CHAR_CAP`` characters, so a document
|
|
5
|
+
larger than that cap had everything past the cap silently dropped — the model
|
|
6
|
+
never saw it, and nothing in the adaptive-retry path could recover it ("a single
|
|
7
|
+
file larger than the budget ... packing can't shrink one big file").
|
|
8
|
+
|
|
9
|
+
This module splits an oversized *splittable text* document (Markdown, plain
|
|
10
|
+
text, reStructuredText) into contiguous ``FileSlice`` units at heading /
|
|
11
|
+
paragraph / line boundaries so the whole file gets extracted across several
|
|
12
|
+
units. Every slice of a file reports the **parent file path** as its source, so
|
|
13
|
+
the resulting nodes are never fragmented per-slice — they merge by source_file
|
|
14
|
+
exactly as if the file had been extracted in one pass.
|
|
15
|
+
|
|
16
|
+
Only plain-text documents are sliced: code files need whole-symbol context, and
|
|
17
|
+
PDFs/images are read through their own extractors and have no char-offset model.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
# Plain-text document types where boundary-based slicing is meaningful and where
|
|
26
|
+
# `_file_to_text` is a straight ``read_text`` (so a char range matches the bytes
|
|
27
|
+
# the model is shown). Deliberately excludes code (.py, .ts, ...) and binary
|
|
28
|
+
# docs (.pdf) — those are never sliced.
|
|
29
|
+
#
|
|
30
|
+
# This set has to keep pace with ``detect.DOC_EXTENSIONS``: anything classified
|
|
31
|
+
# as a document reaches the semantic pass, and anything the pass sees that is
|
|
32
|
+
# NOT listed here is silently cut at ``_FILE_CHAR_CAP`` by ``_read_files``. The
|
|
33
|
+
# two lists drifted as DOC_EXTENSIONS grew — .qmd, .skill, .html, .yaml and .yml
|
|
34
|
+
# were documents that never got sliced, so a 38k-character one reached the model
|
|
35
|
+
# as its first 20k with no warning and no partial marker (#2900).
|
|
36
|
+
# ``tests/test_oversized_document_slicing.py`` pins the relationship so a future
|
|
37
|
+
# addition to DOC_EXTENSIONS fails loudly instead of quietly losing content.
|
|
38
|
+
_SPLITTABLE_TEXT_SUFFIXES = frozenset({
|
|
39
|
+
".md", ".mdx", ".markdown", ".txt", ".rst",
|
|
40
|
+
".qmd", ".skill", ".html", ".yaml", ".yml",
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
# Document types whose BYTES are not what the model is shown. `llm._file_to_text`
|
|
44
|
+
# routes these through a converter, so a character range has to be taken over the
|
|
45
|
+
# converted text, never over the file. Kept separate from the set above because
|
|
46
|
+
# they are splittable for a different reason and via a different reader.
|
|
47
|
+
_CONVERTED_TEXT_SUFFIXES = frozenset({".pdf"})
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _pdf_text(path: Path) -> str:
|
|
51
|
+
"""Extracted text of a PDF — the same string `llm._file_to_text` builds.
|
|
52
|
+
|
|
53
|
+
Imported lazily from ``detect`` so this module keeps no import-time
|
|
54
|
+
dependency on the extraction stack (``llm`` imports *this* module, so the
|
|
55
|
+
reverse direction would be a cycle).
|
|
56
|
+
"""
|
|
57
|
+
from graphify.detect import extract_pdf_text
|
|
58
|
+
return extract_pdf_text(path)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
# Slicing a PDF means extracting its text, and the slicing pass asks for the same
|
|
62
|
+
# file several times: once to measure it, then once per slice as the prompt is
|
|
63
|
+
# built. Memoised on (path, size, mtime_ns) so a corpus of papers is parsed once
|
|
64
|
+
# rather than once per slice, and so a file rewritten mid-run is re-read instead
|
|
65
|
+
# of served a stale body. Bounded: the entries are whole documents, and a large
|
|
66
|
+
# corpus should not pin all of them in memory.
|
|
67
|
+
_CONVERTED_TEXT_CACHE: "dict[tuple, str]" = {}
|
|
68
|
+
_CONVERTED_TEXT_CACHE_MAX = 64
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def unit_source_text(path: Path) -> str:
|
|
72
|
+
"""The text a unit contributes to the prompt, whatever its container.
|
|
73
|
+
|
|
74
|
+
Plain-text files are read directly; converted types (PDF) go through their
|
|
75
|
+
converter. Both `expand_oversized_files` and `read_slice_text` use this, so
|
|
76
|
+
the offsets a slice carries always index the same string the model sees.
|
|
77
|
+
"""
|
|
78
|
+
if path.suffix.lower() not in _CONVERTED_TEXT_SUFFIXES:
|
|
79
|
+
return path.read_text(encoding="utf-8", errors="replace")
|
|
80
|
+
try:
|
|
81
|
+
st = path.stat()
|
|
82
|
+
key = (str(path), st.st_size, st.st_mtime_ns)
|
|
83
|
+
except OSError:
|
|
84
|
+
return _pdf_text(path)
|
|
85
|
+
hit = _CONVERTED_TEXT_CACHE.get(key)
|
|
86
|
+
if hit is not None:
|
|
87
|
+
return hit
|
|
88
|
+
text = _pdf_text(path)
|
|
89
|
+
if len(_CONVERTED_TEXT_CACHE) >= _CONVERTED_TEXT_CACHE_MAX:
|
|
90
|
+
_CONVERTED_TEXT_CACHE.clear()
|
|
91
|
+
_CONVERTED_TEXT_CACHE[key] = text
|
|
92
|
+
return text
|
|
93
|
+
|
|
94
|
+
# Boundary preferences, strongest first. A Markdown heading (``\n#``) keeps a
|
|
95
|
+
# section with its title; a blank line keeps a paragraph intact; a bare newline
|
|
96
|
+
# avoids cutting mid-line. If none is found in the window we hard-cut.
|
|
97
|
+
_BOUNDARY_SEPARATORS = ("\n#", "\n\n", "\n")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass(frozen=True)
|
|
101
|
+
class FileSlice:
|
|
102
|
+
"""A contiguous ``[start, end)`` character range of a splittable text file.
|
|
103
|
+
|
|
104
|
+
``index``/``total`` are for logging only. ``path`` is the real file on disk;
|
|
105
|
+
the slice always reports ``path`` as its source so slices don't fragment the
|
|
106
|
+
graph.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
path: Path
|
|
110
|
+
start: int
|
|
111
|
+
end: int
|
|
112
|
+
index: int
|
|
113
|
+
total: int
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# A unit of extraction work: either a whole file (``Path``) or one slice of one.
|
|
117
|
+
Unit = "Path | FileSlice"
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def unit_path(unit: "Path | FileSlice") -> Path:
|
|
121
|
+
"""The on-disk path a unit belongs to (the parent file for a slice)."""
|
|
122
|
+
return unit.path if isinstance(unit, FileSlice) else unit
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def is_splittable_text(path: Path) -> bool:
|
|
126
|
+
"""True for document types that may be sliced.
|
|
127
|
+
|
|
128
|
+
Covers plain text read straight off disk and converted types (PDF) whose
|
|
129
|
+
text is produced by a converter. Both are sliceable because
|
|
130
|
+
:func:`unit_source_text` gives the slicing pass the same string the prompt
|
|
131
|
+
will carry; what disqualifies a type is having no text at all (an image) or
|
|
132
|
+
text the reader cannot address by character offset.
|
|
133
|
+
"""
|
|
134
|
+
suffix = path.suffix.lower()
|
|
135
|
+
return suffix in _SPLITTABLE_TEXT_SUFFIXES or suffix in _CONVERTED_TEXT_SUFFIXES
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _best_cut(text: str, start: int, end: int) -> int:
|
|
139
|
+
"""Return a cut index in ``(start, end]`` at the strongest nearby boundary.
|
|
140
|
+
|
|
141
|
+
Searches the window ``text[start:end]`` for the latest heading, then blank
|
|
142
|
+
line, then newline, and returns the index just *after* it (a heading cuts
|
|
143
|
+
just *before* the ``#`` so the heading leads the next slice). Falls back to a
|
|
144
|
+
hard cut at ``end`` when the window has no usable boundary, which still makes
|
|
145
|
+
forward progress because ``end > start``.
|
|
146
|
+
"""
|
|
147
|
+
window = text[start:end]
|
|
148
|
+
for sep in _BOUNDARY_SEPARATORS:
|
|
149
|
+
idx = window.rfind(sep)
|
|
150
|
+
if idx > 0: # a boundary strictly inside the window (non-empty prev slice)
|
|
151
|
+
if sep == "\n#":
|
|
152
|
+
return start + idx + 1 # keep the newline with the previous slice
|
|
153
|
+
return start + idx + len(sep)
|
|
154
|
+
return end
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def slice_boundaries(text: str, max_chars: int) -> list[tuple[int, int]]:
|
|
158
|
+
"""Contiguous ``(start, end)`` ranges covering all of ``text``, each ≤ max_chars.
|
|
159
|
+
|
|
160
|
+
Ranges are gap-free and non-overlapping, so concatenating the slices
|
|
161
|
+
reproduces ``text`` exactly — no content is dropped.
|
|
162
|
+
"""
|
|
163
|
+
n = len(text)
|
|
164
|
+
if n <= max_chars:
|
|
165
|
+
return [(0, n)]
|
|
166
|
+
bounds: list[tuple[int, int]] = []
|
|
167
|
+
pos = 0
|
|
168
|
+
while pos < n:
|
|
169
|
+
hard = min(pos + max_chars, n)
|
|
170
|
+
end = _best_cut(text, pos, hard) if hard < n else n
|
|
171
|
+
if end <= pos: # defensive: never stall
|
|
172
|
+
end = hard
|
|
173
|
+
bounds.append((pos, end))
|
|
174
|
+
pos = end
|
|
175
|
+
return bounds
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def expand_oversized_files(
|
|
179
|
+
files: list[Path], max_chars: int
|
|
180
|
+
) -> list["Path | FileSlice"]:
|
|
181
|
+
"""Replace each oversized splittable-text file with a list of ``FileSlice``s.
|
|
182
|
+
|
|
183
|
+
Files at or below ``max_chars`` (and all non-splittable files) pass through
|
|
184
|
+
unchanged as ``Path``, so behaviour is identical for everything that already
|
|
185
|
+
fit. Unreadable files pass through untouched (the reader handles the error).
|
|
186
|
+
"""
|
|
187
|
+
out: list["Path | FileSlice"] = []
|
|
188
|
+
for f in files:
|
|
189
|
+
if not is_splittable_text(f):
|
|
190
|
+
out.append(f)
|
|
191
|
+
continue
|
|
192
|
+
try:
|
|
193
|
+
# The CONVERTED text for a PDF, so the boundaries below index the
|
|
194
|
+
# same string read_slice_text will later slice and the prompt will
|
|
195
|
+
# carry — not the container's bytes (#2906).
|
|
196
|
+
text = unit_source_text(f)
|
|
197
|
+
except OSError:
|
|
198
|
+
out.append(f)
|
|
199
|
+
continue
|
|
200
|
+
if len(text) <= max_chars:
|
|
201
|
+
out.append(f)
|
|
202
|
+
continue
|
|
203
|
+
ranges = slice_boundaries(text, max_chars)
|
|
204
|
+
total = len(ranges)
|
|
205
|
+
for i, (s, e) in enumerate(ranges):
|
|
206
|
+
out.append(FileSlice(path=f, start=s, end=e, index=i, total=total))
|
|
207
|
+
return out
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def read_slice_text(fs: FileSlice) -> str:
|
|
211
|
+
"""Read just this slice's characters from its parent file.
|
|
212
|
+
|
|
213
|
+
Goes through :func:`unit_source_text`, so a PDF slice indexes the extracted
|
|
214
|
+
text rather than the container's bytes — the offsets `expand_oversized_files`
|
|
215
|
+
computed and the string the prompt carries are then the same string (#2906).
|
|
216
|
+
"""
|
|
217
|
+
return unit_source_text(fs.path)[fs.start:fs.end]
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def bisect_slice(fs: FileSlice) -> tuple[FileSlice, FileSlice] | None:
|
|
221
|
+
"""Split a slice into two halves at a newline near its midpoint, or None.
|
|
222
|
+
|
|
223
|
+
Used by the adaptive-retry path when a single slice still overflows the
|
|
224
|
+
model's output: halving it produces a smaller response. Returns None when the
|
|
225
|
+
slice is already too small to split meaningfully.
|
|
226
|
+
"""
|
|
227
|
+
if fs.end - fs.start <= 1:
|
|
228
|
+
return None
|
|
229
|
+
# Index the SAME string the slice offsets were computed against and the
|
|
230
|
+
# prompt carries: for a PDF that is the extracted text via unit_source_text,
|
|
231
|
+
# not the raw container bytes. Reading the container here (the old behavior)
|
|
232
|
+
# searched for the newline cut in binary coordinates, so a compressed PDF
|
|
233
|
+
# slice could cut mid-line or past the text end (#2906). Any converter/read
|
|
234
|
+
# failure means we cannot split, so fall back to None (treated as atomic).
|
|
235
|
+
try:
|
|
236
|
+
text = unit_source_text(fs.path)
|
|
237
|
+
except Exception:
|
|
238
|
+
return None
|
|
239
|
+
mid = (fs.start + fs.end) // 2
|
|
240
|
+
nl = text.find("\n", mid, fs.end)
|
|
241
|
+
cut = nl + 1 if (nl != -1 and fs.start < nl + 1 < fs.end) else mid
|
|
242
|
+
if not (fs.start < cut < fs.end):
|
|
243
|
+
return None
|
|
244
|
+
left = FileSlice(fs.path, fs.start, cut, fs.index, fs.total)
|
|
245
|
+
right = FileSlice(fs.path, cut, fs.end, fs.index, fs.total)
|
|
246
|
+
return left, right
|
graphify/global_graph.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import json
|
|
3
|
+
import hashlib
|
|
4
|
+
import sys
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
import networkx as nx
|
|
8
|
+
from networkx.readwrite import json_graph as _jg
|
|
9
|
+
|
|
10
|
+
_GLOBAL_DIR = Path.home() / ".graphify"
|
|
11
|
+
_GLOBAL_GRAPH = _GLOBAL_DIR / "global-graph.json"
|
|
12
|
+
_GLOBAL_MANIFEST = _GLOBAL_DIR / "global-manifest.json"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _load_manifest() -> dict:
|
|
16
|
+
if _GLOBAL_MANIFEST.exists():
|
|
17
|
+
try:
|
|
18
|
+
return json.loads(_GLOBAL_MANIFEST.read_text(encoding="utf-8"))
|
|
19
|
+
except Exception as exc:
|
|
20
|
+
# Don't silently wipe the user's manifest on a parse error: that
|
|
21
|
+
# deletes every tracked repo. Back the bad file up and surface the
|
|
22
|
+
# error so the user can recover or report it.
|
|
23
|
+
backup = _GLOBAL_MANIFEST.with_suffix(
|
|
24
|
+
_GLOBAL_MANIFEST.suffix + f".corrupt.{int(datetime.now(timezone.utc).timestamp())}"
|
|
25
|
+
)
|
|
26
|
+
try:
|
|
27
|
+
_GLOBAL_MANIFEST.rename(backup)
|
|
28
|
+
print(
|
|
29
|
+
f"[graphify global] manifest at {_GLOBAL_MANIFEST} failed to parse ({exc}); "
|
|
30
|
+
f"moved to {backup} and starting fresh. Restore from the backup if this was "
|
|
31
|
+
f"unexpected.",
|
|
32
|
+
file=sys.stderr,
|
|
33
|
+
)
|
|
34
|
+
except Exception as rename_exc:
|
|
35
|
+
print(
|
|
36
|
+
f"[graphify global] manifest at {_GLOBAL_MANIFEST} failed to parse ({exc}) "
|
|
37
|
+
f"and could not be backed up ({rename_exc}). Starting fresh.",
|
|
38
|
+
file=sys.stderr,
|
|
39
|
+
)
|
|
40
|
+
return {"version": 1, "repos": {}}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _save_manifest(manifest: dict) -> None:
|
|
44
|
+
_GLOBAL_DIR.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
from graphify.paths import write_json_atomic
|
|
46
|
+
write_json_atomic(_GLOBAL_MANIFEST, manifest, indent=2)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _load_global_graph() -> nx.Graph:
|
|
50
|
+
if _GLOBAL_GRAPH.exists():
|
|
51
|
+
from graphify.security import check_graph_file_size_cap
|
|
52
|
+
check_graph_file_size_cap(_GLOBAL_GRAPH)
|
|
53
|
+
data = json.loads(_GLOBAL_GRAPH.read_text(encoding="utf-8"))
|
|
54
|
+
if "links" not in data and "edges" in data:
|
|
55
|
+
data = dict(data, links=data["edges"])
|
|
56
|
+
try:
|
|
57
|
+
return _jg.node_link_graph(data, edges="links")
|
|
58
|
+
except TypeError:
|
|
59
|
+
return _jg.node_link_graph(data)
|
|
60
|
+
return nx.Graph()
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _save_global_graph(G: nx.Graph) -> None:
|
|
64
|
+
_GLOBAL_DIR.mkdir(parents=True, exist_ok=True)
|
|
65
|
+
try:
|
|
66
|
+
data = _jg.node_link_data(G, edges="links")
|
|
67
|
+
except TypeError:
|
|
68
|
+
data = _jg.node_link_data(G)
|
|
69
|
+
from graphify.paths import write_json_atomic
|
|
70
|
+
write_json_atomic(_GLOBAL_GRAPH, data, indent=2)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _file_hash(path: Path) -> str:
|
|
74
|
+
h = hashlib.sha256()
|
|
75
|
+
h.update(path.read_bytes())
|
|
76
|
+
return h.hexdigest()[:16]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def global_add(source_path: Path, repo_tag: str) -> dict:
|
|
80
|
+
"""Add or update a project graph in the global graph.
|
|
81
|
+
|
|
82
|
+
Returns a summary dict with keys: repo_tag, nodes_added, nodes_removed, skipped,
|
|
83
|
+
cross_repo_calls.
|
|
84
|
+
Skipped=True means the source graph hasn't changed since last add.
|
|
85
|
+
"""
|
|
86
|
+
from graphify.build import prefix_graph_for_global, prune_repo_from_graph
|
|
87
|
+
|
|
88
|
+
if not source_path.exists():
|
|
89
|
+
raise FileNotFoundError(f"graph not found: {source_path}")
|
|
90
|
+
|
|
91
|
+
manifest = _load_manifest()
|
|
92
|
+
src_hash = _file_hash(source_path)
|
|
93
|
+
|
|
94
|
+
existing = manifest["repos"].get(repo_tag, {})
|
|
95
|
+
existing_path = existing.get("source_path", "")
|
|
96
|
+
if existing_path and existing_path != str(source_path.resolve()):
|
|
97
|
+
print(
|
|
98
|
+
f"[graphify global] warning: repo tag '{repo_tag}' previously pointed to "
|
|
99
|
+
f"{existing_path!r}, now updating to {str(source_path.resolve())!r}. "
|
|
100
|
+
f"Use --as <tag> to give it a different name.",
|
|
101
|
+
file=sys.stderr,
|
|
102
|
+
)
|
|
103
|
+
if existing.get("source_hash") == src_hash:
|
|
104
|
+
return {"repo_tag": repo_tag, "nodes_added": 0, "nodes_removed": 0, "skipped": True,
|
|
105
|
+
"cross_repo_calls": 0}
|
|
106
|
+
|
|
107
|
+
# Load source graph
|
|
108
|
+
from graphify.security import check_graph_file_size_cap
|
|
109
|
+
check_graph_file_size_cap(source_path)
|
|
110
|
+
data = json.loads(source_path.read_text(encoding="utf-8"))
|
|
111
|
+
if "links" not in data and "edges" in data:
|
|
112
|
+
data = dict(data, links=data["edges"])
|
|
113
|
+
try:
|
|
114
|
+
src_G = _jg.node_link_graph(data, edges="links")
|
|
115
|
+
except TypeError:
|
|
116
|
+
src_G = _jg.node_link_graph(data)
|
|
117
|
+
|
|
118
|
+
# Prefix IDs for cross-project isolation
|
|
119
|
+
prefixed = prefix_graph_for_global(src_G, repo_tag)
|
|
120
|
+
|
|
121
|
+
# Load global graph and prune stale nodes for this repo
|
|
122
|
+
G = _load_global_graph()
|
|
123
|
+
removed = prune_repo_from_graph(G, repo_tag)
|
|
124
|
+
|
|
125
|
+
# Merge external-library nodes (no source_file) by label to avoid duplication
|
|
126
|
+
external_labels = {
|
|
127
|
+
d.get("label", ""): n
|
|
128
|
+
for n, d in G.nodes(data=True)
|
|
129
|
+
if not d.get("source_file") and d.get("label")
|
|
130
|
+
}
|
|
131
|
+
# Map each deduplicated external onto the existing global node so that
|
|
132
|
+
# edges incident to it can be rewired instead of dropped.
|
|
133
|
+
remap = {}
|
|
134
|
+
for node, data in prefixed.nodes(data=True):
|
|
135
|
+
if not data.get("source_file") and data.get("label") in external_labels:
|
|
136
|
+
remap[node] = external_labels[data["label"]]
|
|
137
|
+
|
|
138
|
+
# Compose: add prefixed nodes (except deduplicated externals) into global graph
|
|
139
|
+
for node, data in prefixed.nodes(data=True):
|
|
140
|
+
if node not in remap:
|
|
141
|
+
G.add_node(node, **data)
|
|
142
|
+
for u, v, data in prefixed.edges(data=True):
|
|
143
|
+
u = remap.get(u, u)
|
|
144
|
+
v = remap.get(v, v)
|
|
145
|
+
if u != v: # don't introduce self-loops via remapping
|
|
146
|
+
G.add_edge(u, v, **data)
|
|
147
|
+
|
|
148
|
+
added = prefixed.number_of_nodes() - len(remap)
|
|
149
|
+
# A member call parked on a caller node (#3152) may be answered by a repo
|
|
150
|
+
# already in the global graph, or by this one for a repo added earlier. The
|
|
151
|
+
# pass recomputes its own output, so adding repos one at a time lands where a
|
|
152
|
+
# single merge-graphs of the same inputs would.
|
|
153
|
+
from graphify.cross_repo_calls import link_cross_repo_member_calls
|
|
154
|
+
|
|
155
|
+
cross_repo_calls = link_cross_repo_member_calls(G)
|
|
156
|
+
_save_global_graph(G)
|
|
157
|
+
|
|
158
|
+
manifest["repos"][repo_tag] = {
|
|
159
|
+
"added_at": datetime.now(timezone.utc).isoformat(),
|
|
160
|
+
"source_path": str(source_path.resolve()),
|
|
161
|
+
"node_count": added,
|
|
162
|
+
"edge_count": prefixed.number_of_edges(),
|
|
163
|
+
"source_hash": src_hash,
|
|
164
|
+
}
|
|
165
|
+
_save_manifest(manifest)
|
|
166
|
+
|
|
167
|
+
return {"repo_tag": repo_tag, "nodes_added": added, "nodes_removed": removed,
|
|
168
|
+
"skipped": False, "cross_repo_calls": cross_repo_calls}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def global_remove(repo_tag: str) -> int:
|
|
172
|
+
"""Remove all nodes for repo_tag from the global graph. Returns count removed."""
|
|
173
|
+
from graphify.build import prune_repo_from_graph
|
|
174
|
+
|
|
175
|
+
manifest = _load_manifest()
|
|
176
|
+
if repo_tag not in manifest["repos"]:
|
|
177
|
+
raise KeyError(f"repo '{repo_tag}' not in global graph")
|
|
178
|
+
|
|
179
|
+
G = _load_global_graph()
|
|
180
|
+
removed = prune_repo_from_graph(G, repo_tag)
|
|
181
|
+
_save_global_graph(G)
|
|
182
|
+
|
|
183
|
+
del manifest["repos"][repo_tag]
|
|
184
|
+
_save_manifest(manifest)
|
|
185
|
+
return removed
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def global_list() -> dict:
|
|
189
|
+
"""Return the manifest repos dict."""
|
|
190
|
+
return _load_manifest().get("repos", {})
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def global_path() -> Path:
|
|
194
|
+
return _GLOBAL_GRAPH
|