graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/build.py
ADDED
|
@@ -0,0 +1,2300 @@
|
|
|
1
|
+
# assemble node+edge dicts into a NetworkX graph, preserving edge direction
|
|
2
|
+
#
|
|
3
|
+
# Node deduplication — three layers:
|
|
4
|
+
#
|
|
5
|
+
# 1. Within a file (AST): each extractor tracks a `seen_ids` set. A node ID is
|
|
6
|
+
# emitted at most once per file, so duplicate class/function definitions in
|
|
7
|
+
# the same source file are collapsed to the first occurrence.
|
|
8
|
+
#
|
|
9
|
+
# 2. Between files (build): NetworkX G.add_node() is idempotent — calling it
|
|
10
|
+
# twice with the same ID overwrites the attributes with the second call's
|
|
11
|
+
# values. Nodes are added in extraction order (AST first, then semantic),
|
|
12
|
+
# so if the same entity is extracted by both passes the semantic node
|
|
13
|
+
# silently overwrites the AST node. This is intentional: semantic nodes
|
|
14
|
+
# carry richer labels and cross-file context, while AST nodes have precise
|
|
15
|
+
# source_location. If you need to change the priority, reorder extractions
|
|
16
|
+
# passed to build().
|
|
17
|
+
#
|
|
18
|
+
# 3. Semantic merge (skill): before calling build(), the skill merges cached
|
|
19
|
+
# and new semantic results using an explicit `seen` set keyed on node["id"],
|
|
20
|
+
# so duplicates across cache hits and new extractions are resolved there
|
|
21
|
+
# before any graph construction happens.
|
|
22
|
+
#
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
import json
|
|
25
|
+
import math
|
|
26
|
+
import os
|
|
27
|
+
import re
|
|
28
|
+
import sys
|
|
29
|
+
import unicodedata
|
|
30
|
+
from collections.abc import Iterable
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
import networkx as nx
|
|
33
|
+
from .ids import make_id, normalize_id as _normalize_id
|
|
34
|
+
from .paths import default_graph_json as _default_graph_json
|
|
35
|
+
from .paths import is_absolute_any_platform as _is_abs
|
|
36
|
+
from .validate import validate_extraction
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# Deterministic (AST) extractors emit source_location "L<line>"; the semantic
|
|
40
|
+
# extraction spec emits null. Used by _is_ast_tier as a shape fallback for
|
|
41
|
+
# legacy items that predate the _origin marker (#2334).
|
|
42
|
+
_AST_LOC_RE = re.compile(r"^L\d")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _is_ast_tier(item: dict) -> bool:
|
|
46
|
+
"""AST vs semantic tier. _origin wins when present; unstamped legacy items
|
|
47
|
+
(pre-0.9.16) fall back to shape: deterministic extractors emit
|
|
48
|
+
source_location 'L<line>', the semantic spec emits null (#2334)."""
|
|
49
|
+
o = item.get("_origin")
|
|
50
|
+
if o is not None:
|
|
51
|
+
return o == "ast"
|
|
52
|
+
loc = item.get("source_location")
|
|
53
|
+
return isinstance(loc, str) and bool(_AST_LOC_RE.match(loc))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# Relations that say only "these two symbols appear together", with no claim about
|
|
57
|
+
# HOW. An extractor that finds a specific fact for a pair — a call, an import, an
|
|
58
|
+
# inheritance — routinely emits one of these for the same pair as well, so when the
|
|
59
|
+
# simple graph collapses the pair to one edge, the generic one must never be the
|
|
60
|
+
# survivor. Deliberately a small denylist rather than a full precedence order over
|
|
61
|
+
# every relation: ranking `contains` against `calls` would be inventing a
|
|
62
|
+
# cross-axis judgement, whereas "specific beats generic" is the only comparison
|
|
63
|
+
# this collapse actually needs.
|
|
64
|
+
_GENERIC_RELATIONS: frozenset[str] = frozenset({"references", "uses", "mentions"})
|
|
65
|
+
|
|
66
|
+
# Language interop families, keyed by extension, for the cross-language phantom-edge
|
|
67
|
+
# guard in the edge loop below. Families group by REAL interop (JS/TS share a module
|
|
68
|
+
# graph; C/C++/ObjC share a compilation unit via headers; JVM langs share bytecode),
|
|
69
|
+
# so a legitimate TS->JS import or C impl->header call survives, while a Python
|
|
70
|
+
# `import time` binding to a `time.ts` (#1749) or a cross-language INFERRED `calls`
|
|
71
|
+
# edge (#1547/#1556) is dropped. Kept local to build.py (not imported from extract.py,
|
|
72
|
+
# which imports build.py — a cycle) and deliberately mirrors extract._LANG_FAMILY_BY_EXT.
|
|
73
|
+
_EDGE_LANG_FAMILY: dict[str, str] = {
|
|
74
|
+
".py": "py", ".pyi": "py",
|
|
75
|
+
".js": "js", ".mjs": "js", ".cjs": "js", ".jsx": "js",
|
|
76
|
+
".ts": "js", ".tsx": "js", ".mts": "js", ".cts": "js",
|
|
77
|
+
".go": "go", ".rs": "rs",
|
|
78
|
+
".java": "jvm", ".kt": "jvm", ".scala": "jvm", ".groovy": "jvm",
|
|
79
|
+
".c": "c", ".h": "c", ".cc": "c", ".cpp": "c", ".hpp": "c",
|
|
80
|
+
".cxx": "c", ".hh": "c", ".hxx": "c",
|
|
81
|
+
".cu": "c", ".cuh": "c", ".metal": "c", ".m": "c", ".mm": "c",
|
|
82
|
+
".rb": "rb", ".rake": "rb", ".php": "php", ".cs": "cs", ".swift": "swift", ".lua": "lua",
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# Synonym mapper for known invalid file_type values that LLM subagents commonly
|
|
87
|
+
# emit. Keeps semantic intent close (markdown→document, tool→code) and falls
|
|
88
|
+
# back to "concept" for any other invalid value (see #840).
|
|
89
|
+
_FILE_TYPE_SYNONYMS = {
|
|
90
|
+
"markdown": "document",
|
|
91
|
+
"text": "document",
|
|
92
|
+
"tool": "code",
|
|
93
|
+
"library": "code",
|
|
94
|
+
"pattern": "concept",
|
|
95
|
+
"principle": "concept",
|
|
96
|
+
"constraint": "concept",
|
|
97
|
+
"tech": "concept",
|
|
98
|
+
"technology": "concept",
|
|
99
|
+
"data-source": "concept",
|
|
100
|
+
"data_source": "concept",
|
|
101
|
+
"gotcha": "concept",
|
|
102
|
+
"framework": "concept",
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# Hyperedge member lists are canonically keyed `nodes` (see graphify/llm.py
|
|
107
|
+
# extraction spec), but LLM/subagent drift and externally-supplied graph.json
|
|
108
|
+
# sometimes emit `members` or `node_ids`. _normalize_hyperedge_members folds
|
|
109
|
+
# those aliases into `nodes` at ingest so every downstream consumer reads one
|
|
110
|
+
# canonical key — mirroring the `from`/`to` edge-endpoint tolerance below.
|
|
111
|
+
_HE_MEMBER_ALIASES = ("members", "node_ids")
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _coerce_hyperedge_member_refs(he: dict, members: list) -> list:
|
|
115
|
+
"""Coerce a hyperedge member list to hashable scalar ids, deduped in order.
|
|
116
|
+
|
|
117
|
+
LLM/subagent drift sometimes emits a member as an object (``{"id": "a_ts"}``)
|
|
118
|
+
instead of a bare id string. Left uncoerced, the dict member is unhashable,
|
|
119
|
+
so the semantic-rekey pass's ``_rekey.get(n, n)`` raised ``TypeError`` and
|
|
120
|
+
aborted the whole merge — destroying a completed extraction (#2486). Object
|
|
121
|
+
members collapse to their non-empty ``id`` (numeric ids str-coerced via
|
|
122
|
+
``_coerce_id``, matching #2326); members with no usable id are dropped with
|
|
123
|
+
a stderr WARNING naming the hyperedge, never a crash. Hashable scalar refs
|
|
124
|
+
pass through unchanged. A hyperedge that loses every member this way falls
|
|
125
|
+
to the existing no-valid-members drop-with-warning in ``build_from_json``.
|
|
126
|
+
"""
|
|
127
|
+
seen: set = set()
|
|
128
|
+
coerced: list = []
|
|
129
|
+
for ref in members:
|
|
130
|
+
if isinstance(ref, dict):
|
|
131
|
+
inner = _coerce_id(ref.get("id"))
|
|
132
|
+
if inner in (None, "") or not _hashable(inner):
|
|
133
|
+
print(
|
|
134
|
+
f"[graphify] WARNING: hyperedge "
|
|
135
|
+
f"'{he.get('id', '?')}' has a member object with no usable "
|
|
136
|
+
f"'id' ({ref!r}); dropping that member.",
|
|
137
|
+
file=sys.stderr,
|
|
138
|
+
)
|
|
139
|
+
continue
|
|
140
|
+
ref = inner
|
|
141
|
+
elif not _hashable(ref):
|
|
142
|
+
print(
|
|
143
|
+
f"[graphify] WARNING: hyperedge "
|
|
144
|
+
f"'{he.get('id', '?')}' has an unusable member reference "
|
|
145
|
+
f"{ref!r}; dropping that member.",
|
|
146
|
+
file=sys.stderr,
|
|
147
|
+
)
|
|
148
|
+
continue
|
|
149
|
+
if ref in seen:
|
|
150
|
+
continue
|
|
151
|
+
seen.add(ref)
|
|
152
|
+
coerced.append(ref)
|
|
153
|
+
return coerced
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _normalize_hyperedge_members(he: object) -> None:
|
|
157
|
+
"""Canonicalize a hyperedge's member list onto the `nodes` key, in place.
|
|
158
|
+
|
|
159
|
+
If `nodes` is already a list it wins (canonical), and only stray alias keys
|
|
160
|
+
are dropped. Otherwise the first alias (`members`, then `node_ids`) that is a
|
|
161
|
+
list is moved to `nodes`, with a single stderr WARNING naming the hyperedge
|
|
162
|
+
id and alias used. Leftover alias keys are always removed so downstream code
|
|
163
|
+
never re-reads them. Whichever branch supplied the list, member VALUES are
|
|
164
|
+
coerced to hashable scalar ids and deduped preserving order (#2486) — see
|
|
165
|
+
``_coerce_hyperedge_member_refs``.
|
|
166
|
+
"""
|
|
167
|
+
if not isinstance(he, dict):
|
|
168
|
+
return
|
|
169
|
+
if isinstance(he.get("nodes"), list):
|
|
170
|
+
he["nodes"] = _coerce_hyperedge_member_refs(he, he["nodes"])
|
|
171
|
+
else:
|
|
172
|
+
for alias in _HE_MEMBER_ALIASES:
|
|
173
|
+
val = he.get(alias)
|
|
174
|
+
if isinstance(val, list):
|
|
175
|
+
he["nodes"] = _coerce_hyperedge_member_refs(he, val)
|
|
176
|
+
print(
|
|
177
|
+
f"[graphify] WARNING: hyperedge "
|
|
178
|
+
f"'{he.get('id', '?')}' uses field '{alias}' instead of "
|
|
179
|
+
f"'nodes'; normalizing.",
|
|
180
|
+
file=sys.stderr,
|
|
181
|
+
)
|
|
182
|
+
break
|
|
183
|
+
# Drop any leftover alias keys regardless of which branch ran above.
|
|
184
|
+
for alias in _HE_MEMBER_ALIASES:
|
|
185
|
+
he.pop(alias, None)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _fold_node_aliases(node: dict) -> None:
|
|
189
|
+
"""Fold legacy node field aliases onto canonical keys, in place (#2194).
|
|
190
|
+
|
|
191
|
+
``name`` -> ``label`` and ``path`` -> ``source_file``. Uses an empty-check
|
|
192
|
+
(not mere key presence) so a node carrying ``label: ""``/``None`` next to a
|
|
193
|
+
real ``name`` is healed too. When the canonical field already holds a value
|
|
194
|
+
it wins and the alias key is left untouched. Without this fold an alias-only
|
|
195
|
+
node enters the graph with no label/source_file: it fails validation, gets
|
|
196
|
+
``norm_label == ""`` (invisible to query/explain), and is excluded from every
|
|
197
|
+
label-keyed merge/dedup — a permanent ghost that ``graphify update``
|
|
198
|
+
re-feeds through build_from_json forever.
|
|
199
|
+
"""
|
|
200
|
+
if not node.get("label") and isinstance(node.get("name"), str) and node["name"]:
|
|
201
|
+
node["label"] = node.pop("name")
|
|
202
|
+
if not node.get("source_file") and isinstance(node.get("path"), str) and node["path"]:
|
|
203
|
+
node["source_file"] = node.pop("path")
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _fold_edge_aliases(edge: dict) -> None:
|
|
207
|
+
"""Fold legacy edge field aliases onto canonical keys, in place (#2194).
|
|
208
|
+
|
|
209
|
+
``type`` -> ``relation``. A ``confidence_score`` float with no ``confidence``
|
|
210
|
+
enum backfills ``confidence: "INFERRED"`` — never EXTRACTED (alias recovery
|
|
211
|
+
is not provenance) and never a threshold mapping of the float. The
|
|
212
|
+
``confidence_score`` key itself is NOT popped: it is a legitimate companion
|
|
213
|
+
field that the edge loop sanitizes and to_json round-trips.
|
|
214
|
+
|
|
215
|
+
A NUMERIC ``confidence`` (pre-enum graphs stored the LLM pass's float —
|
|
216
|
+
1.0/0.95/0.9/0.85 — directly in the field) normalizes to ``INFERRED``:
|
|
217
|
+
numeric confidences only ever came from the LLM semantic pass, and
|
|
218
|
+
LLM-derived edges are INFERRED by definition. The original float moves to
|
|
219
|
+
``confidence_score`` unless an explicit one is already present (the
|
|
220
|
+
companion field is the authority). Without this fold, every reload of a
|
|
221
|
+
pre-enum graph re-warns once per legacy edge, forever. ``bool`` is
|
|
222
|
+
excluded despite subclassing ``int``: ``True`` is not a score.
|
|
223
|
+
"""
|
|
224
|
+
if not edge.get("relation") and isinstance(edge.get("type"), str) and edge["type"]:
|
|
225
|
+
edge["relation"] = edge.pop("type")
|
|
226
|
+
_conf = edge.get("confidence")
|
|
227
|
+
if isinstance(_conf, (int, float)) and not isinstance(_conf, bool):
|
|
228
|
+
if edge.get("confidence_score") is None:
|
|
229
|
+
edge["confidence_score"] = float(_conf)
|
|
230
|
+
edge["confidence"] = "INFERRED"
|
|
231
|
+
if not edge.get("confidence") and edge.get("confidence_score") is not None:
|
|
232
|
+
edge["confidence"] = "INFERRED"
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _coerce_id(value: object) -> object:
|
|
236
|
+
"""Return a str for a numeric id, else the value unchanged.
|
|
237
|
+
|
|
238
|
+
``bool`` is excluded despite subclassing ``int``: an id of ``True`` is not a
|
|
239
|
+
number the model meant to name a node, and ``"True"`` would invent a label.
|
|
240
|
+
"""
|
|
241
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
242
|
+
return value
|
|
243
|
+
return str(value)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _hashable(value: object) -> bool:
|
|
247
|
+
"""True when value can be a dict key / set member (same probe as the
|
|
248
|
+
inline ``try: hash(m)`` in build_from_json's hyperedge revalidation)."""
|
|
249
|
+
try:
|
|
250
|
+
hash(value)
|
|
251
|
+
except TypeError:
|
|
252
|
+
return False
|
|
253
|
+
return True
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _coerce_non_string_ids(extraction: dict) -> None:
|
|
257
|
+
"""Coerce numeric node ids and edge/hyperedge references to str, in place (#2326).
|
|
258
|
+
|
|
259
|
+
A backend can emit ``{"id": 10}`` where the schema says ``{"id": "10"}``.
|
|
260
|
+
Every id consumer downstream assumes ``str``, so one int id aborted the build
|
|
261
|
+
in three places: ``_pick_winner``'s ``_CHUNK_SUFFIX.search(n["id"])`` raised
|
|
262
|
+
``TypeError: expected string or bytes-like object``, and ``build_from_json``'s
|
|
263
|
+
``sorted(node_set)`` raised ``'<' not supported between instances of 'str'
|
|
264
|
+
and 'int'`` — the latter for a lone node with nothing to dedup at all.
|
|
265
|
+
Coercing keeps the node and its edges rather than dropping either, which is
|
|
266
|
+
the same tolerate-and-heal treatment loose backend output already gets at the
|
|
267
|
+
parse chokepoint (#1631) and in the alias folds (#2194).
|
|
268
|
+
|
|
269
|
+
Endpoints and hyperedge members are coerced with the nodes, not after: a
|
|
270
|
+
node-only coercion would renumber ``10`` to ``"10"`` and leave every edge
|
|
271
|
+
pointing at the vanished ``10``, trading a loud crash for a silently
|
|
272
|
+
disconnected graph. The legacy ``from``/``to`` endpoint aliases are included
|
|
273
|
+
because dedup reads them directly (#803).
|
|
274
|
+
|
|
275
|
+
Runs in BOTH ``build`` (before dedup, which keys on id) and
|
|
276
|
+
``build_from_json`` (the direct entry that reloads a persisted graph), for
|
|
277
|
+
the same two-site reason as the ``_fold_node_aliases`` fold (#2194). It is
|
|
278
|
+
idempotent, so the nested call on the ``build`` path is a no-op.
|
|
279
|
+
|
|
280
|
+
Non-numeric non-str ids (``None``, lists, dicts) are left alone for
|
|
281
|
+
``validate_extraction`` to report: ``str(None) == "None"`` would fabricate a
|
|
282
|
+
node id that no edge references.
|
|
283
|
+
"""
|
|
284
|
+
for node in extraction.get("nodes") or ():
|
|
285
|
+
if isinstance(node, dict) and "id" in node:
|
|
286
|
+
node["id"] = _coerce_id(node["id"])
|
|
287
|
+
for edge in extraction.get("edges") or ():
|
|
288
|
+
if not isinstance(edge, dict):
|
|
289
|
+
continue
|
|
290
|
+
for key in ("source", "target", "from", "to"):
|
|
291
|
+
if key in edge:
|
|
292
|
+
edge[key] = _coerce_id(edge[key])
|
|
293
|
+
for he in extraction.get("hyperedges") or ():
|
|
294
|
+
if not isinstance(he, dict):
|
|
295
|
+
continue
|
|
296
|
+
members = he.get("nodes")
|
|
297
|
+
if isinstance(members, list):
|
|
298
|
+
he["nodes"] = [_coerce_id(ref) for ref in members]
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _norm_source_file(p: str | None, root: str | None = None) -> str | None:
|
|
302
|
+
"""Normalize path separators and relativize absolute paths.
|
|
303
|
+
|
|
304
|
+
Converts backslashes to forward slashes (Windows compatibility) and, when
|
|
305
|
+
root is provided, strips the absolute prefix from paths produced by semantic
|
|
306
|
+
subagents so source_file is always repo-relative (fixes #932).
|
|
307
|
+
"""
|
|
308
|
+
if not p:
|
|
309
|
+
return p
|
|
310
|
+
p = p.replace("\\", "/")
|
|
311
|
+
if root and _is_abs(p):
|
|
312
|
+
try:
|
|
313
|
+
p = Path(p).relative_to(root).as_posix()
|
|
314
|
+
except ValueError:
|
|
315
|
+
# Lexical relative_to failed. Retry with both sides fully resolved:
|
|
316
|
+
# a symlinked scan root (macOS /var -> /private/var, or a symlinked
|
|
317
|
+
# home/worktree) makes the raw prefixes differ even though they point
|
|
318
|
+
# at the same dir, which otherwise silently defeats prune/replace
|
|
319
|
+
# matching. Only the slow path resolves, so the common lexical match
|
|
320
|
+
# stays filesystem-free.
|
|
321
|
+
try:
|
|
322
|
+
p = Path(p).resolve().relative_to(Path(root).resolve()).as_posix()
|
|
323
|
+
except (ValueError, OSError):
|
|
324
|
+
pass
|
|
325
|
+
return p
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def _abs_identity(p: str | None, root: str | None = None) -> str | None:
|
|
329
|
+
"""Return a form-insensitive absolute identity for a source_file.
|
|
330
|
+
|
|
331
|
+
prune/replace matching in build_merge otherwise compares raw strings against
|
|
332
|
+
``_norm_source_file`` output, so a node whose source_file survived in a THIRD
|
|
333
|
+
form — absolute where prune_sources is relative, or vice versa, or a symlinked
|
|
334
|
+
root — slips past every equality check and its nodes/edges are never pruned
|
|
335
|
+
(silent survival of a deleted file's graph, #2012). Anchoring relative paths
|
|
336
|
+
at ``root`` and resolving both sides to a canonical absolute posix path gives
|
|
337
|
+
a fallback that matches regardless of which form each side happens to hold.
|
|
338
|
+
"""
|
|
339
|
+
if not p:
|
|
340
|
+
return None
|
|
341
|
+
q = p.replace("\\", "/")
|
|
342
|
+
pp = Path(q)
|
|
343
|
+
if not _is_abs(q) and root:
|
|
344
|
+
pp = Path(root) / q
|
|
345
|
+
try:
|
|
346
|
+
return pp.resolve().as_posix()
|
|
347
|
+
except OSError:
|
|
348
|
+
return pp.as_posix()
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _is_file_node_label(label: "str | None", source_file: "str | None") -> bool:
|
|
352
|
+
"""Whether *label* is a file node's label for *source_file* — the bare
|
|
353
|
+
basename, OR a directory-qualified suffix produced by the disambiguation pass
|
|
354
|
+
below (#2032). Used both to recognize file nodes when relabeling and by the
|
|
355
|
+
downstream file-node predicates (analyze/tree/serve)."""
|
|
356
|
+
if not label or not source_file:
|
|
357
|
+
return False
|
|
358
|
+
sf = str(source_file).replace("\\", "/")
|
|
359
|
+
lbl = str(label)
|
|
360
|
+
if lbl == sf.rsplit("/", 1)[-1]:
|
|
361
|
+
return True
|
|
362
|
+
return "/" in lbl and (sf == lbl or sf.endswith("/" + lbl))
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def _shortest_unique_suffix(sf: str, all_sfs: "set[str]") -> str:
|
|
366
|
+
"""Shortest trailing path suffix (basename + k parent dirs) of *sf* that is
|
|
367
|
+
unique among *all_sfs*. `a/b/index.ts` vs `c/b/index.ts` -> `a/b/index.ts`;
|
|
368
|
+
`x/index.ts` vs `y/index.ts` -> `x/index.ts`. Derived from the path (never the
|
|
369
|
+
current label) so relabeling is idempotent across incremental rebuilds."""
|
|
370
|
+
parts = [p for p in sf.replace("\\", "/").split("/") if p]
|
|
371
|
+
others = [
|
|
372
|
+
[p for p in o.replace("\\", "/").split("/") if p]
|
|
373
|
+
for o in all_sfs if o != sf
|
|
374
|
+
]
|
|
375
|
+
for k in range(1, len(parts) + 1):
|
|
376
|
+
suffix = parts[-k:]
|
|
377
|
+
if all(o[-k:] != suffix for o in others):
|
|
378
|
+
return "/".join(suffix)
|
|
379
|
+
return "/".join(parts)
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _file_label_reassignments(items: "list[tuple]") -> dict:
|
|
383
|
+
"""Given (key, label, source_file) triples, return {key: new_label} for file
|
|
384
|
+
nodes whose basename collides with another's — the shortest unique
|
|
385
|
+
directory-qualified suffix (#2032). Keys of non-colliding/basename-unique file
|
|
386
|
+
nodes are omitted (their label stays bare)."""
|
|
387
|
+
from collections import defaultdict
|
|
388
|
+
groups: dict[str, list[tuple]] = defaultdict(list)
|
|
389
|
+
for key, label, sf in items:
|
|
390
|
+
if sf and label and _is_file_node_label(str(label), str(sf)):
|
|
391
|
+
basename = str(sf).replace("\\", "/").rsplit("/", 1)[-1]
|
|
392
|
+
groups[basename].append((key, str(sf)))
|
|
393
|
+
out: dict = {}
|
|
394
|
+
for members in groups.values():
|
|
395
|
+
distinct = {sf for _, sf in members}
|
|
396
|
+
if len(distinct) < 2:
|
|
397
|
+
continue # no collision — leave the bare basename label
|
|
398
|
+
for key, sf in members:
|
|
399
|
+
out[key] = _shortest_unique_suffix(sf, distinct)
|
|
400
|
+
return out
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _disambiguate_file_node_labels(G: "nx.Graph") -> None:
|
|
404
|
+
"""Relabel colliding-basename file nodes on a graph (#2032). Ids/edges are
|
|
405
|
+
never changed — only display labels. Idempotent (labels derive from
|
|
406
|
+
source_file, not the current possibly-qualified label)."""
|
|
407
|
+
items = [(nid, a.get("label"), a.get("source_file")) for nid, a in G.nodes(data=True)]
|
|
408
|
+
for nid, new_label in _file_label_reassignments(items).items():
|
|
409
|
+
G.nodes[nid]["label"] = new_label
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def disambiguate_file_labels_in_nodes(nodes: "list") -> None:
|
|
413
|
+
"""Relabel colliding-basename file nodes on a raw node-dict list, in place
|
|
414
|
+
(#2032). Used by the extract --no-cluster path, which writes the merged
|
|
415
|
+
extraction directly without going through build_from_json."""
|
|
416
|
+
items = [
|
|
417
|
+
(i, n.get("label"), n.get("source_file"))
|
|
418
|
+
for i, n in enumerate(nodes) if isinstance(n, dict)
|
|
419
|
+
]
|
|
420
|
+
for i, new_label in _file_label_reassignments(items).items():
|
|
421
|
+
nodes[i]["label"] = new_label
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _infer_merge_root(graph_path: Path) -> str | None:
|
|
425
|
+
"""Best-effort scan root for relativizing paths in build_merge when the caller
|
|
426
|
+
passes no ``root`` (#1571).
|
|
427
|
+
|
|
428
|
+
Prefers the committed ``graphify-out/.graphify_root`` marker — the authoritative
|
|
429
|
+
scan root graphify records at build/watch time (#686/#1423) — then falls back to
|
|
430
|
+
the directory that contains the output dir (``graph.json``'s grandparent, i.e.
|
|
431
|
+
``<root>/graphify-out/graph.json`` -> ``<root>``). The grandparent heuristic is
|
|
432
|
+
applied only when ``graph.json``'s own directory actually looks like a graphify
|
|
433
|
+
out-dir (named like one, or holding the marker/manifest); for arbitrary layouts
|
|
434
|
+
(``<root>/graph.json``, #2446) the grandparent is NOT the scan root, and guessing
|
|
435
|
+
it made absolute prune_sources silently no-op. Returns None if neither resolves,
|
|
436
|
+
in which case normalization is a no-op (prior behavior) and prune matching can
|
|
437
|
+
still recover via :func:`_derive_prune_root`.
|
|
438
|
+
"""
|
|
439
|
+
parent = graph_path.parent
|
|
440
|
+
try:
|
|
441
|
+
marker = parent / ".graphify_root"
|
|
442
|
+
if marker.exists():
|
|
443
|
+
recorded = marker.read_text(encoding="utf-8-sig").strip()
|
|
444
|
+
if recorded:
|
|
445
|
+
return str(Path(recorded).resolve())
|
|
446
|
+
except OSError:
|
|
447
|
+
pass
|
|
448
|
+
from .paths import GRAPHIFY_OUT
|
|
449
|
+
try:
|
|
450
|
+
if (
|
|
451
|
+
parent.name == Path(GRAPHIFY_OUT).name
|
|
452
|
+
or (parent / ".graphify_root").exists()
|
|
453
|
+
or (parent / "manifest.json").exists()
|
|
454
|
+
):
|
|
455
|
+
return str(parent.parent.resolve())
|
|
456
|
+
except Exception:
|
|
457
|
+
pass
|
|
458
|
+
return None
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _build_prune_sets(
|
|
462
|
+
prune_sources: "list[str] | None",
|
|
463
|
+
eff_root: "str | None",
|
|
464
|
+
new_sources: "set[str]",
|
|
465
|
+
) -> "tuple[dict[str, str], dict[str, str]]":
|
|
466
|
+
"""Prune match sets for build_merge / merge_raw_extraction.
|
|
467
|
+
|
|
468
|
+
Returns ``(prune_set, prune_abs)`` mapping every match form of each prune
|
|
469
|
+
entry — the raw string, the :func:`_norm_source_file` relative form, and the
|
|
470
|
+
:func:`_abs_identity` absolute form — back to the ORIGINAL entry, so callers
|
|
471
|
+
can report which entries actually matched something (#2446). Membership
|
|
472
|
+
tests read like the old set-based code (``sf in prune_set``).
|
|
473
|
+
|
|
474
|
+
"Replace" wins over a contradictory "delete" of the same source (#1796), in
|
|
475
|
+
both string and absolute-identity space (#2012): every form belonging to a
|
|
476
|
+
source re-extracted this run is removed.
|
|
477
|
+
"""
|
|
478
|
+
prune_set: dict[str, str] = {}
|
|
479
|
+
prune_abs: dict[str, str] = {}
|
|
480
|
+
for p in (prune_sources or []):
|
|
481
|
+
if not p:
|
|
482
|
+
continue
|
|
483
|
+
prune_set.setdefault(p, p)
|
|
484
|
+
norm = _norm_source_file(p, eff_root)
|
|
485
|
+
if norm:
|
|
486
|
+
prune_set.setdefault(norm, p)
|
|
487
|
+
a = _abs_identity(p, eff_root)
|
|
488
|
+
if a:
|
|
489
|
+
prune_abs.setdefault(a, p)
|
|
490
|
+
for s in new_sources:
|
|
491
|
+
prune_set.pop(s, None)
|
|
492
|
+
a = _abs_identity(s, eff_root)
|
|
493
|
+
if a:
|
|
494
|
+
prune_abs.pop(a, None)
|
|
495
|
+
return prune_set, prune_abs
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _derive_prune_root(prune_sources: "list[str]", stored_sfs: "set[str]") -> "str | None":
|
|
499
|
+
"""Derive the scan root from absolute prune paths that matched nothing (#2446).
|
|
500
|
+
|
|
501
|
+
When build_merge/merge_raw_extraction receive absolute prune_sources but no
|
|
502
|
+
``root``, :func:`_infer_merge_root`'s guess can be wrong for non-standard
|
|
503
|
+
layouts (``graph.json`` not under ``<root>/graphify-out/``), so every prune
|
|
504
|
+
entry silently no-ops. Each absolute prune path ``P`` that ends with a
|
|
505
|
+
stored RELATIVE source_file ``S`` implies the candidate root
|
|
506
|
+
``P[:-len(S)-1]``. Within one graph all relative source_files share a single
|
|
507
|
+
scan root, so the candidate is accepted only when it is unique and
|
|
508
|
+
consistent across every suffix-matched entry; on ambiguity (or no suffix
|
|
509
|
+
match at all) returns None and the caller falls through to the zero-match
|
|
510
|
+
warning.
|
|
511
|
+
"""
|
|
512
|
+
rel_sfs = [
|
|
513
|
+
sf.replace("\\", "/")
|
|
514
|
+
for sf in stored_sfs
|
|
515
|
+
if sf and isinstance(sf, str) and not _is_abs(sf.replace("\\", "/"))
|
|
516
|
+
]
|
|
517
|
+
if not rel_sfs:
|
|
518
|
+
return None
|
|
519
|
+
roots: set[str] = set()
|
|
520
|
+
for p in prune_sources:
|
|
521
|
+
if not p or not isinstance(p, str):
|
|
522
|
+
continue
|
|
523
|
+
q = p.replace("\\", "/")
|
|
524
|
+
if not _is_abs(q):
|
|
525
|
+
continue
|
|
526
|
+
hits = {q[: -len(s) - 1] for s in rel_sfs if q.endswith("/" + s)}
|
|
527
|
+
if len(hits) > 1:
|
|
528
|
+
return None # one entry implies two different roots — ambiguous
|
|
529
|
+
roots |= hits
|
|
530
|
+
if len(roots) == 1:
|
|
531
|
+
return next(iter(roots))
|
|
532
|
+
return None
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def edge_data(G: nx.Graph, u: str, v: str) -> dict:
|
|
536
|
+
"""Return one edge attribute dict for (u, v), tolerating MultiGraph.
|
|
537
|
+
|
|
538
|
+
For MultiGraph/MultiDiGraph there can be multiple parallel edges;
|
|
539
|
+
this returns the first one (sufficient for callers that only need
|
|
540
|
+
relation/confidence for rendering). Fixes #796.
|
|
541
|
+
"""
|
|
542
|
+
raw = G[u][v]
|
|
543
|
+
if isinstance(G, (nx.MultiGraph, nx.MultiDiGraph)):
|
|
544
|
+
return next(iter(raw.values()), {})
|
|
545
|
+
return raw
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
def edge_datas(G: nx.Graph, u: str, v: str) -> list[dict]:
|
|
549
|
+
"""Return every edge attribute dict for (u, v); always a list."""
|
|
550
|
+
raw = G[u][v]
|
|
551
|
+
if isinstance(G, (nx.MultiGraph, nx.MultiDiGraph)):
|
|
552
|
+
return list(raw.values())
|
|
553
|
+
return [raw]
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def dedupe_nodes(nodes: list[dict]) -> list[dict]:
|
|
557
|
+
"""Collapse nodes sharing an ``id``, last-writer-wins on attributes.
|
|
558
|
+
|
|
559
|
+
Mirrors what ``build_from_json``'s ``G.add_node`` does implicitly (idempotent;
|
|
560
|
+
a later node overwrites an earlier one's attributes). The ``--no-cluster``
|
|
561
|
+
write path dumps the raw node list without building a graph, so same-id nodes
|
|
562
|
+
— e.g. a Swift ``type=module`` anchor emitted once per importing file (#1327)
|
|
563
|
+
— would otherwise appear as duplicates. Insertion order follows each id's
|
|
564
|
+
first appearance; the retained dict is the last one seen.
|
|
565
|
+
"""
|
|
566
|
+
by_id: dict = {}
|
|
567
|
+
for n in nodes:
|
|
568
|
+
nid = n.get("id")
|
|
569
|
+
if nid is None:
|
|
570
|
+
continue
|
|
571
|
+
by_id[nid] = n
|
|
572
|
+
return list(by_id.values())
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
def dedupe_edges(edges: list[dict]) -> list[dict]:
|
|
576
|
+
"""Collapse exact parallel edges by ``(source, target, relation)``, keeping the
|
|
577
|
+
first occurrence.
|
|
578
|
+
|
|
579
|
+
The clustered build path runs edges through a NetworkX ``DiGraph``, which
|
|
580
|
+
collapses parallel edges automatically. The ``--no-cluster`` and incremental
|
|
581
|
+
``update`` write paths bypass NetworkX and concatenate edge lists raw, so
|
|
582
|
+
duplicates accumulate and edge counts become non-deterministic across build
|
|
583
|
+
modes / repeated updates (#1317). Deduping on the connectivity identity is
|
|
584
|
+
zero-signal-loss and restores idempotency. Callers that intentionally keep
|
|
585
|
+
parallel edges (multigraph output) must not use this.
|
|
586
|
+
"""
|
|
587
|
+
seen: set[tuple] = set()
|
|
588
|
+
out: list[dict] = []
|
|
589
|
+
for e in edges:
|
|
590
|
+
key = (e.get("source"), e.get("target"), e.get("relation"))
|
|
591
|
+
if key in seen:
|
|
592
|
+
continue
|
|
593
|
+
seen.add(key)
|
|
594
|
+
out.append(e)
|
|
595
|
+
return out
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def _old_file_stems(rel: Path) -> list[str]:
|
|
599
|
+
"""Pre-migration stem forms a semantic fragment may have used for ``rel``.
|
|
600
|
+
|
|
601
|
+
Ordered longest-first so prefix stripping is greedy and unambiguous:
|
|
602
|
+
- one-parent form: ``parent.stem`` (the old _file_stem rule, #550-era)
|
|
603
|
+
- zero-parent form: ``stem`` (the old llm.py prompt rule, #1509)
|
|
604
|
+
"""
|
|
605
|
+
forms: list[str] = []
|
|
606
|
+
parent = rel.parent.name
|
|
607
|
+
if parent and parent not in (".", ""):
|
|
608
|
+
forms.append(make_id(f"{parent}.{rel.stem}"))
|
|
609
|
+
forms.append(make_id(rel.stem))
|
|
610
|
+
# Dedupe while preserving order (top-level files collapse both forms).
|
|
611
|
+
seen: set[str] = set()
|
|
612
|
+
return [f for f in forms if f and not (f in seen or seen.add(f))]
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def _semantic_id_remap(nodes: list, root: str | None) -> dict:
|
|
616
|
+
"""Re-derive non-AST node ids from ``source_file`` using the canonical
|
|
617
|
+
full-path stem, so a cached/LLM fragment carrying a pre-migration short id
|
|
618
|
+
reconciles with the AST node instead of spawning a ghost (#1504/#1509).
|
|
619
|
+
|
|
620
|
+
Drift-proof by construction: the new id is computed from ``source_file`` in
|
|
621
|
+
code, never trusted from the fragment's own ``id`` string. AST-origin nodes
|
|
622
|
+
are skipped (they are already canonical via the extract() post-pass)."""
|
|
623
|
+
from graphify.extractors.base import _file_stem # local: avoid import cost at module load
|
|
624
|
+
|
|
625
|
+
remap: dict[str, str] = {}
|
|
626
|
+
for node in nodes:
|
|
627
|
+
if not isinstance(node, dict):
|
|
628
|
+
continue
|
|
629
|
+
if _is_ast_tier(node):
|
|
630
|
+
continue
|
|
631
|
+
nid = node.get("id")
|
|
632
|
+
sf = node.get("source_file")
|
|
633
|
+
if not nid or not isinstance(nid, str) or not sf:
|
|
634
|
+
continue
|
|
635
|
+
sf_norm = _norm_source_file(str(sf), root) or str(sf)
|
|
636
|
+
rel = Path(sf_norm)
|
|
637
|
+
if _is_abs(sf_norm):
|
|
638
|
+
# Can't relativize (no/failed root) — leave the id untouched rather
|
|
639
|
+
# than bake an on-disk path into it. Tested for BOTH platforms: a
|
|
640
|
+
# graph built on Linux/CI carries POSIX-absolute source_files that
|
|
641
|
+
# WindowsPath.is_absolute() calls relative, which leaked the whole
|
|
642
|
+
# build directory into node IDs when updated on Windows (#2618).
|
|
643
|
+
continue
|
|
644
|
+
if not rel.name:
|
|
645
|
+
# source_file equals the scan root, so _norm_source_file relativized it
|
|
646
|
+
# to Path('.') — a project-level node with no per-file identity to remap.
|
|
647
|
+
# Leave its id untouched (and avoid _file_stem's empty-name crash, #1618).
|
|
648
|
+
continue
|
|
649
|
+
new_stem = make_id(_file_stem(rel))
|
|
650
|
+
if not new_stem:
|
|
651
|
+
continue
|
|
652
|
+
norm_nid = _normalize_id(nid)
|
|
653
|
+
# Idempotency guard (#1917): an id already carrying its canonical stem is
|
|
654
|
+
# done — do not re-run the legacy branch on it. When the canonical stem
|
|
655
|
+
# contains a shorter legacy stem as a prefix (parent dir name == file
|
|
656
|
+
# stem, e.g. `.claude/CLAUDE.md` -> `claude_claude` over legacy `claude`),
|
|
657
|
+
# an already-migrated id like `claude_claude_x` still matches the legacy
|
|
658
|
+
# `claude_` prefix below and would gain another stem segment on every
|
|
659
|
+
# build, defeating the same_topology/no_change short-circuits. Mirrors the
|
|
660
|
+
# canonical check in graph_has_legacy_ids.
|
|
661
|
+
if norm_nid == new_stem or norm_nid.startswith(new_stem + "_"):
|
|
662
|
+
continue
|
|
663
|
+
new_id: str | None = None
|
|
664
|
+
old_forms = _old_file_stems(rel)
|
|
665
|
+
# #2197: on Windows, detect() can emit an ABSOLUTE source_file, and a
|
|
666
|
+
# semantic fragment's id derived from that absolute path (e.g.
|
|
667
|
+
# d_projects_myrepo_docs_dataflow) matches neither the canonical
|
|
668
|
+
# relative stem nor the legacy short forms above — so while source_file
|
|
669
|
+
# itself is healed by _norm_source_file, the id would ghost against the
|
|
670
|
+
# existing graph's docs_dataflow. When the raw path was absolute and
|
|
671
|
+
# relativized under root, treat the raw-absolute stem as one more
|
|
672
|
+
# old-stem form — the semantic-side twin of extract.py's absolute-form
|
|
673
|
+
# id registration. It is the longest form, so it goes first (greedy
|
|
674
|
+
# prefix stripping, same ordering rule as _old_file_stems).
|
|
675
|
+
sf_raw = str(sf).replace("\\", "/")
|
|
676
|
+
if sf_raw != sf_norm and _is_abs(sf_raw):
|
|
677
|
+
abs_stem = make_id(_file_stem(Path(sf_raw)))
|
|
678
|
+
if abs_stem and abs_stem != new_stem and abs_stem not in old_forms:
|
|
679
|
+
old_forms.insert(0, abs_stem)
|
|
680
|
+
for old_stem in old_forms:
|
|
681
|
+
if old_stem == new_stem:
|
|
682
|
+
continue # already canonical for this form
|
|
683
|
+
if norm_nid == old_stem:
|
|
684
|
+
new_id = new_stem # the file node itself
|
|
685
|
+
break
|
|
686
|
+
prefix = old_stem + "_"
|
|
687
|
+
if norm_nid.startswith(prefix):
|
|
688
|
+
entity = norm_nid[len(prefix):]
|
|
689
|
+
new_id = make_id(new_stem, entity)
|
|
690
|
+
break
|
|
691
|
+
if new_id and new_id != nid:
|
|
692
|
+
remap[nid] = new_id
|
|
693
|
+
return remap
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
# MCP node kinds whose ID is GLOBAL by design — deliberately shared across every
|
|
697
|
+
# config file that mentions them (`mcp_command_npx`, `mcp_package_...`,
|
|
698
|
+
# `env_var_...`), so it is not derived from ``source_file`` at all (#2408). The
|
|
699
|
+
# file-scoped kinds (`mcp_config_file`, `mcp_server`) ARE stem-derived and stay
|
|
700
|
+
# subject to legacy detection.
|
|
701
|
+
_MCP_GLOBAL_ID_KINDS = frozenset({"mcp_command", "mcp_package", "env_var"})
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def _has_global_id(node: dict) -> bool:
|
|
705
|
+
"""Whether ``node``'s ID is global by construction rather than file-derived."""
|
|
706
|
+
meta = node.get("metadata")
|
|
707
|
+
if not isinstance(meta, dict):
|
|
708
|
+
return False
|
|
709
|
+
return meta.get("mcp_kind") in _MCP_GLOBAL_ID_KINDS
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def graph_has_legacy_ids(nodes: list, root: str | Path | None = None, sample: int = 300) -> bool:
|
|
713
|
+
"""Whether a loaded graph still uses pre-#1504 node IDs (parent-dir / filename
|
|
714
|
+
stem) rather than the full repo-relative path. Read-only consumers (query,
|
|
715
|
+
serve) use this to nudge the user to rebuild, since they don't re-extract.
|
|
716
|
+
|
|
717
|
+
Heuristic and cheap: only **file-level** nodes (source_location ``L1``) are
|
|
718
|
+
inspected, because their ID is unambiguously the file stem. Symbol nodes are
|
|
719
|
+
skipped — some extractors scope a symbol by package/directory (Go's
|
|
720
|
+
``_make_id(pkg_dir, name)`` → ``sub_thing``), which can coincide with an old
|
|
721
|
+
file-stem form and would otherwise false-positive. Nodes whose ID is global by
|
|
722
|
+
construction (see ``_MCP_GLOBAL_ID_KINDS``) are skipped for the same reason.
|
|
723
|
+
Returns True as soon as one file node's ID matches an OLD stem form but not the
|
|
724
|
+
canonical full-path form."""
|
|
725
|
+
from graphify.extractors.base import _file_stem
|
|
726
|
+
_r = str(root) if root is not None else None
|
|
727
|
+
checked = 0
|
|
728
|
+
for node in nodes:
|
|
729
|
+
if not isinstance(node, dict):
|
|
730
|
+
continue
|
|
731
|
+
if str(node.get("source_location") or "") != "L1":
|
|
732
|
+
continue # only file-level nodes carry an unambiguous file-stem ID
|
|
733
|
+
if _has_global_id(node):
|
|
734
|
+
# #2408: MCP ingest stamps every node it emits with line 1 (JSON has no
|
|
735
|
+
# line info), so globally-scoped nodes slip past the L1 proxy for
|
|
736
|
+
# "file-level". For `sub/.mcp.json` the old bare stem is `mcp` while the
|
|
737
|
+
# canonical stem is `sub_mcp`, so a perfectly valid `mcp_command_npx`
|
|
738
|
+
# reads as a legacy `mcp_`-prefixed id and warns on every fresh build.
|
|
739
|
+
# (A root-level `.mcp.json` never tripped it: there `mcp` IS canonical.)
|
|
740
|
+
continue
|
|
741
|
+
nid = node.get("id")
|
|
742
|
+
sf = node.get("source_file")
|
|
743
|
+
if not nid or not isinstance(nid, str) or not sf:
|
|
744
|
+
continue
|
|
745
|
+
sf_norm = _norm_source_file(str(sf), _r) or str(sf)
|
|
746
|
+
rel = Path(sf_norm)
|
|
747
|
+
if _is_abs(sf_norm):
|
|
748
|
+
continue
|
|
749
|
+
if not rel.name:
|
|
750
|
+
continue # source_file == scan root -> Path('.'), no file stem (#1618)
|
|
751
|
+
new_stem = make_id(_file_stem(rel))
|
|
752
|
+
if not new_stem:
|
|
753
|
+
continue
|
|
754
|
+
norm = _normalize_id(nid)
|
|
755
|
+
if norm == new_stem or norm.startswith(new_stem + "_"):
|
|
756
|
+
checked += 1
|
|
757
|
+
else:
|
|
758
|
+
for old in _old_file_stems(rel):
|
|
759
|
+
if old != new_stem and (norm == old or norm.startswith(old + "_")):
|
|
760
|
+
return True
|
|
761
|
+
checked += 1
|
|
762
|
+
if checked >= sample:
|
|
763
|
+
break
|
|
764
|
+
return False
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
def _doc_twin_remap(nodes: list) -> dict[str, str]:
|
|
768
|
+
"""Map a markdown quick-scan's bare doc node ``<slug>`` to the semantic
|
|
769
|
+
``<slug>_doc`` node for the SAME file (#1799).
|
|
770
|
+
|
|
771
|
+
The markdown quick-scan (``extract_markdown``) mints a file node with the
|
|
772
|
+
bare id ``_make_id(path)`` while the semantic pass mints ``<slug>_doc`` for
|
|
773
|
+
the same document. A ``graphify update`` after a semantic build leaves both,
|
|
774
|
+
splitting the file's edges across two disconnected nodes. Canonicalize to the
|
|
775
|
+
semantic ``_doc`` node (it carries the richer references/hyperedges). Gated to
|
|
776
|
+
``file_type == "document"`` on BOTH twins with an identical ``source_file``,
|
|
777
|
+
so an unrelated code symbol ``foo`` and ``foo_doc`` never merge.
|
|
778
|
+
"""
|
|
779
|
+
by_id: dict[str, dict] = {}
|
|
780
|
+
for n in nodes:
|
|
781
|
+
if isinstance(n, dict) and n.get("id"):
|
|
782
|
+
by_id[str(n["id"])] = n
|
|
783
|
+
remap: dict[str, str] = {}
|
|
784
|
+
for nid, node in by_id.items():
|
|
785
|
+
if not nid.endswith("_doc"):
|
|
786
|
+
continue
|
|
787
|
+
bare = by_id.get(nid[:-4])
|
|
788
|
+
if bare is None:
|
|
789
|
+
continue
|
|
790
|
+
sf = node.get("source_file")
|
|
791
|
+
if not sf or bare.get("source_file") != sf:
|
|
792
|
+
continue
|
|
793
|
+
if node.get("file_type") != "document" or bare.get("file_type") != "document":
|
|
794
|
+
continue
|
|
795
|
+
remap[nid[:-4]] = nid
|
|
796
|
+
return remap
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
def build_from_json(extraction: dict, *, directed: bool = False, root: str | Path | None = None) -> nx.Graph:
|
|
800
|
+
"""Build a NetworkX graph from an extraction dict.
|
|
801
|
+
|
|
802
|
+
directed=True produces a DiGraph that preserves edge direction (source→target).
|
|
803
|
+
directed=False (default) produces an undirected Graph for backward compatibility.
|
|
804
|
+
root: if given, absolute source_file paths from semantic subagents are made
|
|
805
|
+
relative to root so all nodes share a consistent path key (#932).
|
|
806
|
+
"""
|
|
807
|
+
_root = str(Path(root).resolve()) if root else None
|
|
808
|
+
# NetworkX <= 3.1 serialised edges as "links"; remap to "edges" for compatibility.
|
|
809
|
+
if "edges" not in extraction and "links" in extraction:
|
|
810
|
+
extraction = dict(extraction, edges=extraction["links"])
|
|
811
|
+
|
|
812
|
+
# Hyperedge persistence is dual-slot (#2485): to_json writes BOTH a
|
|
813
|
+
# top-level `hyperedges` key AND the nested `graph.hyperedges` (node_link
|
|
814
|
+
# graph attrs), but node_link_data-only writers emit just the nested slot.
|
|
815
|
+
# Fold the nested slot onto the top-level key ONCE, so every downstream
|
|
816
|
+
# pass (_coerce_non_string_ids, _normalize_hyperedge_members, the member
|
|
817
|
+
# revalidation before G.graph["hyperedges"] is set) reads one location.
|
|
818
|
+
if "hyperedges" not in extraction and isinstance(
|
|
819
|
+
(extraction.get("graph") or {}).get("hyperedges"), list
|
|
820
|
+
):
|
|
821
|
+
extraction = dict(extraction, hyperedges=extraction["graph"]["hyperedges"])
|
|
822
|
+
|
|
823
|
+
# Numeric ids from a loose backend become str before anything keys on them
|
|
824
|
+
# (#2326) — after the links remap so aliased edges are covered too.
|
|
825
|
+
_coerce_non_string_ids(extraction)
|
|
826
|
+
|
|
827
|
+
# Canonicalize legacy node/edge schema before validation.
|
|
828
|
+
for node in extraction.get("nodes", []):
|
|
829
|
+
if not isinstance(node, dict):
|
|
830
|
+
continue
|
|
831
|
+
if "source" in node and "source_file" not in node:
|
|
832
|
+
# Count edges that reference this node so the warning is actionable (#479)
|
|
833
|
+
node_id = node.get("id", "?")
|
|
834
|
+
affected_edges = sum(
|
|
835
|
+
1 for e in extraction.get("edges", [])
|
|
836
|
+
if e.get("source") == node_id or e.get("target") == node_id
|
|
837
|
+
)
|
|
838
|
+
print(
|
|
839
|
+
f"[graphify] WARNING: node '{node_id}' uses field 'source' instead of "
|
|
840
|
+
f"'source_file' — {affected_edges} edge(s) may be misrouted. "
|
|
841
|
+
f"Rename the field to 'source_file' to silence this warning.",
|
|
842
|
+
file=sys.stderr,
|
|
843
|
+
)
|
|
844
|
+
node["source_file"] = node.pop("source")
|
|
845
|
+
# Fold the remaining legacy node aliases (`name`->`label`,
|
|
846
|
+
# `path`->`source_file`, #2194) before validation and before the
|
|
847
|
+
# semantic-rekey / ghost-merge passes below, all of which key on
|
|
848
|
+
# label/source_file and would otherwise skip the node entirely.
|
|
849
|
+
_fold_node_aliases(node)
|
|
850
|
+
# Default missing/None file_type to "concept" so legacy graph.json
|
|
851
|
+
# entries (and stub nodes preserved by `_rebuild_code` from older
|
|
852
|
+
# graphify versions that didn't always populate file_type) don't
|
|
853
|
+
# trigger spurious "invalid file_type 'None'" validator warnings (#660).
|
|
854
|
+
if node.get("file_type") in (None, ""):
|
|
855
|
+
node["file_type"] = "concept"
|
|
856
|
+
ft = node.get("file_type", "")
|
|
857
|
+
if ft and ft not in {"code", "document", "paper", "image", "rationale", "concept"}:
|
|
858
|
+
node["file_type"] = _FILE_TYPE_SYNONYMS.get(ft, "concept")
|
|
859
|
+
|
|
860
|
+
# Canonicalize hyperedge member lists (#1561): producers sometimes key the
|
|
861
|
+
# member list `members`/`node_ids` instead of `nodes`. Fold aliases onto
|
|
862
|
+
# `nodes` here — BEFORE validation and the semantic-rekey loop below — so
|
|
863
|
+
# every downstream consumer (rekey, source_file relativize, to_json) reads
|
|
864
|
+
# one canonical key, the same way edge endpoints alias from/to at build.
|
|
865
|
+
for he in extraction.get("hyperedges", []) or []:
|
|
866
|
+
_normalize_hyperedge_members(he)
|
|
867
|
+
|
|
868
|
+
# Fold legacy edge field aliases (`type`->`relation`,
|
|
869
|
+
# `confidence_score`->`confidence`, #2194) BEFORE validation. The existing
|
|
870
|
+
# from/to endpoint fold lives in the edge loop further down, which runs
|
|
871
|
+
# after validate_extraction — too late for fields the validator requires.
|
|
872
|
+
for edge in extraction.get("edges", []):
|
|
873
|
+
if isinstance(edge, dict):
|
|
874
|
+
_fold_edge_aliases(edge)
|
|
875
|
+
|
|
876
|
+
errors = validate_extraction(extraction)
|
|
877
|
+
# Dangling edges (stdlib/external imports) are expected - only warn about real schema errors.
|
|
878
|
+
real_errors = [e for e in errors if "does not match any node id" not in e]
|
|
879
|
+
if real_errors:
|
|
880
|
+
# Break the warning down by cause (#2194): a mixed batch used to surface
|
|
881
|
+
# only real_errors[0], hiding every other failure mode. Group on the
|
|
882
|
+
# "missing required field 'X'" suffix and report per-cause counts plus
|
|
883
|
+
# one example each, so the operator sees the full shape of the damage.
|
|
884
|
+
by_cause: dict[str, list[str]] = {}
|
|
885
|
+
for err in real_errors:
|
|
886
|
+
m = re.search(r"missing required field '[^']*'", err)
|
|
887
|
+
by_cause.setdefault(m.group(0) if m else "other schema issue", []).append(err)
|
|
888
|
+
breakdown = "; ".join(
|
|
889
|
+
f"{len(errs)}x {cause} (e.g. {errs[0]})" for cause, errs in by_cause.items()
|
|
890
|
+
)
|
|
891
|
+
print(
|
|
892
|
+
f"[graphify] Extraction warning ({len(real_errors)} issues): {breakdown}",
|
|
893
|
+
file=sys.stderr,
|
|
894
|
+
)
|
|
895
|
+
# Deterministic semantic re-key (#1504/#1509): the node-ID stem is now the
|
|
896
|
+
# full repo-relative path (docs/v1/api/README.md -> docs_v1_api_readme), but
|
|
897
|
+
# the semantic cache is UNVERSIONED, so a cached/LLM fragment can still carry
|
|
898
|
+
# an OLD short id whose stem was just the immediate parent dir (api_readme),
|
|
899
|
+
# or a prompt-drifting id with zero parent dirs (readme). Rather than trust
|
|
900
|
+
# LLM prose to emit the right stem, we re-derive every non-AST node's id from
|
|
901
|
+
# its own source_file in code, so a drifted fragment physically reconciles
|
|
902
|
+
# with the AST node instead of spawning a ghost / a re-bill. AST-origin nodes
|
|
903
|
+
# already carry canonical ids (the extract() id-remap post-pass guarantees it)
|
|
904
|
+
# and are left untouched.
|
|
905
|
+
_rekey: dict[str, str] = _semantic_id_remap(extraction.get("nodes", []), _root)
|
|
906
|
+
if _rekey:
|
|
907
|
+
for node in extraction.get("nodes", []):
|
|
908
|
+
if isinstance(node, dict) and node.get("id") in _rekey:
|
|
909
|
+
node["id"] = _rekey[node["id"]]
|
|
910
|
+
for edge in extraction.get("edges", []):
|
|
911
|
+
if not isinstance(edge, dict):
|
|
912
|
+
continue
|
|
913
|
+
if edge.get("source") in _rekey:
|
|
914
|
+
edge["source"] = _rekey[edge["source"]]
|
|
915
|
+
if edge.get("target") in _rekey:
|
|
916
|
+
edge["target"] = _rekey[edge["target"]]
|
|
917
|
+
for he in extraction.get("hyperedges", []) or []:
|
|
918
|
+
if isinstance(he, dict) and isinstance(he.get("nodes"), list):
|
|
919
|
+
# Guard on hashability (#2486): _normalize_hyperedge_members
|
|
920
|
+
# has already coerced members above, but a still-unhashable ref
|
|
921
|
+
# must pass through rather than abort the merge on dict.get.
|
|
922
|
+
he["nodes"] = [
|
|
923
|
+
_rekey.get(n, n) if _hashable(n) else n for n in he["nodes"]
|
|
924
|
+
]
|
|
925
|
+
|
|
926
|
+
# Merge markdown quick-scan bare doc nodes into their semantic `_doc` twin
|
|
927
|
+
# for the same file, so a document is one node regardless of which pipeline
|
|
928
|
+
# touched it last (#1799).
|
|
929
|
+
_doc_remap = _doc_twin_remap(extraction.get("nodes", []))
|
|
930
|
+
if _doc_remap:
|
|
931
|
+
extraction["nodes"] = [
|
|
932
|
+
n for n in extraction.get("nodes", [])
|
|
933
|
+
if not (isinstance(n, dict) and n.get("id") in _doc_remap)
|
|
934
|
+
]
|
|
935
|
+
_new_edges = []
|
|
936
|
+
for edge in extraction.get("edges", []):
|
|
937
|
+
if isinstance(edge, dict):
|
|
938
|
+
s0, t0 = edge.get("source"), edge.get("target")
|
|
939
|
+
if s0 in _doc_remap:
|
|
940
|
+
edge["source"] = _doc_remap[s0]
|
|
941
|
+
if t0 in _doc_remap:
|
|
942
|
+
edge["target"] = _doc_remap[t0]
|
|
943
|
+
# Drop only self-loops the remap itself collapsed (a bare->_doc
|
|
944
|
+
# link becoming doc->doc); leave any pre-existing self-loop alone.
|
|
945
|
+
if edge.get("source") == edge.get("target") and (s0 in _doc_remap or t0 in _doc_remap):
|
|
946
|
+
continue
|
|
947
|
+
_new_edges.append(edge)
|
|
948
|
+
extraction["edges"] = _new_edges
|
|
949
|
+
for he in extraction.get("hyperedges", []) or []:
|
|
950
|
+
if isinstance(he, dict) and isinstance(he.get("nodes"), list):
|
|
951
|
+
# Same hashability guard as the _rekey pass above (#2486).
|
|
952
|
+
he["nodes"] = [
|
|
953
|
+
_doc_remap.get(n, n) if _hashable(n) else n for n in he["nodes"]
|
|
954
|
+
]
|
|
955
|
+
|
|
956
|
+
G: nx.Graph = nx.DiGraph() if directed else nx.Graph()
|
|
957
|
+
for node in extraction.get("nodes", []):
|
|
958
|
+
# Skip dict nodes with a missing or non-hashable id (e.g. a list emitted
|
|
959
|
+
# by a buggy LLM extraction) so NetworkX add_node never raises
|
|
960
|
+
# TypeError: unhashable type. Non-dict nodes are deliberately left to
|
|
961
|
+
# raise as before, so callers that probe build for shape errors (e.g.
|
|
962
|
+
# the multigraph diagnostic) still observe the malformed shape.
|
|
963
|
+
if isinstance(node, dict):
|
|
964
|
+
if "id" not in node:
|
|
965
|
+
continue
|
|
966
|
+
try:
|
|
967
|
+
hash(node["id"])
|
|
968
|
+
except TypeError:
|
|
969
|
+
print(
|
|
970
|
+
f"[graphify] WARNING: skipping node with non-hashable id "
|
|
971
|
+
f"{node['id']!r} (must be a string).",
|
|
972
|
+
file=sys.stderr,
|
|
973
|
+
)
|
|
974
|
+
continue
|
|
975
|
+
if "source_file" in node:
|
|
976
|
+
node["source_file"] = _norm_source_file(node["source_file"], _root)
|
|
977
|
+
# definition_file names a file inside the scanned tree exactly like
|
|
978
|
+
# source_file (the #2990 decl/def merge stamps it from the impl's
|
|
979
|
+
# source_file BEFORE this normalization runs), so it must be made
|
|
980
|
+
# portable the same way - it used to ship absolute, leaking the
|
|
981
|
+
# build host's layout into graph.json and MCP get_node (#3223).
|
|
982
|
+
if "definition_file" in node:
|
|
983
|
+
node["definition_file"] = _norm_source_file(node["definition_file"], _root)
|
|
984
|
+
G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"})
|
|
985
|
+
node_set = set(G.nodes())
|
|
986
|
+
|
|
987
|
+
# #1145 (extended): merge LLM ghost-duplicate nodes into AST canonical nodes.
|
|
988
|
+
# Original bug: AST uses parent-qualified IDs (mingpt_bpe_get_pairs) while LLM
|
|
989
|
+
# uses bare-stem IDs (bpe_get_pairs) — different IDs, same symbol.
|
|
990
|
+
# Original fix only caught LLM nodes with source_location=None; LLM now
|
|
991
|
+
# populates source_location, so those ghosts survived. Extended fix: use
|
|
992
|
+
# _origin=="ast" as the canonical signal. AST nodes always win; any non-AST
|
|
993
|
+
# node sharing (basename, label) with an AST node is a ghost.
|
|
994
|
+
_loc_nodes: dict[tuple[str, str], str] = {} # (source_file, label) -> canonical node id
|
|
995
|
+
_loc_collisions: set[tuple[str, str]] = set() # keys shared by 2+ AST nodes
|
|
996
|
+
_noloc_nodes: dict[tuple[str, str], str] = {} # (source_file, label) -> ghost node id
|
|
997
|
+
_ast_file_nodes: list[tuple[str, str]] = [] # (node_id, source_file) for AST file-self nodes (#3344)
|
|
998
|
+
|
|
999
|
+
# Pass 1: collect canonical nodes — AST-origin nodes take precedence over LLM nodes.
|
|
1000
|
+
# When 2+ AST nodes share a key (same-named symbols in same-named files across
|
|
1001
|
+
# directories, e.g. render in two index.ts), the key is ambiguous: merging a
|
|
1002
|
+
# ghost would pick an arbitrary winner via set-iteration order (#1257). Track
|
|
1003
|
+
# those keys so Pass 2 skips them — same conservatism as
|
|
1004
|
+
# _rewire_unique_stub_nodes, which only merges when exactly one real def exists.
|
|
1005
|
+
# Iterate in a deterministic (sorted) order, not set-iteration order, so the
|
|
1006
|
+
# canonical winner and the ambiguity decisions below don't flip run-to-run
|
|
1007
|
+
# with CPython's per-process string-hash seed (#1753) — the same reason the
|
|
1008
|
+
# edge-iteration loop further down sorts on purpose.
|
|
1009
|
+
for nid in sorted(node_set):
|
|
1010
|
+
attrs = G.nodes[nid]
|
|
1011
|
+
label = str(attrs.get("label", "")).strip()
|
|
1012
|
+
sf = str(attrs.get("source_file", ""))
|
|
1013
|
+
if not label or not sf:
|
|
1014
|
+
continue
|
|
1015
|
+
# Strict _origin check on purpose — NOT _is_ast_tier (#2334): existing
|
|
1016
|
+
# graph items are backfilled with _origin at load time, so inside a
|
|
1017
|
+
# build the only unstamped items are fresh SEMANTIC chunks (extract()
|
|
1018
|
+
# always stamps AST output). Those may carry drifted 'L<line>'
|
|
1019
|
+
# source_locations (the very ghosts #1145-extended collapses), and the
|
|
1020
|
+
# shape fallback would misread them as AST — turning two same-file LLM
|
|
1021
|
+
# duplicates into a fake AST/AST collision that blocks their merge.
|
|
1022
|
+
is_ast = attrs.get("_origin") == "ast"
|
|
1023
|
+
if attrs.get("source_location") or is_ast:
|
|
1024
|
+
# Key on the FULL normalized source_file, not the bare basename
|
|
1025
|
+
# (#2068): the AST/LLM ghost twins of #1145 always share the same
|
|
1026
|
+
# source_file (different ids, same file), so full-path keying still
|
|
1027
|
+
# collapses them, while unrelated same-basename nodes in DIFFERENT
|
|
1028
|
+
# directories (docs/a/index.md vs docs/b/index.md) now get distinct
|
|
1029
|
+
# keys and are never falsely merged. This subsumes the #1753/#1257
|
|
1030
|
+
# cross-file ambiguity guard, which is why the non-AST branch below
|
|
1031
|
+
# no longer needs it.
|
|
1032
|
+
key = (sf, label)
|
|
1033
|
+
if is_ast:
|
|
1034
|
+
# Two AST nodes on the same key (same file, same label) is an
|
|
1035
|
+
# ambiguous collision.
|
|
1036
|
+
if key in _loc_nodes and G.nodes[_loc_nodes[key]].get("_origin") == "ast":
|
|
1037
|
+
_loc_collisions.add(key)
|
|
1038
|
+
# AST-origin nodes always overwrite a prior non-AST entry.
|
|
1039
|
+
_loc_nodes[key] = nid
|
|
1040
|
+
# #3344: a self-referential semantic pass (e.g. re-extracting a
|
|
1041
|
+
# saved graphify-out/memory/*.md query answer) mints a NEW
|
|
1042
|
+
# non-AST node for every bare file path/basename it mentions in
|
|
1043
|
+
# prose ("App.tsx", "customer-app/index.ts"), stamped with
|
|
1044
|
+
# source_file = the memory doc being read, NOT the file named in
|
|
1045
|
+
# the prose. That wrong source_file means such a ghost can never
|
|
1046
|
+
# hit the (sf, label) key above — same underlying bug as #1145,
|
|
1047
|
+
# just with the (source_file, label) *pair* broken instead of
|
|
1048
|
+
# only the id. Record every AST node whose own label already
|
|
1049
|
+
# names its own file (_is_file_node_label — the same "is this a
|
|
1050
|
+
# file node" predicate the label-disambiguation pass uses) so
|
|
1051
|
+
# Pass 2b below can catch these by label alone.
|
|
1052
|
+
if _is_file_node_label(label, sf):
|
|
1053
|
+
_ast_file_nodes.append((nid, sf))
|
|
1054
|
+
else:
|
|
1055
|
+
# First non-AST node for this (file, label) wins as canonical; a
|
|
1056
|
+
# later same-key node is a genuine same-file duplicate and still
|
|
1057
|
+
# collapses in Pass 2.
|
|
1058
|
+
_loc_nodes.setdefault(key, nid)
|
|
1059
|
+
|
|
1060
|
+
# Pass 2: find ghosts — non-AST nodes that have an AST canonical twin.
|
|
1061
|
+
for nid in sorted(node_set):
|
|
1062
|
+
attrs = G.nodes[nid]
|
|
1063
|
+
if attrs.get("_origin") == "ast":
|
|
1064
|
+
continue # AST nodes are never ghosts (strict check — see Pass 1)
|
|
1065
|
+
label = str(attrs.get("label", "")).strip()
|
|
1066
|
+
sf = str(attrs.get("source_file", ""))
|
|
1067
|
+
if not label or not sf:
|
|
1068
|
+
continue
|
|
1069
|
+
key = (sf, label)
|
|
1070
|
+
if key in _loc_collisions:
|
|
1071
|
+
continue # ambiguous key: no safe canonical winner, leave ghost intact
|
|
1072
|
+
if key in _loc_nodes and _loc_nodes[key] != nid:
|
|
1073
|
+
_noloc_nodes[key] = nid
|
|
1074
|
+
# For every ghost that has an AST counterpart, record a remap.
|
|
1075
|
+
_ghost_remap: dict[str, str] = {} # ghost_id -> canonical_id
|
|
1076
|
+
for key, sem_id in _noloc_nodes.items():
|
|
1077
|
+
ast_id = _loc_nodes.get(key)
|
|
1078
|
+
if ast_id is not None:
|
|
1079
|
+
_ghost_remap[sem_id] = ast_id
|
|
1080
|
+
|
|
1081
|
+
# Pass 2b (#3344): catch ghosts the (source_file, label) key above cannot,
|
|
1082
|
+
# because their source_file is simply wrong — a semantic pass over a
|
|
1083
|
+
# document that only *mentions* a file (a saved graphify-out/memory/*.md
|
|
1084
|
+
# query answer, a README, an ADR) stamps the file's own name as a new
|
|
1085
|
+
# node's label but the DOCUMENT's path as source_file, since it has no way
|
|
1086
|
+
# to know the mentioned file's real path. Resolve these by label alone
|
|
1087
|
+
# against every AST file-self node collected in Pass 1, reusing
|
|
1088
|
+
# _is_file_node_label so "App.tsx" matches source_file ".../App.tsx" and
|
|
1089
|
+
# "customer-app/index.ts" matches ".../apps/customer-app/index.ts" (a
|
|
1090
|
+
# directory-qualified suffix, e.g. the leading "apps/" the prose dropped).
|
|
1091
|
+
# Conservative by construction: a label matching 0 or 2+ AST files is left
|
|
1092
|
+
# alone (0 = no known file, 2+ = genuinely ambiguous — same "no safe
|
|
1093
|
+
# canonical winner" rule Pass 2's _loc_collisions already applies).
|
|
1094
|
+
if _ast_file_nodes:
|
|
1095
|
+
for nid in sorted(node_set):
|
|
1096
|
+
if nid in _ghost_remap:
|
|
1097
|
+
continue # already resolved by the exact (sf, label) key
|
|
1098
|
+
attrs = G.nodes[nid]
|
|
1099
|
+
if attrs.get("_origin") == "ast":
|
|
1100
|
+
continue
|
|
1101
|
+
label = str(attrs.get("label", "")).strip()
|
|
1102
|
+
if not label:
|
|
1103
|
+
continue
|
|
1104
|
+
matches = {
|
|
1105
|
+
ast_id for ast_id, ast_sf in _ast_file_nodes
|
|
1106
|
+
if _is_file_node_label(label, ast_sf)
|
|
1107
|
+
}
|
|
1108
|
+
if len(matches) == 1:
|
|
1109
|
+
_ghost_remap[nid] = next(iter(matches))
|
|
1110
|
+
|
|
1111
|
+
# Remove ghost nodes from the graph; edges will be re-pointed via norm_to_id.
|
|
1112
|
+
for ghost_id in _ghost_remap:
|
|
1113
|
+
G.remove_node(ghost_id)
|
|
1114
|
+
node_set.discard(ghost_id)
|
|
1115
|
+
|
|
1116
|
+
# Normalized ID map: lets edges survive when the LLM generates IDs with
|
|
1117
|
+
# slightly different casing or punctuation than the AST extractor.
|
|
1118
|
+
# e.g. "Session_ValidateToken" maps to "session_validatetoken".
|
|
1119
|
+
norm_to_id: dict[str, str] = {_normalize_id(nid): nid for nid in node_set}
|
|
1120
|
+
# Also map ghost IDs to their canonical AST replacements.
|
|
1121
|
+
for ghost_id, canonical_id in _ghost_remap.items():
|
|
1122
|
+
norm_to_id[_normalize_id(ghost_id)] = canonical_id
|
|
1123
|
+
norm_to_id[ghost_id] = canonical_id
|
|
1124
|
+
# Pre-migration alias index (#1504): register each canonical node's OLD-stem id
|
|
1125
|
+
# forms as aliases so a stale-id edge endpoint coming from an un-re-keyed
|
|
1126
|
+
# fragment (e.g. an incremental update whose fragment references a symbol in a
|
|
1127
|
+
# file that was NOT re-extracted) still resolves to the migrated node instead
|
|
1128
|
+
# of dangling. Only fills gaps — never overrides a real node id.
|
|
1129
|
+
#
|
|
1130
|
+
# The old-stem form drops the extension and (for the file node itself) every
|
|
1131
|
+
# directory but the immediate parent, so it collapses easily: "ping.h" and
|
|
1132
|
+
# "ping.php" in different directories both alias to bare "ping". Collecting
|
|
1133
|
+
# every candidate for an alias BEFORE committing any of them — and only
|
|
1134
|
+
# committing when exactly one candidate claims it — keeps this a precise
|
|
1135
|
+
# re-keying aid instead of a silent cross-file (and cross-language) merge.
|
|
1136
|
+
# Without this, a dangling edge to a bare, deliberately-unscoped fallback id
|
|
1137
|
+
# (e.g. the C/C++ extractor's last-resort target for an #include it couldn't
|
|
1138
|
+
# resolve to a real path) could ride this alias onto whichever unrelated
|
|
1139
|
+
# same-stem file happened to be inserted first into ``node_set`` — a Python
|
|
1140
|
+
# set, so "first" is hash-order, not anything meaningful.
|
|
1141
|
+
#
|
|
1142
|
+
# A file node's OWN id is not always a clean ``new_stem`` prefix: when a
|
|
1143
|
+
# same-directory ``.h``/``.cpp`` pair collides on their shared pre-extension
|
|
1144
|
+
# id, _disambiguate_colliding_node_ids salts both apart into ids like
|
|
1145
|
+
# ``tools_aolserver_utility_h_tools_aolserver_utility`` — which no longer
|
|
1146
|
+
# string-prefixes cleanly for the suffix math below. Detecting "this IS the
|
|
1147
|
+
# file node" by label (every file node's label is its own basename,
|
|
1148
|
+
# regardless of id mangling) instead of by id shape keeps a salted file node
|
|
1149
|
+
# in the alias competition, so a genuine collision (a C header AND an
|
|
1150
|
+
# unrelated same-named PHP script) is still caught as ambiguous instead of
|
|
1151
|
+
# the header silently dropping out of the race and leaving the PHP file as
|
|
1152
|
+
# the lone (wrong) "unambiguous" winner.
|
|
1153
|
+
from graphify.extractors.base import _file_stem as _fs
|
|
1154
|
+
_alias_candidates: dict[str, set[str]] = {}
|
|
1155
|
+
for nid in node_set:
|
|
1156
|
+
attrs = G.nodes[nid]
|
|
1157
|
+
sf = attrs.get("source_file")
|
|
1158
|
+
if not sf:
|
|
1159
|
+
continue
|
|
1160
|
+
rel = Path(str(sf))
|
|
1161
|
+
if _is_abs(str(sf)):
|
|
1162
|
+
continue
|
|
1163
|
+
new_stem = make_id(_fs(rel))
|
|
1164
|
+
if str(attrs.get("label", "")) == rel.name:
|
|
1165
|
+
suffix = "" # this node IS the file, whatever its (possibly salted) id
|
|
1166
|
+
else:
|
|
1167
|
+
suffix = ""
|
|
1168
|
+
if _normalize_id(nid).startswith(new_stem):
|
|
1169
|
+
suffix = _normalize_id(nid)[len(new_stem):] # leading "_entity" or ""
|
|
1170
|
+
for old_stem in _old_file_stems(rel):
|
|
1171
|
+
if old_stem == new_stem:
|
|
1172
|
+
continue
|
|
1173
|
+
alias = old_stem + suffix
|
|
1174
|
+
_alias_candidates.setdefault(_normalize_id(alias), set()).add(nid)
|
|
1175
|
+
_alias_candidates.setdefault(alias, set()).add(nid)
|
|
1176
|
+
for alias_key, candidates in _alias_candidates.items():
|
|
1177
|
+
if len(candidates) == 1:
|
|
1178
|
+
norm_to_id.setdefault(alias_key, next(iter(candidates)))
|
|
1179
|
+
# Iterate edges in a deterministic order. The graph is undirected and stores
|
|
1180
|
+
# direction in _src/_tgt; when two edges collapse onto the same node pair the
|
|
1181
|
+
# last write wins, so an unstable iteration order flips _src/_tgt run-to-run
|
|
1182
|
+
# and makes the serialized graph churn. Sorting fixes the last-write outcome.
|
|
1183
|
+
for edge in sorted(
|
|
1184
|
+
extraction.get("edges", []),
|
|
1185
|
+
key=lambda e: (
|
|
1186
|
+
str(e.get("source", e.get("from", ""))),
|
|
1187
|
+
str(e.get("target", e.get("to", ""))),
|
|
1188
|
+
str(e.get("relation", "")),
|
|
1189
|
+
),
|
|
1190
|
+
):
|
|
1191
|
+
if "source" not in edge and "from" in edge:
|
|
1192
|
+
edge["source"] = edge["from"]
|
|
1193
|
+
if "target" not in edge and "to" in edge:
|
|
1194
|
+
edge["target"] = edge["to"]
|
|
1195
|
+
if "source" not in edge or "target" not in edge:
|
|
1196
|
+
continue
|
|
1197
|
+
src, tgt = edge["source"], edge["target"]
|
|
1198
|
+
# Skip edges with non-hashable endpoints (e.g. a list emitted by a buggy
|
|
1199
|
+
# LLM extraction) so the `not in node_set` membership test below never
|
|
1200
|
+
# raises TypeError: unhashable type. The validator already reported these.
|
|
1201
|
+
try:
|
|
1202
|
+
hash(src)
|
|
1203
|
+
hash(tgt)
|
|
1204
|
+
except TypeError:
|
|
1205
|
+
print(
|
|
1206
|
+
f"[graphify] WARNING: skipping edge with non-hashable endpoint "
|
|
1207
|
+
f"(source={src!r}, target={tgt!r}).",
|
|
1208
|
+
file=sys.stderr,
|
|
1209
|
+
)
|
|
1210
|
+
continue
|
|
1211
|
+
# Remap mismatched IDs via normalization before dropping the edge.
|
|
1212
|
+
if src not in node_set:
|
|
1213
|
+
src = norm_to_id.get(_normalize_id(src), src)
|
|
1214
|
+
if tgt not in node_set:
|
|
1215
|
+
tgt = norm_to_id.get(_normalize_id(tgt), tgt)
|
|
1216
|
+
if src not in node_set or tgt not in node_set:
|
|
1217
|
+
continue # skip edges to external/stdlib nodes - expected, not an error
|
|
1218
|
+
# `target_file` is a transient import-disambiguation salt hint (#1814)
|
|
1219
|
+
# with no downstream reader; it holds an absolute path, so it must never
|
|
1220
|
+
# be persisted. Disambiguation already pops it off fresh extractions —
|
|
1221
|
+
# dropping it here as well keeps a pre-fix graph's stale absolute hint
|
|
1222
|
+
# from surviving an incremental build_merge, which re-serializes base
|
|
1223
|
+
# edges through here without re-running disambiguation.
|
|
1224
|
+
# `local_alias` is the same shape of transient hint (#2082): it exists only
|
|
1225
|
+
# for the module arm of _resolve_python_member_calls to match an aliased
|
|
1226
|
+
# import receiver, and extract() already drops it once that pass has run.
|
|
1227
|
+
# Dropping it here too covers a stale pre-fix graph re-serialized through
|
|
1228
|
+
# an incremental build_merge, same rationale as target_file above.
|
|
1229
|
+
# Sanitize numeric edge fields (#1960): an explicit ``"weight": null`` in
|
|
1230
|
+
# the extraction JSON survives ``.get("weight", 1.0)`` (the key is present,
|
|
1231
|
+
# so the default never applies) and reaches Louvain/Leiden as None,
|
|
1232
|
+
# crashing modularity arithmetic with a TypeError (graspologic's Leiden
|
|
1233
|
+
# even panics on NaN). Coerce to float and fall back to the schema default
|
|
1234
|
+
# of 1.0 for anything the clustering backends reject — None, non-numeric
|
|
1235
|
+
# strings, NaN/inf, negatives — while numeric strings coerce cleanly.
|
|
1236
|
+
# Repair (not drop) the key so graph.json round-trips a clean value and a
|
|
1237
|
+
# cluster-only/--update reload never re-ingests the null.
|
|
1238
|
+
attrs = {k: v for k, v in edge.items() if k not in ("source", "target", "target_file", "local_alias")}
|
|
1239
|
+
for _num_key in ("weight", "confidence_score"):
|
|
1240
|
+
if _num_key in attrs:
|
|
1241
|
+
try:
|
|
1242
|
+
_num_val = float(attrs[_num_key])
|
|
1243
|
+
except (TypeError, ValueError):
|
|
1244
|
+
_num_val = 1.0
|
|
1245
|
+
if not math.isfinite(_num_val) or _num_val < 0:
|
|
1246
|
+
_num_val = 1.0
|
|
1247
|
+
attrs[_num_key] = _num_val
|
|
1248
|
+
# Backfill source_file from the endpoint nodes (every node carries one).
|
|
1249
|
+
# Semantic/LLM edges occasionally omit it, which downstream validation
|
|
1250
|
+
# flags and leaves query results with no file reference (#1279).
|
|
1251
|
+
if not attrs.get("source_file"):
|
|
1252
|
+
attrs["source_file"] = (
|
|
1253
|
+
G.nodes[src].get("source_file")
|
|
1254
|
+
or G.nodes[tgt].get("source_file")
|
|
1255
|
+
or ""
|
|
1256
|
+
)
|
|
1257
|
+
if "source_file" in attrs:
|
|
1258
|
+
attrs["source_file"] = _norm_source_file(attrs["source_file"], _root)
|
|
1259
|
+
if attrs.get("definition_file"):
|
|
1260
|
+
# Same portability rule as source_file (#3223); heals a graph
|
|
1261
|
+
# written before the fix on its next rebuild.
|
|
1262
|
+
attrs["definition_file"] = _norm_source_file(attrs["definition_file"], _root)
|
|
1263
|
+
# Drop cross-language phantom edges — the same short names (render, parse,
|
|
1264
|
+
# time, ...) recur across language boundaries, so an unresolved target can
|
|
1265
|
+
# bind to a same-named node in another language. The extraction spec forbids
|
|
1266
|
+
# this for `calls`; it is equally invalid for `imports`/`references` (a
|
|
1267
|
+
# Python `import time` must not bind to a `time.ts`, #1749).
|
|
1268
|
+
_edge_rel = attrs.get("relation")
|
|
1269
|
+
if _edge_rel in ("calls", "imports", "imports_from", "references"):
|
|
1270
|
+
src_ext = Path(G.nodes[src].get("source_file") or "").suffix.lower()
|
|
1271
|
+
tgt_ext = Path(G.nodes[tgt].get("source_file") or "").suffix.lower()
|
|
1272
|
+
src_fam = _EDGE_LANG_FAMILY.get(src_ext)
|
|
1273
|
+
tgt_fam = _EDGE_LANG_FAMILY.get(tgt_ext)
|
|
1274
|
+
if _edge_rel == "calls":
|
|
1275
|
+
# Unchanged #1547/#1556 behavior: only INFERRED calls, and drop as
|
|
1276
|
+
# soon as either family differs (an unknown ext counts as different).
|
|
1277
|
+
if (
|
|
1278
|
+
attrs.get("confidence") == "INFERRED"
|
|
1279
|
+
and src_ext and tgt_ext and src_fam != tgt_fam
|
|
1280
|
+
):
|
|
1281
|
+
continue
|
|
1282
|
+
else:
|
|
1283
|
+
# imports/references: drop only when BOTH endpoints are known code
|
|
1284
|
+
# languages of different families, so a config->code reference
|
|
1285
|
+
# (unknown ext, e.g. a manifest) is never mistaken for a phantom.
|
|
1286
|
+
if src_fam is not None and tgt_fam is not None and src_fam != tgt_fam:
|
|
1287
|
+
continue
|
|
1288
|
+
# A file-level import or re-export cannot carry useful connectivity when
|
|
1289
|
+
# both endpoints resolve to the same node. This most often happens when
|
|
1290
|
+
# the target is an unresolved bare module name (``builtins``, ``poseidon``)
|
|
1291
|
+
# that the legacy-ID alias index above mistakes for the importing file's
|
|
1292
|
+
# own old stem. It also covers a nested module importing its parent file:
|
|
1293
|
+
# at file-node granularity that relationship necessarily collapses. Keep
|
|
1294
|
+
# other self-edges, notably recursive ``calls``, because those are real
|
|
1295
|
+
# program structure rather than import-resolution artifacts.
|
|
1296
|
+
if src == tgt and _edge_rel in ("imports", "imports_from", "re_exports"):
|
|
1297
|
+
continue
|
|
1298
|
+
# Preserve original edge direction - undirected graphs lose it otherwise,
|
|
1299
|
+
# causing display functions to show edges backwards.
|
|
1300
|
+
attrs["_src"] = src
|
|
1301
|
+
attrs["_tgt"] = tgt
|
|
1302
|
+
# When the graph is undirected and the same node pair appears twice with
|
|
1303
|
+
# the same relation but opposite directions (e.g. a `calls` b and b `calls` a),
|
|
1304
|
+
# nx.Graph collapses them into one edge. The deterministic sort above means
|
|
1305
|
+
# the lexicographically-later direction would systematically overwrite the
|
|
1306
|
+
# earlier one's _src/_tgt, silently flipping the surviving edge's caller
|
|
1307
|
+
# and callee. First-seen direction wins instead — drop the redundant
|
|
1308
|
+
# reverse-direction duplicate so the original direction is preserved (#1061).
|
|
1309
|
+
if not G.is_directed() and G.has_edge(src, tgt):
|
|
1310
|
+
existing = edge_data(G, src, tgt)
|
|
1311
|
+
if existing.get("relation") == attrs.get("relation") and (
|
|
1312
|
+
existing.get("_src") == tgt and existing.get("_tgt") == src
|
|
1313
|
+
):
|
|
1314
|
+
continue
|
|
1315
|
+
# A pair that already carries a SPECIFIC relation must not be downgraded
|
|
1316
|
+
# to a generic one. Only one edge survives per pair here, and the sort
|
|
1317
|
+
# above orders same-pair edges by relation name, so "last write wins"
|
|
1318
|
+
# resolved the winner alphabetically — which put `references` after
|
|
1319
|
+
# `calls` and `uses` after everything. On graphify's own corpus that
|
|
1320
|
+
# rewrote all 144 pairs where the extraction found both `calls` and
|
|
1321
|
+
# `references` into plain `references`, and callflow's relation filter
|
|
1322
|
+
# does not include `references`, so those call sites left the call graph
|
|
1323
|
+
# entirely. Alphabetical order carries no meaning; keeping the specific
|
|
1324
|
+
# fact does. The reverse (specific arriving after generic) still
|
|
1325
|
+
# overwrites, so the outcome no longer depends on edge order at all.
|
|
1326
|
+
if G.has_edge(src, tgt):
|
|
1327
|
+
existing_rel = edge_data(G, src, tgt).get("relation")
|
|
1328
|
+
if (
|
|
1329
|
+
attrs.get("relation") in _GENERIC_RELATIONS
|
|
1330
|
+
and existing_rel is not None
|
|
1331
|
+
and existing_rel not in _GENERIC_RELATIONS
|
|
1332
|
+
):
|
|
1333
|
+
continue
|
|
1334
|
+
G.add_edge(src, tgt, **attrs)
|
|
1335
|
+
hyperedges = extraction.get("hyperedges", [])
|
|
1336
|
+
if hyperedges:
|
|
1337
|
+
# Relativize hyperedge source_file the same way nodes and edges are
|
|
1338
|
+
# (above), so to_json — which has no root and writes G.graph["hyperedges"]
|
|
1339
|
+
# verbatim — never leaks an absolute path from a semantic subagent (#1418).
|
|
1340
|
+
kept_hyperedges = []
|
|
1341
|
+
for he in hyperedges:
|
|
1342
|
+
if isinstance(he, dict) and he.get("source_file"):
|
|
1343
|
+
he["source_file"] = _norm_source_file(he["source_file"], _root)
|
|
1344
|
+
# Validate members against the built node set (#1916): a hyperedge
|
|
1345
|
+
# member absent from the graph used to be copied into
|
|
1346
|
+
# G.graph["hyperedges"] verbatim and reach graph.json dangling,
|
|
1347
|
+
# even from a live (non-cache) extraction. Mirror the pairwise-edge
|
|
1348
|
+
# handling above: remap mismatched ids via normalization first,
|
|
1349
|
+
# then drop members that still don't resolve; drop the hyperedge
|
|
1350
|
+
# itself when no valid member remains (single-member hyperedges
|
|
1351
|
+
# are legal in this codebase, e.g. a per-file flow, so we prune
|
|
1352
|
+
# rather than require two survivors).
|
|
1353
|
+
if isinstance(he, dict) and isinstance(he.get("nodes"), list):
|
|
1354
|
+
valid_members = []
|
|
1355
|
+
for m in he["nodes"]:
|
|
1356
|
+
try:
|
|
1357
|
+
hash(m)
|
|
1358
|
+
except TypeError:
|
|
1359
|
+
continue
|
|
1360
|
+
if m not in node_set and isinstance(m, str):
|
|
1361
|
+
m = norm_to_id.get(_normalize_id(m), m)
|
|
1362
|
+
if m in node_set:
|
|
1363
|
+
valid_members.append(m)
|
|
1364
|
+
if not valid_members:
|
|
1365
|
+
print(
|
|
1366
|
+
f"[graphify] WARNING: dropping hyperedge "
|
|
1367
|
+
f"{he.get('id', '?')!r} — none of its members "
|
|
1368
|
+
f"{he.get('nodes')!r} match built nodes.",
|
|
1369
|
+
file=sys.stderr,
|
|
1370
|
+
)
|
|
1371
|
+
continue
|
|
1372
|
+
if valid_members != he["nodes"]:
|
|
1373
|
+
he["nodes"] = valid_members
|
|
1374
|
+
kept_hyperedges.append(he)
|
|
1375
|
+
if kept_hyperedges:
|
|
1376
|
+
G.graph["hyperedges"] = kept_hyperedges
|
|
1377
|
+
else:
|
|
1378
|
+
# Full wipeout (#2485): every incoming hyperedge failed member
|
|
1379
|
+
# revalidation. Store an EXPLICIT empty list — distinct from
|
|
1380
|
+
# "this graph never carried hyperedge metadata" — and say loudly
|
|
1381
|
+
# that the persisted set is about to be emptied, so the per-edge
|
|
1382
|
+
# warnings above can't scroll past unnoticed.
|
|
1383
|
+
G.graph["hyperedges"] = []
|
|
1384
|
+
print(
|
|
1385
|
+
f"[graphify] WARNING: all {len(hyperedges)} hyperedge(s) were "
|
|
1386
|
+
f"dropped by member revalidation; graph.json's hyperedge set "
|
|
1387
|
+
f"will be emptied on the next export.",
|
|
1388
|
+
file=sys.stderr,
|
|
1389
|
+
)
|
|
1390
|
+
# Runs LAST, after the alias-competition above (which relies on file-node
|
|
1391
|
+
# labels still being bare basenames): give colliding-basename file nodes a
|
|
1392
|
+
# directory-qualified display label so lookup/discovery can disambiguate
|
|
1393
|
+
# them (#2032). Labels only — ids and edges are untouched.
|
|
1394
|
+
_disambiguate_file_node_labels(G)
|
|
1395
|
+
return G
|
|
1396
|
+
|
|
1397
|
+
|
|
1398
|
+
def build(
|
|
1399
|
+
extractions: list[dict],
|
|
1400
|
+
*,
|
|
1401
|
+
directed: bool = False,
|
|
1402
|
+
dedup: bool = True,
|
|
1403
|
+
dedup_llm_backend: str | None = None,
|
|
1404
|
+
root: str | Path | None = None,
|
|
1405
|
+
protected_ids: "set[str] | None" = None,
|
|
1406
|
+
) -> nx.Graph:
|
|
1407
|
+
"""Merge multiple extraction results into one graph.
|
|
1408
|
+
|
|
1409
|
+
directed=True produces a DiGraph that preserves edge direction (source→target).
|
|
1410
|
+
directed=False (default) produces an undirected Graph for backward compatibility.
|
|
1411
|
+
dedup=True (default) runs entity deduplication before building the graph.
|
|
1412
|
+
dedup_llm_backend: if set (e.g. "gemini", "claude", or "kimi"), uses LLM to resolve
|
|
1413
|
+
ambiguous pairs in the 75–92 Jaro-Winkler score zone.
|
|
1414
|
+
root: if given, absolute source_file paths are made relative to root (#932).
|
|
1415
|
+
protected_ids: optional set of node IDs to protect from being collapsed with
|
|
1416
|
+
other protected nodes during incremental merge (#3477).
|
|
1417
|
+
|
|
1418
|
+
With dedup disabled, extractions are merged in order and the last node's
|
|
1419
|
+
attributes win (NetworkX add_node overwrites). With dedup enabled, nodes
|
|
1420
|
+
sharing an ID use a deterministic survivor and retain missing attributes
|
|
1421
|
+
from duplicate records of the same source entity. Genuine cross-file ID
|
|
1422
|
+
collisions remain isolated and are reported.
|
|
1423
|
+
"""
|
|
1424
|
+
from graphify.dedup import deduplicate_entities
|
|
1425
|
+
combined: dict = {"nodes": [], "edges": [], "hyperedges": [], "input_tokens": 0, "output_tokens": 0}
|
|
1426
|
+
for ext in extractions:
|
|
1427
|
+
combined["nodes"].extend(ext.get("nodes", []))
|
|
1428
|
+
combined["edges"].extend(ext.get("edges", []))
|
|
1429
|
+
combined["hyperedges"].extend(ext.get("hyperedges", []))
|
|
1430
|
+
combined["input_tokens"] += ext.get("input_tokens", 0)
|
|
1431
|
+
combined["output_tokens"] += ext.get("output_tokens", 0)
|
|
1432
|
+
_root = str(Path(root).resolve()) if root else None
|
|
1433
|
+
if dedup and combined["nodes"]:
|
|
1434
|
+
# Numeric ids must be str before dedup, which keys on them and would
|
|
1435
|
+
# raise TypeError in _pick_winner's regex search (#2326). build_from_json
|
|
1436
|
+
# coerces too, but that runs after dedup — too late for this path.
|
|
1437
|
+
_coerce_non_string_ids(combined)
|
|
1438
|
+
# Fold legacy node field aliases before dedup (#2194): dedup runs BEFORE
|
|
1439
|
+
# build_from_json and keys on `label`, so a `name`/`path` alias node
|
|
1440
|
+
# would be invisible to it and only label-dedup one build later, after
|
|
1441
|
+
# build_from_json's own fold has healed the persisted graph.json.
|
|
1442
|
+
for n in combined["nodes"]:
|
|
1443
|
+
if isinstance(n, dict):
|
|
1444
|
+
_fold_node_aliases(n)
|
|
1445
|
+
# Normalize source_file and definition_file to the build root before
|
|
1446
|
+
# deduplication (#3472), so exact-ID collision checks and same-file
|
|
1447
|
+
# attribute merging operate on canonical repo-relative paths rather
|
|
1448
|
+
# than false-flagging absolute paths from semantic subagents as
|
|
1449
|
+
# different files.
|
|
1450
|
+
if "source_file" in n:
|
|
1451
|
+
n["source_file"] = _norm_source_file(n["source_file"], _root)
|
|
1452
|
+
if "definition_file" in n:
|
|
1453
|
+
n["definition_file"] = _norm_source_file(n["definition_file"], _root)
|
|
1454
|
+
combined["nodes"], combined["edges"] = deduplicate_entities(
|
|
1455
|
+
combined["nodes"], combined["edges"], communities={},
|
|
1456
|
+
dedup_llm_backend=dedup_llm_backend, root=_root,
|
|
1457
|
+
# Hyperedge members reference node ids too, so they need the same
|
|
1458
|
+
# survivor rewiring the edges get (#2805).
|
|
1459
|
+
hyperedges=combined.get("hyperedges"),
|
|
1460
|
+
protected_ids=protected_ids,
|
|
1461
|
+
)
|
|
1462
|
+
return build_from_json(combined, directed=directed, root=_root)
|
|
1463
|
+
|
|
1464
|
+
|
|
1465
|
+
def _norm_label(label: str | None) -> str:
|
|
1466
|
+
"""Canonical dedup key — Unicode-aware, preserves CJK/word characters."""
|
|
1467
|
+
if not isinstance(label, str):
|
|
1468
|
+
label = "" if label is None else str(label)
|
|
1469
|
+
label = unicodedata.normalize("NFKC", label)
|
|
1470
|
+
return re.sub(r"[\W_ ]+", " ", label.casefold(), flags=re.UNICODE).strip()
|
|
1471
|
+
|
|
1472
|
+
|
|
1473
|
+
def deduplicate_by_label(nodes: list[dict], edges: list[dict]) -> tuple[list[dict], list[dict]]:
|
|
1474
|
+
"""Merge nodes that share a normalised label, rewriting edge references.
|
|
1475
|
+
|
|
1476
|
+
Prefers IDs without chunk suffixes (_c\\d+) and shorter IDs when tied.
|
|
1477
|
+
Drops self-loops created by the merge.
|
|
1478
|
+
|
|
1479
|
+
Dormant: this is NOT wired into ``build()`` — the active dedup path is
|
|
1480
|
+
``deduplicate_entities`` (imported and called in ``build``), which supersedes
|
|
1481
|
+
it. The previous "Called in build() automatically" note was never true. It
|
|
1482
|
+
also merges by label alone with no ``file_type`` guard, so it must not be
|
|
1483
|
+
enabled for code nodes: same-label symbols from different files/packages
|
|
1484
|
+
(e.g. two ``Account`` types) would collapse into one — the cross-file
|
|
1485
|
+
conflation ``deduplicate_entities`` deliberately avoids for code (#1205).
|
|
1486
|
+
"""
|
|
1487
|
+
_CHUNK_SUFFIX = re.compile(r"_c\d+$")
|
|
1488
|
+
canonical: dict[str, dict] = {} # norm_label -> surviving node
|
|
1489
|
+
remap: dict[str, str] = {} # old_id -> surviving_id
|
|
1490
|
+
|
|
1491
|
+
for node in nodes:
|
|
1492
|
+
key = _norm_label(node.get("label", node.get("id", "")))
|
|
1493
|
+
if not key:
|
|
1494
|
+
continue
|
|
1495
|
+
existing = canonical.get(key)
|
|
1496
|
+
if existing is None:
|
|
1497
|
+
canonical[key] = node
|
|
1498
|
+
else:
|
|
1499
|
+
has_suffix = bool(_CHUNK_SUFFIX.search(node["id"]))
|
|
1500
|
+
existing_has_suffix = bool(_CHUNK_SUFFIX.search(existing["id"]))
|
|
1501
|
+
if has_suffix and not existing_has_suffix:
|
|
1502
|
+
remap[node["id"]] = existing["id"]
|
|
1503
|
+
elif existing_has_suffix and not has_suffix:
|
|
1504
|
+
remap[existing["id"]] = node["id"]
|
|
1505
|
+
canonical[key] = node
|
|
1506
|
+
elif len(node["id"]) < len(existing["id"]):
|
|
1507
|
+
remap[existing["id"]] = node["id"]
|
|
1508
|
+
canonical[key] = node
|
|
1509
|
+
else:
|
|
1510
|
+
remap[node["id"]] = existing["id"]
|
|
1511
|
+
|
|
1512
|
+
if not remap:
|
|
1513
|
+
return nodes, edges
|
|
1514
|
+
|
|
1515
|
+
print(f"[graphify] Deduplicated {len(remap)} duplicate node(s) by label.", file=sys.stderr)
|
|
1516
|
+
deduped_nodes = list(canonical.values())
|
|
1517
|
+
deduped_edges = []
|
|
1518
|
+
for edge in edges:
|
|
1519
|
+
e = dict(edge)
|
|
1520
|
+
e["source"] = remap.get(e["source"], e["source"])
|
|
1521
|
+
e["target"] = remap.get(e["target"], e["target"])
|
|
1522
|
+
if e["source"] != e["target"]:
|
|
1523
|
+
deduped_edges.append(e)
|
|
1524
|
+
return deduped_nodes, deduped_edges
|
|
1525
|
+
|
|
1526
|
+
|
|
1527
|
+
def _load_existing_graph(graph_path: Path) -> "tuple[list, list, list, bool] | None":
|
|
1528
|
+
"""Load (nodes, edges, hyperedges, directed) from an existing graph.json for
|
|
1529
|
+
an incremental merge, accepting both the ``links`` and ``edges`` spellings.
|
|
1530
|
+
|
|
1531
|
+
Reads the JSON directly instead of going through node_link_graph().
|
|
1532
|
+
The latter rebuilds an undirected nx.Graph and then enumerating
|
|
1533
|
+
edges() yields endpoints based on node insertion order, which
|
|
1534
|
+
silently flips directional edges (e.g. `calls`) when the callee
|
|
1535
|
+
was inserted before the caller. The _src/_tgt direction-preserving
|
|
1536
|
+
attrs are popped before saving in export.py, so going through the
|
|
1537
|
+
NetworkX round-trip loses direction permanently (#760).
|
|
1538
|
+
|
|
1539
|
+
Returns None when the file does not exist. Raises RuntimeError when it
|
|
1540
|
+
exists but cannot be parsed — callers must refuse to overwrite rather
|
|
1541
|
+
than silently replace a possibly-recoverable graph.
|
|
1542
|
+
"""
|
|
1543
|
+
if not graph_path.exists():
|
|
1544
|
+
return None
|
|
1545
|
+
from graphify.security import check_graph_file_size_cap
|
|
1546
|
+
check_graph_file_size_cap(graph_path)
|
|
1547
|
+
try:
|
|
1548
|
+
data = json.loads(graph_path.read_text(encoding="utf-8"))
|
|
1549
|
+
except (json.JSONDecodeError, OSError) as exc:
|
|
1550
|
+
raise RuntimeError(
|
|
1551
|
+
f"Cannot read {graph_path} for incremental merge: {exc}. "
|
|
1552
|
+
"Delete the file and run a full rebuild."
|
|
1553
|
+
) from exc
|
|
1554
|
+
links_key = "links" if "links" in data else "edges"
|
|
1555
|
+
nodes = list(data.get("nodes", []))
|
|
1556
|
+
edges = list(data.get(links_key, []))
|
|
1557
|
+
# Backfill tier provenance on legacy items (#2334): _origin is stamped at
|
|
1558
|
+
# extraction time only (extract.py for AST, and the semantic path never
|
|
1559
|
+
# stamps), so pre-0.9.16 graphs and externally-merged fragments carry
|
|
1560
|
+
# unstamped items. Stamp them via the _is_ast_tier shape fallback so the
|
|
1561
|
+
# graph self-heals on the next write and every downstream tier decision
|
|
1562
|
+
# (build_merge replace, watch reconcile) reads an explicit marker.
|
|
1563
|
+
for item in nodes:
|
|
1564
|
+
if isinstance(item, dict):
|
|
1565
|
+
item.setdefault("_origin", "ast" if _is_ast_tier(item) else "semantic")
|
|
1566
|
+
for item in edges:
|
|
1567
|
+
if isinstance(item, dict):
|
|
1568
|
+
item.setdefault("_origin", "ast" if _is_ast_tier(item) else "semantic")
|
|
1569
|
+
return (
|
|
1570
|
+
nodes,
|
|
1571
|
+
edges,
|
|
1572
|
+
list(data.get("hyperedges", [])),
|
|
1573
|
+
bool(data.get("directed", False)),
|
|
1574
|
+
)
|
|
1575
|
+
|
|
1576
|
+
|
|
1577
|
+
def _tier_replacement_sources(
|
|
1578
|
+
chunks: "Iterable[dict]",
|
|
1579
|
+
root: "str | Path | None" = None,
|
|
1580
|
+
ast_sources: "Iterable[str | Path] | None" = None,
|
|
1581
|
+
) -> tuple[set[str], set[str]]:
|
|
1582
|
+
"""Compute (new_ast_sources, new_sem_sources) for tier-scoped replacement.
|
|
1583
|
+
|
|
1584
|
+
#3411: AST replacement ownership is derived from explicit extraction
|
|
1585
|
+
provenance (ast_sources or chunk-level "extracted_sources"), so cross-file
|
|
1586
|
+
stub nodes emitted by extractors (e.g. .sln project stubs or ProjectReference
|
|
1587
|
+
stubs) do not pollute the replacement set and wipe the target project's
|
|
1588
|
+
nodes/edges. Falls back to AST node source_file only when no provenance is
|
|
1589
|
+
provided.
|
|
1590
|
+
"""
|
|
1591
|
+
explicit_ast_sources: set[str] = set()
|
|
1592
|
+
if ast_sources is not None:
|
|
1593
|
+
for s in ast_sources:
|
|
1594
|
+
if s:
|
|
1595
|
+
explicit_ast_sources.add(str(s))
|
|
1596
|
+
for ch in chunks:
|
|
1597
|
+
if isinstance(ch, dict):
|
|
1598
|
+
for s in (ch.get("extracted_sources") or []):
|
|
1599
|
+
if s:
|
|
1600
|
+
explicit_ast_sources.add(str(s))
|
|
1601
|
+
|
|
1602
|
+
new_ast_sources: set[str] = set()
|
|
1603
|
+
new_sem_sources: set[str] = set()
|
|
1604
|
+
|
|
1605
|
+
if explicit_ast_sources:
|
|
1606
|
+
for sf in explicit_ast_sources:
|
|
1607
|
+
new_ast_sources.add(sf)
|
|
1608
|
+
norm = _norm_source_file(sf, root)
|
|
1609
|
+
if norm:
|
|
1610
|
+
new_ast_sources.add(norm)
|
|
1611
|
+
else:
|
|
1612
|
+
for ch in chunks:
|
|
1613
|
+
if not isinstance(ch, dict):
|
|
1614
|
+
continue
|
|
1615
|
+
for n in ch.get("nodes", []):
|
|
1616
|
+
if not isinstance(n, dict):
|
|
1617
|
+
continue
|
|
1618
|
+
sf = n.get("source_file")
|
|
1619
|
+
if not sf or not _is_ast_tier(n):
|
|
1620
|
+
continue
|
|
1621
|
+
new_ast_sources.add(sf)
|
|
1622
|
+
norm = _norm_source_file(sf, root)
|
|
1623
|
+
if norm:
|
|
1624
|
+
new_ast_sources.add(norm)
|
|
1625
|
+
|
|
1626
|
+
for ch in chunks:
|
|
1627
|
+
if not isinstance(ch, dict):
|
|
1628
|
+
continue
|
|
1629
|
+
for n in ch.get("nodes", []):
|
|
1630
|
+
if not isinstance(n, dict):
|
|
1631
|
+
continue
|
|
1632
|
+
sf = n.get("source_file")
|
|
1633
|
+
if not sf or _is_ast_tier(n):
|
|
1634
|
+
continue
|
|
1635
|
+
new_sem_sources.add(sf)
|
|
1636
|
+
norm = _norm_source_file(sf, root)
|
|
1637
|
+
if norm:
|
|
1638
|
+
new_sem_sources.add(norm)
|
|
1639
|
+
|
|
1640
|
+
return new_ast_sources, new_sem_sources
|
|
1641
|
+
|
|
1642
|
+
|
|
1643
|
+
def merge_raw_extraction(
|
|
1644
|
+
new: dict,
|
|
1645
|
+
graph_path: str | Path,
|
|
1646
|
+
prune_sources: "list[str] | None" = None,
|
|
1647
|
+
root: "str | Path | None" = None,
|
|
1648
|
+
*,
|
|
1649
|
+
ast_sources: "Iterable[str | Path] | None" = None,
|
|
1650
|
+
) -> dict:
|
|
1651
|
+
"""Merge the existing raw graph.json forward into a fresh raw extraction
|
|
1652
|
+
(the ``extract --no-cluster`` incremental path, #2169).
|
|
1653
|
+
|
|
1654
|
+
Replace/prune semantics mirror :func:`build_merge` exactly, so the raw and
|
|
1655
|
+
clustered incremental paths can't drift:
|
|
1656
|
+
|
|
1657
|
+
- sources re-extracted this run REPLACE their prior contribution PER TIER
|
|
1658
|
+
(#2333/#2336, #3411): existing nodes/edges/hyperedges owned by them are dropped
|
|
1659
|
+
only when the new extraction contains the same tier (AST vs semantic,
|
|
1660
|
+
per :func:`_is_ast_tier`) for that source, matched in both raw and
|
|
1661
|
+
:func:`_norm_source_file` form (#1007). AST replacement ownership is derived
|
|
1662
|
+
from explicit extraction provenance (``ast_sources`` or chunk-level
|
|
1663
|
+
``extracted_sources``), falling back to AST node source_file only when no
|
|
1664
|
+
provenance is provided (#3411);
|
|
1665
|
+
- ``prune_sources`` (deleted / excluded / graph-stale files) are dropped,
|
|
1666
|
+
with the ``_abs_identity`` third-form fallback (#2012), and "replace" wins
|
|
1667
|
+
over a contradictory "delete" of a re-extracted source (#1796);
|
|
1668
|
+
- everything else — nodes/edges/hyperedges owned by unchanged files — is
|
|
1669
|
+
carried forward unchanged.
|
|
1670
|
+
|
|
1671
|
+
Survivors are PREPENDED to ``new``'s lists (existing-first), so the caller's
|
|
1672
|
+
``dedupe_nodes`` last-writer-wins keeps fresh attributes for re-extracted
|
|
1673
|
+
nodes while ``dedupe_edges`` first-wins never resurrects a replaced edge
|
|
1674
|
+
(replaced sources' edges were already dropped above). Token counters and
|
|
1675
|
+
every other key of ``new`` are left untouched. Returns ``new``, mutated in
|
|
1676
|
+
place. Raises RuntimeError (via :func:`_load_existing_graph`) when the
|
|
1677
|
+
existing graph is present but unparseable — the caller must refuse to
|
|
1678
|
+
overwrite it. No-op when ``graph_path`` does not exist.
|
|
1679
|
+
"""
|
|
1680
|
+
graph_path = Path(graph_path)
|
|
1681
|
+
loaded = _load_existing_graph(graph_path)
|
|
1682
|
+
if loaded is None:
|
|
1683
|
+
return new
|
|
1684
|
+
existing_nodes, existing_edges, existing_hyperedges, _ = loaded
|
|
1685
|
+
|
|
1686
|
+
_eff_root = (
|
|
1687
|
+
str(Path(root).resolve()) if root is not None
|
|
1688
|
+
else _infer_merge_root(graph_path)
|
|
1689
|
+
)
|
|
1690
|
+
|
|
1691
|
+
# Tier-scoped replace, mirroring build_merge (#2333/#2336, COEXIST, #3411): a
|
|
1692
|
+
# source re-extracted this run replaces only the tier(s) actually present
|
|
1693
|
+
# in the new extraction, so an AST-only re-extract keeps the file's
|
|
1694
|
+
# semantic layer and vice versa. AST replacement ownership is derived from
|
|
1695
|
+
# explicit extraction provenance (ast_sources or "extracted_sources"),
|
|
1696
|
+
# falling back to AST node source_files only when no provenance is provided.
|
|
1697
|
+
new_ast_sources, new_sem_sources = _tier_replacement_sources(
|
|
1698
|
+
[new], root=_eff_root, ast_sources=ast_sources
|
|
1699
|
+
)
|
|
1700
|
+
new_sources: set[str] = new_ast_sources | new_sem_sources
|
|
1701
|
+
|
|
1702
|
+
# "Replace" wins over a contradictory "delete" of the same source (#1796),
|
|
1703
|
+
# in both string and absolute-identity space (#2012) — as in build_merge.
|
|
1704
|
+
prune_set, prune_abs = _build_prune_sets(prune_sources, _eff_root, new_sources)
|
|
1705
|
+
_prune_root = _eff_root
|
|
1706
|
+
|
|
1707
|
+
def _prune_hit(sf: "str | None") -> bool:
|
|
1708
|
+
if not sf:
|
|
1709
|
+
return False
|
|
1710
|
+
if sf in prune_set:
|
|
1711
|
+
return True
|
|
1712
|
+
norm = _norm_source_file(sf, _prune_root)
|
|
1713
|
+
if norm and norm in prune_set:
|
|
1714
|
+
return True
|
|
1715
|
+
a = _abs_identity(sf, _prune_root)
|
|
1716
|
+
return bool(a) and a in prune_abs
|
|
1717
|
+
|
|
1718
|
+
# #2446: when NONE of the prune entries match anything stored, the guessed
|
|
1719
|
+
# _eff_root is usually wrong (non-standard layout, no marker) and every
|
|
1720
|
+
# prune silently no-ops. Derive the root by suffix-matching the absolute
|
|
1721
|
+
# prune paths against the stored relative source_files and retry — same
|
|
1722
|
+
# fallback as build_merge.
|
|
1723
|
+
if prune_set or prune_abs:
|
|
1724
|
+
_stored_sfs = {
|
|
1725
|
+
item.get("source_file")
|
|
1726
|
+
for seq in (existing_nodes, existing_edges, existing_hyperedges)
|
|
1727
|
+
for item in seq if isinstance(item, dict)
|
|
1728
|
+
}
|
|
1729
|
+
_stored_sfs.discard(None)
|
|
1730
|
+
if not any(_prune_hit(sf) for sf in _stored_sfs):
|
|
1731
|
+
_derived = _derive_prune_root(prune_sources or [], _stored_sfs)
|
|
1732
|
+
if _derived is not None and _derived != _prune_root:
|
|
1733
|
+
_prune_root = _derived
|
|
1734
|
+
prune_set, prune_abs = _build_prune_sets(
|
|
1735
|
+
prune_sources, _prune_root, new_sources
|
|
1736
|
+
)
|
|
1737
|
+
|
|
1738
|
+
def _dropped(item: dict) -> bool:
|
|
1739
|
+
if not isinstance(item, dict):
|
|
1740
|
+
return True
|
|
1741
|
+
sf = item.get("source_file")
|
|
1742
|
+
# Tier-scoped replace: an item is superseded only when ITS OWN tier
|
|
1743
|
+
# re-extracted its source. Hyperedges are semantic-tier (no _origin,
|
|
1744
|
+
# null source_location), so an AST-only re-extract carries them.
|
|
1745
|
+
# Deletion pruning below stays tier-blind.
|
|
1746
|
+
own = new_ast_sources if _is_ast_tier(item) else new_sem_sources
|
|
1747
|
+
if sf in own or _norm_source_file(sf, _eff_root) in own:
|
|
1748
|
+
return True # re-extracted this run — replaced by the new chunk
|
|
1749
|
+
if not sf:
|
|
1750
|
+
return False # unowned — carry forward
|
|
1751
|
+
return _prune_hit(sf)
|
|
1752
|
+
|
|
1753
|
+
# #3203: Check for unverified semantic shrink on re-extracted sources.
|
|
1754
|
+
unverified_semantic_shrink: dict[str, tuple[int, int]] = {}
|
|
1755
|
+
if new_sem_sources:
|
|
1756
|
+
prior_sem_counts: dict[str, int] = {}
|
|
1757
|
+
for n in existing_nodes:
|
|
1758
|
+
if isinstance(n, dict) and not _is_ast_tier(n):
|
|
1759
|
+
sf = n.get("source_file")
|
|
1760
|
+
if sf:
|
|
1761
|
+
canon = _norm_source_file(sf, _eff_root) or sf
|
|
1762
|
+
prior_sem_counts[canon] = prior_sem_counts.get(canon, 0) + 1
|
|
1763
|
+
|
|
1764
|
+
fresh_sem_counts: dict[str, int] = {}
|
|
1765
|
+
for n in new.get("nodes", []):
|
|
1766
|
+
if isinstance(n, dict) and not _is_ast_tier(n):
|
|
1767
|
+
sf = n.get("source_file")
|
|
1768
|
+
if sf:
|
|
1769
|
+
canon = _norm_source_file(sf, _eff_root) or sf
|
|
1770
|
+
fresh_sem_counts[canon] = fresh_sem_counts.get(canon, 0) + 1
|
|
1771
|
+
|
|
1772
|
+
for canon_sf, fresh_count in fresh_sem_counts.items():
|
|
1773
|
+
prior_count = prior_sem_counts.get(canon_sf, 0)
|
|
1774
|
+
if prior_count > 1 and fresh_count < prior_count:
|
|
1775
|
+
unverified_semantic_shrink[canon_sf] = (prior_count, fresh_count)
|
|
1776
|
+
|
|
1777
|
+
new["nodes"] = [n for n in existing_nodes if not _dropped(n)] + list(new.get("nodes", []))
|
|
1778
|
+
new["edges"] = [e for e in existing_edges if not _dropped(e)] + list(new.get("edges", []))
|
|
1779
|
+
carried_hyper = [he for he in existing_hyperedges if not _dropped(he)]
|
|
1780
|
+
if carried_hyper or new.get("hyperedges"):
|
|
1781
|
+
new["hyperedges"] = carried_hyper + list(new.get("hyperedges", []))
|
|
1782
|
+
if unverified_semantic_shrink:
|
|
1783
|
+
new["_unverified_semantic_shrink"] = unverified_semantic_shrink
|
|
1784
|
+
return new
|
|
1785
|
+
|
|
1786
|
+
|
|
1787
|
+
def build_merge(
|
|
1788
|
+
new_chunks: list[dict],
|
|
1789
|
+
graph_path: str | Path | None = None,
|
|
1790
|
+
prune_sources: list[str] | None = None,
|
|
1791
|
+
*,
|
|
1792
|
+
directed: bool | None = None,
|
|
1793
|
+
dedup: bool = True,
|
|
1794
|
+
dedup_llm_backend: str | None = None,
|
|
1795
|
+
root: str | Path | None = None,
|
|
1796
|
+
ast_sources: "Iterable[str | Path] | None" = None,
|
|
1797
|
+
) -> nx.Graph:
|
|
1798
|
+
"""Load existing graph.json and return it merged with ``new_chunks``.
|
|
1799
|
+
|
|
1800
|
+
Does NOT write to disk — the caller persists the result, e.g. via
|
|
1801
|
+
``export.to_json(G, communities, graph_path, force=True)`` after
|
|
1802
|
+
clustering. ``graph_path`` is read-only here.
|
|
1803
|
+
|
|
1804
|
+
Re-extracted files REPLACE their prior contribution per tier (#2333/#2336, #3411):
|
|
1805
|
+
a source_file present in new_chunks has its existing nodes/edges dropped
|
|
1806
|
+
for each tier (AST vs semantic, per :func:`_is_ast_tier`) the new chunks
|
|
1807
|
+
actually contain, so a changed file's stale nodes/edges don't accumulate
|
|
1808
|
+
while a one-tier re-extract keeps the other tier's layer intact. AST replacement
|
|
1809
|
+
ownership is derived from explicit extraction provenance (``ast_sources`` or
|
|
1810
|
+
chunk-level ``extracted_sources``), falling back to AST node source_file only
|
|
1811
|
+
when no provenance is provided (#3411). Files absent from new_chunks are
|
|
1812
|
+
preserved unchanged; deleted files are removed via prune_sources (tier-blind).
|
|
1813
|
+
Safe to call repeatedly.
|
|
1814
|
+
root: if given, absolute source_file paths in new_chunks are made relative (#932).
|
|
1815
|
+
directed: if None (default), honor the on-disk graph's own ``directed`` flag
|
|
1816
|
+
when one exists, so an incremental merge can't silently flip a directed
|
|
1817
|
+
graph undirected (#2342). Falls back to False when there is no existing
|
|
1818
|
+
graph to inherit from. An explicit True/False always overrides the on-disk
|
|
1819
|
+
flag.
|
|
1820
|
+
"""
|
|
1821
|
+
# Iterated more than once below (source sets, the hyperedge carry, the
|
|
1822
|
+
# build itself), so a one-shot iterator must be materialised first.
|
|
1823
|
+
new_chunks = list(new_chunks)
|
|
1824
|
+
graph_path = Path(graph_path if graph_path is not None else _default_graph_json())
|
|
1825
|
+
_loaded = _load_existing_graph(graph_path)
|
|
1826
|
+
if _loaded is not None:
|
|
1827
|
+
existing_nodes, existing_edges, existing_hyperedges, existing_directed = _loaded
|
|
1828
|
+
had_graph = True
|
|
1829
|
+
else:
|
|
1830
|
+
existing_nodes = []
|
|
1831
|
+
existing_edges = []
|
|
1832
|
+
existing_hyperedges = []
|
|
1833
|
+
existing_directed = False
|
|
1834
|
+
had_graph = False
|
|
1835
|
+
if directed is None:
|
|
1836
|
+
directed = existing_directed if had_graph else False
|
|
1837
|
+
|
|
1838
|
+
# Effective root for relativizing absolute source_file / prune paths back to the
|
|
1839
|
+
# stored relative source_file keys. When the caller passes root we use it;
|
|
1840
|
+
# otherwise fall back to the graph's recorded scan root, so absolute
|
|
1841
|
+
# prune_sources and new-chunk paths still match even when a caller omits root
|
|
1842
|
+
# (#1571 — the skill's --update runbook calls build_merge without root, so
|
|
1843
|
+
# absolute deleted-file paths never matched the relative node keys and their
|
|
1844
|
+
# nodes survived as ghosts).
|
|
1845
|
+
_eff_root = (
|
|
1846
|
+
str(Path(root).resolve()) if root is not None
|
|
1847
|
+
else _infer_merge_root(graph_path)
|
|
1848
|
+
)
|
|
1849
|
+
|
|
1850
|
+
# Re-extracted files REPLACE their prior contribution. Every source_file
|
|
1851
|
+
# present in new_chunks is dropped from the loaded base before merging, so a
|
|
1852
|
+
# CHANGED file's stale nodes/edges don't accumulate across incremental
|
|
1853
|
+
# updates. Without this, build() merges old+new for the same file and only
|
|
1854
|
+
# exact-duplicate edges collapse — edges/nodes that disappeared from the new
|
|
1855
|
+
# version survive forever. Brand-new files aren't in base, so this is a no-op
|
|
1856
|
+
# for them; genuinely deleted files are still handled via prune_sources.
|
|
1857
|
+
# Matched in both raw and _norm_source_file form because new_chunks may carry
|
|
1858
|
+
# absolute win32 paths while the stored graph keeps relative posix (#1007).
|
|
1859
|
+
# Replacement is tier-scoped (#2333/#2336, COEXIST): each file has two
|
|
1860
|
+
# producers — the deterministic AST pass and the semantic/LLM pass — whose
|
|
1861
|
+
# node sets coexist in the graph. A re-extract of one tier must replace
|
|
1862
|
+
# only that tier's prior contribution, never the other's (a semantic-only
|
|
1863
|
+
# chunk used to delete the file's AST headings). Which tier a NEW chunk
|
|
1864
|
+
# item belongs to is read via _is_ast_tier (existing items were stamped by
|
|
1865
|
+
# _load_existing_graph above).
|
|
1866
|
+
#
|
|
1867
|
+
# #3411: AST replacement ownership is derived from explicit extraction
|
|
1868
|
+
# provenance (ast_sources or chunk-level "extracted_sources"), so cross-file
|
|
1869
|
+
# stub nodes emitted by extractors (e.g. .sln project stubs or ProjectReference
|
|
1870
|
+
# stubs) do not pollute the replacement set and wipe the target project's
|
|
1871
|
+
# nodes/edges. Falls back to AST node source_file only when no provenance is
|
|
1872
|
+
# provided.
|
|
1873
|
+
_replace_root = _eff_root
|
|
1874
|
+
new_ast_sources, new_sem_sources = _tier_replacement_sources(
|
|
1875
|
+
new_chunks, root=_replace_root, ast_sources=ast_sources
|
|
1876
|
+
)
|
|
1877
|
+
new_sources: set[str] = new_ast_sources | new_sem_sources
|
|
1878
|
+
# True on-disk baseline for the #479 shrink accounting at the end (#2497):
|
|
1879
|
+
# the rebind below removes the re-extracted sources' old nodes from
|
|
1880
|
+
# existing_nodes, so any later size comparison against the rebound list can
|
|
1881
|
+
# never see the loss it is meant to catch.
|
|
1882
|
+
_disk_nodes = existing_nodes
|
|
1883
|
+
_disk_n = len(existing_nodes)
|
|
1884
|
+
|
|
1885
|
+
# #3203: Check for unverified semantic shrink on re-extracted sources.
|
|
1886
|
+
# An existing source with prior semantic nodes (> 1) that produces strictly
|
|
1887
|
+
# fewer semantic nodes in this extraction is flagged on G.graph so the CLI
|
|
1888
|
+
# can arm the shrink guard and leave the source unstamped in the manifest.
|
|
1889
|
+
unverified_semantic_shrink: dict[str, tuple[int, int]] = {}
|
|
1890
|
+
if had_graph and new_sem_sources:
|
|
1891
|
+
prior_sem_counts: dict[str, int] = {}
|
|
1892
|
+
for n in _disk_nodes:
|
|
1893
|
+
if isinstance(n, dict) and not _is_ast_tier(n):
|
|
1894
|
+
sf = n.get("source_file")
|
|
1895
|
+
if sf:
|
|
1896
|
+
canon = _norm_source_file(sf, _replace_root) or sf
|
|
1897
|
+
prior_sem_counts[canon] = prior_sem_counts.get(canon, 0) + 1
|
|
1898
|
+
|
|
1899
|
+
fresh_sem_counts: dict[str, int] = {}
|
|
1900
|
+
for ch in new_chunks:
|
|
1901
|
+
for n in ch.get("nodes", []):
|
|
1902
|
+
if isinstance(n, dict) and not _is_ast_tier(n):
|
|
1903
|
+
sf = n.get("source_file")
|
|
1904
|
+
if sf:
|
|
1905
|
+
canon = _norm_source_file(sf, _replace_root) or sf
|
|
1906
|
+
fresh_sem_counts[canon] = fresh_sem_counts.get(canon, 0) + 1
|
|
1907
|
+
|
|
1908
|
+
for canon_sf, fresh_count in fresh_sem_counts.items():
|
|
1909
|
+
prior_count = prior_sem_counts.get(canon_sf, 0)
|
|
1910
|
+
if prior_count > 1 and fresh_count < prior_count:
|
|
1911
|
+
unverified_semantic_shrink[canon_sf] = (prior_count, fresh_count)
|
|
1912
|
+
|
|
1913
|
+
if new_sources:
|
|
1914
|
+
def _kept(item: dict) -> bool:
|
|
1915
|
+
sf = item.get("source_file")
|
|
1916
|
+
own = new_ast_sources if _is_ast_tier(item) else new_sem_sources
|
|
1917
|
+
return sf not in own and _norm_source_file(sf, _replace_root) not in own
|
|
1918
|
+
existing_nodes = [n for n in existing_nodes if _kept(n)]
|
|
1919
|
+
existing_edges = [e for e in existing_edges if _kept(e)]
|
|
1920
|
+
replaced_n = _disk_n - len(existing_nodes)
|
|
1921
|
+
if replaced_n:
|
|
1922
|
+
print(
|
|
1923
|
+
f"[graphify] Replaced {replaced_n} node(s) from re-extracted "
|
|
1924
|
+
f"source file(s).",
|
|
1925
|
+
file=sys.stderr,
|
|
1926
|
+
)
|
|
1927
|
+
|
|
1928
|
+
# Prune set for deleted source files — both the raw form (matches nodes that
|
|
1929
|
+
# kept absolute source_file) and the normalised relative form (matches nodes
|
|
1930
|
+
# relativised by _norm_source_file at build time). .resolve() (via _eff_root)
|
|
1931
|
+
# handles symlinked roots and ".." / "./" segments so Path.relative_to()
|
|
1932
|
+
# succeeds even when the scan root is a symlink. (#1007, #1571)
|
|
1933
|
+
#
|
|
1934
|
+
# A file that was just re-extracted (present in new_chunks) is being REPLACED,
|
|
1935
|
+
# never deleted — so never prune it, even if the caller also lists it in
|
|
1936
|
+
# prune_sources. Otherwise its fresh, just-built nodes are silently removed
|
|
1937
|
+
# (data loss): common when an edit keeps a node's label and the caller follows
|
|
1938
|
+
# the old edit-workflow of passing the changed file in prune_sources (#1796).
|
|
1939
|
+
# "replace" wins over a contradictory "delete" of the same source. Applied in
|
|
1940
|
+
# both string and absolute-identity space so the third-form fallback below
|
|
1941
|
+
# can't resurrect the delete for a re-extracted file (#2012).
|
|
1942
|
+
_prune_root = _eff_root
|
|
1943
|
+
prune_set, prune_abs = _build_prune_sets(prune_sources, _prune_root, new_sources)
|
|
1944
|
+
_matched_prune_entries: set[str] = set()
|
|
1945
|
+
|
|
1946
|
+
def _prune_match(sf: "str | None") -> bool:
|
|
1947
|
+
# Match a node/edge/hyperedge source_file against the prune set in a
|
|
1948
|
+
# form-insensitive way: exact string, normalised-relative, then the
|
|
1949
|
+
# absolute-identity fallback for the third-form case (#2012). Records
|
|
1950
|
+
# WHICH prune entry matched, so the prune report can count only the
|
|
1951
|
+
# entries that actually hit something (#2446).
|
|
1952
|
+
if not sf:
|
|
1953
|
+
return False
|
|
1954
|
+
hit = prune_set.get(sf)
|
|
1955
|
+
if hit is None:
|
|
1956
|
+
norm = _norm_source_file(sf, _prune_root)
|
|
1957
|
+
if norm:
|
|
1958
|
+
hit = prune_set.get(norm)
|
|
1959
|
+
if hit is None:
|
|
1960
|
+
a = _abs_identity(sf, _prune_root)
|
|
1961
|
+
if a:
|
|
1962
|
+
hit = prune_abs.get(a)
|
|
1963
|
+
if hit is None:
|
|
1964
|
+
return False
|
|
1965
|
+
_matched_prune_entries.add(hit)
|
|
1966
|
+
return True
|
|
1967
|
+
|
|
1968
|
+
# #2446: when NONE of the prune entries match anything stored, the guessed
|
|
1969
|
+
# _eff_root is usually wrong (non-standard layout, no marker) and every
|
|
1970
|
+
# prune would silently no-op. Derive the root by suffix-matching the
|
|
1971
|
+
# absolute prune paths against the stored relative source_files and redo
|
|
1972
|
+
# the prune sets with it; on ambiguity fall through to the zero-match
|
|
1973
|
+
# warning below. Runs before the hyperedge carry so hyperedge pruning
|
|
1974
|
+
# benefits too.
|
|
1975
|
+
if prune_set or prune_abs:
|
|
1976
|
+
_stored_sfs = {
|
|
1977
|
+
item.get("source_file")
|
|
1978
|
+
for seq in (_disk_nodes, existing_edges, existing_hyperedges)
|
|
1979
|
+
for item in seq if isinstance(item, dict)
|
|
1980
|
+
}
|
|
1981
|
+
_stored_sfs.discard(None)
|
|
1982
|
+
if not any(_prune_match(sf) for sf in _stored_sfs):
|
|
1983
|
+
_derived = _derive_prune_root(prune_sources or [], _stored_sfs)
|
|
1984
|
+
if _derived is not None and _derived != _prune_root:
|
|
1985
|
+
_prune_root = _derived
|
|
1986
|
+
prune_set, prune_abs = _build_prune_sets(
|
|
1987
|
+
prune_sources, _prune_root, new_sources
|
|
1988
|
+
)
|
|
1989
|
+
|
|
1990
|
+
# Carry forward hyperedges from files that were neither re-extracted nor
|
|
1991
|
+
# deleted (#1574). build() only sees the new chunks' hyperedges, so without
|
|
1992
|
+
# this every --update collapses the graph's hyperedge set down to just the
|
|
1993
|
+
# changed files'. Re-extracted files' prior hyperedges are dropped (their new
|
|
1994
|
+
# version is already in the new chunks — replace-per-source, like
|
|
1995
|
+
# nodes/edges); deleted files' are dropped via prune_set; id-dedup so a
|
|
1996
|
+
# carried hyperedge never duplicates one the new chunks re-emitted. Mirrors
|
|
1997
|
+
# watch.py, which already preserves existing hyperedges across a rebuild.
|
|
1998
|
+
#
|
|
1999
|
+
# The carried set rides INTO build() on the base chunk rather than being
|
|
2000
|
+
# attached to G afterwards (#3102): entity dedup rewires every edge endpoint
|
|
2001
|
+
# and every hyperedge member it sees onto the survivor (#2805), but a
|
|
2002
|
+
# hyperedge attached after the fact kept naming the merged-away node — a
|
|
2003
|
+
# dangling member with no backing node in the written graph.
|
|
2004
|
+
carried_hyperedges: list[dict] = []
|
|
2005
|
+
if existing_hyperedges:
|
|
2006
|
+
carried = carried_hyperedges
|
|
2007
|
+
_new_hyperedge_ids = {
|
|
2008
|
+
he.get("id")
|
|
2009
|
+
for chunk in new_chunks
|
|
2010
|
+
for he in (chunk.get("hyperedges") or [])
|
|
2011
|
+
if isinstance(he, dict) and he.get("id")
|
|
2012
|
+
}
|
|
2013
|
+
for he in existing_hyperedges:
|
|
2014
|
+
if not isinstance(he, dict):
|
|
2015
|
+
continue
|
|
2016
|
+
sf = he.get("source_file")
|
|
2017
|
+
norm = _norm_source_file(sf, _eff_root)
|
|
2018
|
+
# Hyperedges are semantic-tier: only a SEMANTIC re-extract of the
|
|
2019
|
+
# source replaces them. An AST-only re-extract cannot regenerate
|
|
2020
|
+
# hyperedges, so dropping them there would be data loss (#2336).
|
|
2021
|
+
if sf in new_sem_sources or norm in new_sem_sources:
|
|
2022
|
+
continue # semantically re-extracted — replaced by the new chunk's version
|
|
2023
|
+
if _prune_match(sf):
|
|
2024
|
+
continue # deleted — pruned
|
|
2025
|
+
if he.get("id") and he.get("id") in _new_hyperedge_ids:
|
|
2026
|
+
continue # the new chunks re-emitted it — theirs wins
|
|
2027
|
+
carried.append(he)
|
|
2028
|
+
|
|
2029
|
+
base = (
|
|
2030
|
+
[{"nodes": existing_nodes, "edges": existing_edges, "hyperedges": carried_hyperedges}]
|
|
2031
|
+
if had_graph else []
|
|
2032
|
+
)
|
|
2033
|
+
|
|
2034
|
+
# Untouched existing nodes must not be collapsed with each other during dedup (#3477).
|
|
2035
|
+
_protected_ids = {
|
|
2036
|
+
n["id"] for n in existing_nodes
|
|
2037
|
+
if isinstance(n, dict) and n.get("id")
|
|
2038
|
+
} if had_graph else None
|
|
2039
|
+
|
|
2040
|
+
all_chunks = base + list(new_chunks)
|
|
2041
|
+
G = build(
|
|
2042
|
+
all_chunks,
|
|
2043
|
+
directed=directed,
|
|
2044
|
+
dedup=dedup,
|
|
2045
|
+
dedup_llm_backend=dedup_llm_backend,
|
|
2046
|
+
root=_eff_root,
|
|
2047
|
+
protected_ids=_protected_ids,
|
|
2048
|
+
)
|
|
2049
|
+
|
|
2050
|
+
# Prune nodes and edges from deleted source files
|
|
2051
|
+
if prune_sources:
|
|
2052
|
+
# Source-less nodes that are ALREADY isolated before this prune. They are
|
|
2053
|
+
# not this prune's doing, so they must survive it — the sweep below is
|
|
2054
|
+
# scoped to the ones it orphans itself.
|
|
2055
|
+
_isolated_before = {
|
|
2056
|
+
n for n, d in G.nodes(data=True)
|
|
2057
|
+
if not d.get("source_file") and G.degree(n) == 0
|
|
2058
|
+
}
|
|
2059
|
+
to_remove = [
|
|
2060
|
+
n for n, d in G.nodes(data=True)
|
|
2061
|
+
if _prune_match(d.get("source_file"))
|
|
2062
|
+
]
|
|
2063
|
+
G.remove_nodes_from(to_remove)
|
|
2064
|
+
n_nodes = len(to_remove)
|
|
2065
|
+
|
|
2066
|
+
edges_to_remove = [
|
|
2067
|
+
(u, v) for u, v, d in G.edges(data=True)
|
|
2068
|
+
if _prune_match(d.get("source_file"))
|
|
2069
|
+
]
|
|
2070
|
+
if edges_to_remove:
|
|
2071
|
+
G.remove_edges_from(edges_to_remove)
|
|
2072
|
+
|
|
2073
|
+
# Extractors create a per-file node for each IMPORTED EXTERNAL symbol
|
|
2074
|
+
# (`Path` from pathlib, `Counter` from collections), and those carry no
|
|
2075
|
+
# source_file because they are defined outside the corpus. Every edge
|
|
2076
|
+
# they have points at symbols in the one file they were created for, so
|
|
2077
|
+
# pruning that file leaves them at degree 0 — named after a file the
|
|
2078
|
+
# corpus no longer contains, counted in every total that reads the graph,
|
|
2079
|
+
# exported as a note of their own, and unreachable by any future prune
|
|
2080
|
+
# since there is no source_file to match on. Nothing else can collect
|
|
2081
|
+
# them: deletions go through deleted_files, exclusions through
|
|
2082
|
+
# excluded_files (#1908) and _stale_graph_sources (#1909), and all three
|
|
2083
|
+
# match on source_file. A node with neither a source_file nor an edge
|
|
2084
|
+
# names nothing and connects nothing, so dropping it loses no
|
|
2085
|
+
# information (#2807).
|
|
2086
|
+
#
|
|
2087
|
+
# A single pass suffices: external-import stubs are only ever edge
|
|
2088
|
+
# TARGETS (extractors mint them as the target of an imports_from/
|
|
2089
|
+
# references/inherits edge, never as a source), so removing one can
|
|
2090
|
+
# never drop another to degree 0. A future extractor emitting a
|
|
2091
|
+
# stub->stub edge would require iterating this to a fixpoint.
|
|
2092
|
+
orphaned = [
|
|
2093
|
+
n for n, d in G.nodes(data=True)
|
|
2094
|
+
if not d.get("source_file")
|
|
2095
|
+
and G.degree(n) == 0
|
|
2096
|
+
and n not in _isolated_before
|
|
2097
|
+
]
|
|
2098
|
+
if orphaned:
|
|
2099
|
+
G.remove_nodes_from(orphaned)
|
|
2100
|
+
n_nodes += len(orphaned)
|
|
2101
|
+
|
|
2102
|
+
# Report only the prune entries that ACTUALLY matched something — not
|
|
2103
|
+
# len(prune_sources), which counted every entry as pruned-from even
|
|
2104
|
+
# when a root mismatch made most of them no-ops (#2446).
|
|
2105
|
+
n_files = len(_matched_prune_entries)
|
|
2106
|
+
if n_nodes:
|
|
2107
|
+
print(
|
|
2108
|
+
f"[graphify] Pruned {n_nodes} node(s) from {n_files} deleted or "
|
|
2109
|
+
f"excluded source file(s).",
|
|
2110
|
+
file=sys.stderr,
|
|
2111
|
+
)
|
|
2112
|
+
if edges_to_remove:
|
|
2113
|
+
print(
|
|
2114
|
+
f"[graphify] Pruned {len(edges_to_remove)} edge(s) from deleted or "
|
|
2115
|
+
f"excluded source file(s).",
|
|
2116
|
+
file=sys.stderr,
|
|
2117
|
+
)
|
|
2118
|
+
|
|
2119
|
+
if not n_nodes and not edges_to_remove:
|
|
2120
|
+
if (prune_set or prune_abs) and not _matched_prune_entries:
|
|
2121
|
+
# Live prune entries that matched NOTHING usually mean the
|
|
2122
|
+
# effective root is wrong (and the derived-root fallback above
|
|
2123
|
+
# found no consistent candidate) — warn instead of claiming the
|
|
2124
|
+
# graph is "already clean" (#2446).
|
|
2125
|
+
_p0 = next((p for p in prune_sources if p), None)
|
|
2126
|
+
sample_prune = _norm_source_file(_p0, _prune_root) or _p0
|
|
2127
|
+
sample_sf = next(
|
|
2128
|
+
iter(sorted(
|
|
2129
|
+
str(d.get("source_file"))
|
|
2130
|
+
for _, d in G.nodes(data=True) if d.get("source_file")
|
|
2131
|
+
)),
|
|
2132
|
+
None,
|
|
2133
|
+
)
|
|
2134
|
+
print(
|
|
2135
|
+
f"[graphify] WARNING: {len(prune_sources)} prune source(s) "
|
|
2136
|
+
f"matched no nodes or edges — nothing was removed. Prune "
|
|
2137
|
+
f"entry {sample_prune!r} does not correspond to any stored "
|
|
2138
|
+
f"source_file (e.g. {sample_sf!r}). If these files should "
|
|
2139
|
+
f"have been pruned, pass root= to build_merge so absolute "
|
|
2140
|
+
f"paths relativize to the graph's source_file keys. (#2446)",
|
|
2141
|
+
file=sys.stderr,
|
|
2142
|
+
)
|
|
2143
|
+
else:
|
|
2144
|
+
print(
|
|
2145
|
+
f"[graphify] {len(prune_sources)} source file(s) deleted or "
|
|
2146
|
+
f"excluded since last run — no matching nodes or edges in "
|
|
2147
|
+
f"graph, already clean.",
|
|
2148
|
+
file=sys.stderr,
|
|
2149
|
+
)
|
|
2150
|
+
|
|
2151
|
+
# Safety check: refuse to SILENTLY drop nodes (#479, reworked in #2497).
|
|
2152
|
+
# The old count comparison ran against the post-replace `existing_nodes`,
|
|
2153
|
+
# which had already lost the re-extracted sources' old nodes — so it could
|
|
2154
|
+
# never fire when it mattered, and was skipped outright whenever
|
|
2155
|
+
# prune_sources was passed. Mirror watch._check_shrink instead: diff the
|
|
2156
|
+
# on-disk baseline by node identity and excuse only losses explained by
|
|
2157
|
+
# this run's own re-extraction (same tier) or an explicit prune. Skipped
|
|
2158
|
+
# under dedup, where fuzzy merging collapses ids legitimately.
|
|
2159
|
+
#
|
|
2160
|
+
# Residual tradeoff (accepted): a partial re-extraction that under-produces
|
|
2161
|
+
# for ITS OWN file (>= 1 node still present in new_chunks) is excused here —
|
|
2162
|
+
# that failure mode is owned by the extraction layer's incomplete-build
|
|
2163
|
+
# guard (#1951), and refusing it here would reintroduce the #1116
|
|
2164
|
+
# false-refuse for legitimate edits that remove symbols from a file.
|
|
2165
|
+
if had_graph and not dedup:
|
|
2166
|
+
def _in_new_graph(n: dict) -> bool:
|
|
2167
|
+
nid = n.get("id")
|
|
2168
|
+
if nid is None or not _hashable(nid):
|
|
2169
|
+
return True # untrackable identity — never count as lost
|
|
2170
|
+
if nid in G:
|
|
2171
|
+
return True
|
|
2172
|
+
# Doc-twin heal (#1799): build_from_json merges a markdown
|
|
2173
|
+
# quick-scan's bare doc node into its semantic `<id>_doc` twin for
|
|
2174
|
+
# the same file — a legitimate collapse, not a loss.
|
|
2175
|
+
return (
|
|
2176
|
+
isinstance(nid, str)
|
|
2177
|
+
and not nid.endswith("_doc")
|
|
2178
|
+
and n.get("file_type") == "document"
|
|
2179
|
+
and f"{nid}_doc" in G
|
|
2180
|
+
and G.nodes[f"{nid}_doc"].get("source_file") == n.get("source_file")
|
|
2181
|
+
)
|
|
2182
|
+
|
|
2183
|
+
lost = [
|
|
2184
|
+
n for n in _disk_nodes
|
|
2185
|
+
if isinstance(n, dict) and not _in_new_graph(n)
|
|
2186
|
+
]
|
|
2187
|
+
|
|
2188
|
+
def _explained(n: dict) -> bool:
|
|
2189
|
+
sf = n.get("source_file")
|
|
2190
|
+
if not sf:
|
|
2191
|
+
return True
|
|
2192
|
+
own = new_ast_sources if _is_ast_tier(n) else new_sem_sources
|
|
2193
|
+
if sf in own or _norm_source_file(sf, _replace_root) in own:
|
|
2194
|
+
return True # replaced by this run's re-extract (same tier)
|
|
2195
|
+
return _prune_match(sf) # deliberately pruned this run
|
|
2196
|
+
|
|
2197
|
+
unexplained = [n for n in lost if not _explained(n)]
|
|
2198
|
+
if unexplained:
|
|
2199
|
+
raise ValueError(
|
|
2200
|
+
f"graphify: build_merge would drop {len(unexplained)} node(s) "
|
|
2201
|
+
f"from sources that were neither re-extracted nor pruned this "
|
|
2202
|
+
f"run (e.g. {unexplained[0].get('id')!r}); graph would go "
|
|
2203
|
+
f"{_disk_n} → {G.number_of_nodes()} nodes. "
|
|
2204
|
+
f"Pass prune_sources explicitly if you intend to remove them. (#479)"
|
|
2205
|
+
)
|
|
2206
|
+
|
|
2207
|
+
if unverified_semantic_shrink:
|
|
2208
|
+
G.graph["_unverified_semantic_shrink"] = unverified_semantic_shrink
|
|
2209
|
+
|
|
2210
|
+
return G
|
|
2211
|
+
|
|
2212
|
+
|
|
2213
|
+
def prefix_graph_for_global(
|
|
2214
|
+
G: nx.Graph, repo_tag: str, community_offset: int = 0
|
|
2215
|
+
) -> nx.Graph:
|
|
2216
|
+
"""Return a copy of G with all node IDs prefixed with repo_tag::.
|
|
2217
|
+
|
|
2218
|
+
Labels are preserved unchanged (for display). A 'local_id' attribute
|
|
2219
|
+
is added to each node so the original ID can be recovered. Edges and
|
|
2220
|
+
their directional attributes (_src/_tgt) are rewritten to match the new
|
|
2221
|
+
prefixed IDs. The 'repo' attribute is set on every node.
|
|
2222
|
+
|
|
2223
|
+
community_offset shifts each node's integer 'community' id into a shared
|
|
2224
|
+
id space and records the original in 'local_community': every input graph
|
|
2225
|
+
numbers its communities from 0, so ids carried across a merge unchanged
|
|
2226
|
+
collide and the aggregated community view fuses unrelated communities
|
|
2227
|
+
into one meta-node (#3014). 0 (the default) leaves communities untouched.
|
|
2228
|
+
"""
|
|
2229
|
+
relabel = {n: f"{repo_tag}::{n}" for n in G.nodes}
|
|
2230
|
+
H = nx.relabel_nodes(G, relabel, copy=True)
|
|
2231
|
+
for node, data in H.nodes(data=True):
|
|
2232
|
+
data["repo"] = repo_tag
|
|
2233
|
+
data.setdefault("local_id", node.split("::", 1)[1])
|
|
2234
|
+
cid = data.get("community")
|
|
2235
|
+
if community_offset and isinstance(cid, int):
|
|
2236
|
+
data["local_community"] = cid
|
|
2237
|
+
data["community"] = cid + community_offset
|
|
2238
|
+
for u, v, data in H.edges(data=True):
|
|
2239
|
+
if "_src" in data and data["_src"] in relabel:
|
|
2240
|
+
data["_src"] = relabel[data["_src"]]
|
|
2241
|
+
if "_tgt" in data and data["_tgt"] in relabel:
|
|
2242
|
+
data["_tgt"] = relabel[data["_tgt"]]
|
|
2243
|
+
# Out-of-band hyperedges must be relabeled with the nodes (#2484, after
|
|
2244
|
+
# @oleksii-tumanov's diagnosis in PR #1691): relabel_nodes copies graph
|
|
2245
|
+
# attrs by reference, so member ids kept their pre-prefix form and dangled
|
|
2246
|
+
# after a cross-repo merge. Rebuild the list (fresh dicts — the input
|
|
2247
|
+
# graph's list is shared with H) with member ids mapped through the same
|
|
2248
|
+
# relabel table, and prefix the hyperedge id itself so same-named
|
|
2249
|
+
# hyperedges from different repos cannot collide when merged.
|
|
2250
|
+
hyperedges = H.graph.get("hyperedges")
|
|
2251
|
+
if isinstance(hyperedges, list):
|
|
2252
|
+
rewritten = []
|
|
2253
|
+
for he in hyperedges:
|
|
2254
|
+
if isinstance(he, dict):
|
|
2255
|
+
he = dict(he)
|
|
2256
|
+
if isinstance(he.get("nodes"), list):
|
|
2257
|
+
he["nodes"] = [
|
|
2258
|
+
relabel.get(m, m) if _hashable(m) else m
|
|
2259
|
+
for m in he["nodes"]
|
|
2260
|
+
]
|
|
2261
|
+
if he.get("id"):
|
|
2262
|
+
he["id"] = f"{repo_tag}::{he['id']}"
|
|
2263
|
+
rewritten.append(he)
|
|
2264
|
+
H.graph["hyperedges"] = rewritten
|
|
2265
|
+
return H
|
|
2266
|
+
|
|
2267
|
+
|
|
2268
|
+
def distinct_repo_tags(graph_paths: "list[Path]") -> "list[str]":
|
|
2269
|
+
"""Return a unique, human-meaningful repo tag per input graph for merge-graphs.
|
|
2270
|
+
|
|
2271
|
+
The naive tag (the ``graphify-out`` parent dir name) is NOT unique across
|
|
2272
|
+
inputs: ``src/graphify-out`` and ``frontend/src/graphify-out`` both yield
|
|
2273
|
+
``src``. Prefixing both node sets with ``src::`` then makes same-stem nodes
|
|
2274
|
+
(a backend ``src/app.js`` and a frontend ``App.jsx``, both bare ``app``)
|
|
2275
|
+
collide, so ``nx.compose`` silently merges two unrelated entities and invents
|
|
2276
|
+
cross-runtime edges (#1729). Colliding tags are widened with their own parent
|
|
2277
|
+
dir (``frontend_src``), then an index suffix guarantees uniqueness so no two
|
|
2278
|
+
graphs ever share a prefix.
|
|
2279
|
+
"""
|
|
2280
|
+
repo_dirs = [p.parent.parent for p in graph_paths] # graphify-out/.. → repo dir
|
|
2281
|
+
tags = [d.name or "repo" for d in repo_dirs]
|
|
2282
|
+
if len(set(tags)) != len(tags):
|
|
2283
|
+
widened: list[str] = []
|
|
2284
|
+
for d in repo_dirs:
|
|
2285
|
+
parent = d.parent.name
|
|
2286
|
+
widened.append(f"{parent}_{d.name}" if parent and d.name else (d.name or "repo"))
|
|
2287
|
+
tags = widened
|
|
2288
|
+
seen: dict[str, int] = {}
|
|
2289
|
+
unique: list[str] = []
|
|
2290
|
+
for t in tags:
|
|
2291
|
+
seen[t] = seen.get(t, 0) + 1
|
|
2292
|
+
unique.append(t if seen[t] == 1 else f"{t}-{seen[t]}")
|
|
2293
|
+
return unique
|
|
2294
|
+
|
|
2295
|
+
|
|
2296
|
+
def prune_repo_from_graph(G: nx.Graph, repo_tag: str) -> int:
|
|
2297
|
+
"""Remove all nodes tagged with repo_tag from G in-place. Returns count removed."""
|
|
2298
|
+
to_remove = [n for n, d in G.nodes(data=True) if d.get("repo") == repo_tag]
|
|
2299
|
+
G.remove_nodes_from(to_remove)
|
|
2300
|
+
return len(to_remove)
|