graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/export.py
ADDED
|
@@ -0,0 +1,1349 @@
|
|
|
1
|
+
# write graph to HTML, JSON, SVG, GraphML, Obsidian vault, and Neo4j Cypher
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import hashlib
|
|
4
|
+
import html as _html
|
|
5
|
+
import json
|
|
6
|
+
import math
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
import shutil
|
|
10
|
+
import sys
|
|
11
|
+
from collections import Counter
|
|
12
|
+
from datetime import date
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
import networkx as nx
|
|
15
|
+
from networkx.readwrite import json_graph
|
|
16
|
+
from graphify.security import sanitize_label
|
|
17
|
+
from graphify.analyze import _node_community_map
|
|
18
|
+
from graphify.build import edge_data
|
|
19
|
+
from graphify.paths import stem_filename_budget, write_json_atomic, write_text_atomic
|
|
20
|
+
|
|
21
|
+
from graphify.exporters.graphdb import push_to_falkordb, push_to_neo4j # noqa: E402,F401
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# Artifacts worth preserving across rebuilds (non-regenerable without LLM or curation).
|
|
25
|
+
_BACKUP_ARTIFACTS = [
|
|
26
|
+
"graph.json",
|
|
27
|
+
"GRAPH_REPORT.md",
|
|
28
|
+
".graphify_labels.json",
|
|
29
|
+
".graphify_analysis.json",
|
|
30
|
+
"manifest.json",
|
|
31
|
+
".graphify_semantic_marker",
|
|
32
|
+
"cost.json",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def backup_if_protected(out_dir: Path) -> "Path | None":
|
|
37
|
+
"""Snapshot graph artifacts to a dated subfolder before an overwrite.
|
|
38
|
+
|
|
39
|
+
Triggers when graph.json exists AND either:
|
|
40
|
+
- .graphify_semantic_marker is present (graph cost real LLM tokens), or
|
|
41
|
+
- .graphify_labels.json contains at least one non-default community label
|
|
42
|
+
(graph has been curated by a human or skill).
|
|
43
|
+
|
|
44
|
+
Returns the backup folder path, or None if no backup was taken.
|
|
45
|
+
Never raises — backup failure prints a warning but never blocks the write.
|
|
46
|
+
Set GRAPHIFY_NO_BACKUP=1 to disable.
|
|
47
|
+
"""
|
|
48
|
+
if os.environ.get("GRAPHIFY_NO_BACKUP"):
|
|
49
|
+
return None
|
|
50
|
+
out = Path(out_dir)
|
|
51
|
+
if not (out / "graph.json").exists():
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
is_semantic = (out / ".graphify_semantic_marker").exists()
|
|
55
|
+
is_curated = False
|
|
56
|
+
labels_file = out / ".graphify_labels.json"
|
|
57
|
+
if labels_file.exists():
|
|
58
|
+
try:
|
|
59
|
+
labels = json.loads(labels_file.read_text(encoding="utf-8"))
|
|
60
|
+
is_curated = any(v != f"Community {k}" for k, v in labels.items())
|
|
61
|
+
except Exception:
|
|
62
|
+
pass
|
|
63
|
+
|
|
64
|
+
if not is_semantic and not is_curated:
|
|
65
|
+
return None
|
|
66
|
+
|
|
67
|
+
reason = "+".join(filter(None, ["semantic" if is_semantic else "", "curated" if is_curated else ""]))
|
|
68
|
+
today = date.today().isoformat()
|
|
69
|
+
backup_dir = out / today
|
|
70
|
+
graph_src = out / "graph.json"
|
|
71
|
+
|
|
72
|
+
# Skip re-copying if today's backup already has identical graph.json content.
|
|
73
|
+
# If content differs (graph changed since the last backup today), overwrite
|
|
74
|
+
# the backup in place — one folder per day, always the latest pre-overwrite state.
|
|
75
|
+
if backup_dir.exists() and (backup_dir / "graph.json").exists():
|
|
76
|
+
src_hash = hashlib.sha256(graph_src.read_bytes()).hexdigest()
|
|
77
|
+
bak_hash = hashlib.sha256((backup_dir / "graph.json").read_bytes()).hexdigest()
|
|
78
|
+
if src_hash == bak_hash:
|
|
79
|
+
return backup_dir # identical content, nothing to do
|
|
80
|
+
|
|
81
|
+
try:
|
|
82
|
+
backup_dir.mkdir(parents=True, exist_ok=True)
|
|
83
|
+
copied = 0
|
|
84
|
+
for name in _BACKUP_ARTIFACTS:
|
|
85
|
+
src = out / name
|
|
86
|
+
if src.exists():
|
|
87
|
+
try:
|
|
88
|
+
shutil.copy2(src, backup_dir / name)
|
|
89
|
+
copied += 1
|
|
90
|
+
except Exception:
|
|
91
|
+
pass
|
|
92
|
+
if copied:
|
|
93
|
+
print(f"[graphify] backed up {reason} graph ({copied} files) -> {backup_dir.name}/")
|
|
94
|
+
return backup_dir
|
|
95
|
+
except Exception as exc:
|
|
96
|
+
import sys
|
|
97
|
+
print(f"[graphify] warning: backup failed ({exc}) - continuing with overwrite", file=sys.stderr)
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
def _obsidian_tag(name: str) -> str:
|
|
101
|
+
r"""Sanitize a community name for use as an Obsidian tag.
|
|
102
|
+
|
|
103
|
+
Obsidian tags accept letters from any language plus digits, hyphens,
|
|
104
|
+
underscores and slashes; spaces and most punctuation are not allowed, and a
|
|
105
|
+
tag cannot be digits-only. ``\w`` is Unicode-aware in Python 3, so Hangul,
|
|
106
|
+
CJK, Cyrillic and accented Latin survive instead of being stripped (#2862):
|
|
107
|
+
an ASCII-only filter collapsed every non-Latin community label to
|
|
108
|
+
underscores, so every note in that community carried the same tag.
|
|
109
|
+
"""
|
|
110
|
+
tag = re.sub(r"[^\w\-/]", "", name.replace(" ", "_"))
|
|
111
|
+
if not tag.strip("_-/"):
|
|
112
|
+
return "unnamed" # label was punctuation only
|
|
113
|
+
if tag.isdigit():
|
|
114
|
+
return f"c{tag}" # Obsidian ignores digits-only tags
|
|
115
|
+
return tag
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _strip_diacritics(text: str | None) -> str:
|
|
119
|
+
import unicodedata
|
|
120
|
+
if not isinstance(text, str):
|
|
121
|
+
text = "" if text is None else str(text)
|
|
122
|
+
nfkd = unicodedata.normalize("NFKD", text)
|
|
123
|
+
return "".join(c for c in nfkd if not unicodedata.combining(c))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _yaml_str(s: str) -> str:
|
|
127
|
+
"""Escape a value for safe embedding in a YAML double-quoted scalar (F-009).
|
|
128
|
+
|
|
129
|
+
See `graphify.ingest._yaml_str` for the full rationale; duplicated here to
|
|
130
|
+
avoid pulling the URL-fetching `ingest` module into export's dependency
|
|
131
|
+
graph. Handles backslash, double-quote, all line breaks (\\n, \\r,
|
|
132
|
+
U+2028, U+2029), tab, NUL, and other C0/DEL control characters that
|
|
133
|
+
would otherwise let a hostile `source_file` / `community` / etc. break
|
|
134
|
+
out of the YAML scalar and inject sibling keys.
|
|
135
|
+
"""
|
|
136
|
+
if s is None:
|
|
137
|
+
return ""
|
|
138
|
+
out: list[str] = []
|
|
139
|
+
for ch in str(s):
|
|
140
|
+
cp = ord(ch)
|
|
141
|
+
if ch == "\\":
|
|
142
|
+
out.append("\\\\")
|
|
143
|
+
elif ch == '"':
|
|
144
|
+
out.append('\\"')
|
|
145
|
+
elif ch == "\n":
|
|
146
|
+
out.append("\\n")
|
|
147
|
+
elif ch == "\r":
|
|
148
|
+
out.append("\\r")
|
|
149
|
+
elif ch == "\t":
|
|
150
|
+
out.append("\\t")
|
|
151
|
+
elif ch == "\0":
|
|
152
|
+
out.append("\\0")
|
|
153
|
+
elif cp == 0x2028:
|
|
154
|
+
out.append("\\L")
|
|
155
|
+
elif cp == 0x2029:
|
|
156
|
+
out.append("\\P")
|
|
157
|
+
elif cp < 0x20 or cp == 0x7F:
|
|
158
|
+
out.append(f"\\x{cp:02x}")
|
|
159
|
+
else:
|
|
160
|
+
out.append(ch)
|
|
161
|
+
return "".join(out)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
from graphify.exporters.base import COMMUNITY_COLORS # noqa: E402,F401
|
|
165
|
+
|
|
166
|
+
from graphify.exporters.html import to_html # noqa: E402,F401
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
# Fallback scores for an edge that carries a confidence tier but no
|
|
170
|
+
# confidence_score. The INFERRED default was 0.5, which references/extraction-spec.md
|
|
171
|
+
# rules out in as many words — "never omit it, never use 0.5 as a default" — and
|
|
172
|
+
# which is not in the discrete INFERRED set {0.55, 0.65, 0.75, 0.85, 0.95} either.
|
|
173
|
+
# It is now the bottom of that set: a missing score is an absence of evidence
|
|
174
|
+
# about strength, so the honest fallback is the weakest value the rubric allows,
|
|
175
|
+
# not a midpoint that reads as a coin flip (#2813). Every AST emission site now
|
|
176
|
+
# supplies its own score, so this is a backstop rather than a routine path.
|
|
177
|
+
_CONFIDENCE_SCORE_DEFAULTS = {"EXTRACTED": 1.0, "INFERRED": 0.55, "AMBIGUOUS": 0.2}
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def attach_hyperedges(G: nx.Graph, hyperedges: list) -> None:
|
|
181
|
+
"""Store hyperedges in the graph's metadata dict."""
|
|
182
|
+
existing = G.graph.get("hyperedges", [])
|
|
183
|
+
# Skip id-less persisted entries when seeding the dedup set (#2775): the
|
|
184
|
+
# semantic extractor emits hyperedges with no `id` and build.py persists them
|
|
185
|
+
# verbatim, so a prior graph.json can contain id-less hyperedges. A hard
|
|
186
|
+
# `h["id"]` here raised `KeyError: 'id'` on every incremental re-extract,
|
|
187
|
+
# symmetric with the `.get("id")` guard the loop below already applies to the
|
|
188
|
+
# incoming set.
|
|
189
|
+
seen_ids = {h["id"] for h in existing if h.get("id")}
|
|
190
|
+
for h in hyperedges:
|
|
191
|
+
if h.get("id") and h["id"] not in seen_ids:
|
|
192
|
+
existing.append(h)
|
|
193
|
+
seen_ids.add(h["id"])
|
|
194
|
+
G.graph["hyperedges"] = existing
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _git_head(cwd: "str | Path | None" = None) -> str | None:
|
|
198
|
+
"""Return git HEAD for the repo containing ``cwd``, or None outside a repo.
|
|
199
|
+
|
|
200
|
+
``cwd`` selects the repository to ask, exactly as in watch._git_head
|
|
201
|
+
(#2316). Without it the command inherits the caller's working directory,
|
|
202
|
+
which stamps the *invoking* repo's commit when the graph being written
|
|
203
|
+
describes a different repo — provenance must come from the repo the graph
|
|
204
|
+
describes, so callers pass the graph's own location.
|
|
205
|
+
"""
|
|
206
|
+
import subprocess as _sp
|
|
207
|
+
try:
|
|
208
|
+
r = _sp.run(
|
|
209
|
+
["git", "rev-parse", "HEAD"], capture_output=True, text=True, timeout=3,
|
|
210
|
+
cwd=str(cwd) if cwd is not None else None,
|
|
211
|
+
)
|
|
212
|
+
return r.stdout.strip() if r.returncode == 0 else None
|
|
213
|
+
except Exception:
|
|
214
|
+
return None
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# Sentinel: an existing graph.json is present and non-empty but cannot be parsed
|
|
218
|
+
# into a node count (corrupt, mid-write, or structurally wrong). The caller must
|
|
219
|
+
# fail CLOSED on this — the same way to_json's #479 guard refuses to overwrite
|
|
220
|
+
# such a file — because we cannot prove the new graph isn't a silent shrink.
|
|
221
|
+
MALFORMED_GRAPH = object()
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def existing_graph_node_count(path: "str | Path"):
|
|
225
|
+
"""Node count of an existing graph.json.
|
|
226
|
+
|
|
227
|
+
Returns:
|
|
228
|
+
- an ``int`` node count when the file parses;
|
|
229
|
+
- ``None`` when there is verifiably nothing to protect — absent, empty, or
|
|
230
|
+
over the size cap (matching how :func:`to_json` lets the new graph
|
|
231
|
+
replace an empty/oversized file);
|
|
232
|
+
- :data:`MALFORMED_GRAPH` when the file is present and non-empty but
|
|
233
|
+
unparseable — the caller must treat this as fail-closed (refuse to
|
|
234
|
+
overwrite), mirroring to_json's #479 handling of a corrupt/mid-write file.
|
|
235
|
+
|
|
236
|
+
The raw ``--no-cluster`` write path uses this to apply the same #479 shrink
|
|
237
|
+
guard that :func:`to_json` applies inline for the clustered path.
|
|
238
|
+
"""
|
|
239
|
+
p = Path(path)
|
|
240
|
+
if not p.exists():
|
|
241
|
+
return None
|
|
242
|
+
from graphify.security import check_graph_file_size_cap
|
|
243
|
+
try:
|
|
244
|
+
check_graph_file_size_cap(p)
|
|
245
|
+
except Exception:
|
|
246
|
+
# Oversized: reading it to compare would be the DoS the cap guards against.
|
|
247
|
+
return None
|
|
248
|
+
try:
|
|
249
|
+
raw = p.read_text(encoding="utf-8")
|
|
250
|
+
except Exception:
|
|
251
|
+
# Present but unreadable: fail closed if it has bytes, else nothing to lose.
|
|
252
|
+
try:
|
|
253
|
+
return MALFORMED_GRAPH if p.stat().st_size > 0 else None
|
|
254
|
+
except Exception:
|
|
255
|
+
return None
|
|
256
|
+
if not raw.strip():
|
|
257
|
+
return None
|
|
258
|
+
try:
|
|
259
|
+
data = json.loads(raw)
|
|
260
|
+
except Exception:
|
|
261
|
+
return MALFORMED_GRAPH
|
|
262
|
+
nodes = data.get("nodes") if isinstance(data, dict) else None
|
|
263
|
+
return len(nodes) if isinstance(nodes, list) else MALFORMED_GRAPH
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, force: bool = False, built_at_commit: str | None = None, community_labels: dict[int, str] | None = None) -> bool:
|
|
267
|
+
# Safety check: refuse to silently shrink an existing graph (#479)
|
|
268
|
+
existing_path = Path(output_path)
|
|
269
|
+
if not force and existing_path.exists():
|
|
270
|
+
from graphify.security import check_graph_file_size_cap
|
|
271
|
+
try:
|
|
272
|
+
check_graph_file_size_cap(existing_path)
|
|
273
|
+
except Exception:
|
|
274
|
+
# Existing graph.json trips the size cap; reading it to compare would
|
|
275
|
+
# be the very DoS the cap guards against. Can't verify — let the new
|
|
276
|
+
# graph replace the oversized file.
|
|
277
|
+
oversized = True
|
|
278
|
+
else:
|
|
279
|
+
oversized = False
|
|
280
|
+
if not oversized:
|
|
281
|
+
try:
|
|
282
|
+
raw = existing_path.read_text(encoding="utf-8")
|
|
283
|
+
except Exception:
|
|
284
|
+
raw = ""
|
|
285
|
+
if not raw.strip():
|
|
286
|
+
# Empty/whitespace existing file (e.g. a freshly touched path):
|
|
287
|
+
# no nodes to lose, so any new graph is a growth — proceed.
|
|
288
|
+
existing_n = 0
|
|
289
|
+
else:
|
|
290
|
+
try:
|
|
291
|
+
existing_data = json.loads(raw)
|
|
292
|
+
existing_n = len(existing_data.get("nodes", []))
|
|
293
|
+
except Exception as exc:
|
|
294
|
+
# Non-empty but unparseable existing graph (corrupt or a
|
|
295
|
+
# mid-write): we cannot verify the new graph is not a silent
|
|
296
|
+
# shrink. Fail SAFE — refuse rather than overwrite. A
|
|
297
|
+
# fail-OPEN here (the prior behavior) is the silent data-loss
|
|
298
|
+
# path #479 exists to prevent: a transiently unreadable
|
|
299
|
+
# graph.json would let a partial rebuild clobber a good one.
|
|
300
|
+
import sys as _sys
|
|
301
|
+
print(
|
|
302
|
+
f"[graphify] WARNING: existing {existing_path} could not be "
|
|
303
|
+
f"read to verify the new graph is not smaller ({exc}). "
|
|
304
|
+
f"Refusing to overwrite; pass force=True to override.",
|
|
305
|
+
file=_sys.stderr,
|
|
306
|
+
)
|
|
307
|
+
return False
|
|
308
|
+
new_n = G.number_of_nodes()
|
|
309
|
+
if new_n < existing_n:
|
|
310
|
+
import sys as _sys
|
|
311
|
+
print(
|
|
312
|
+
f"[graphify] WARNING: new graph has {new_n} nodes but existing "
|
|
313
|
+
f"graph.json has {existing_n} (net -{existing_n - new_n}). "
|
|
314
|
+
f"Refusing to overwrite. Possible causes: missing chunk files from "
|
|
315
|
+
f"a previous session, or fuzzy dedup collapsed same-named symbols "
|
|
316
|
+
f"across files during an --update on an already-current graph. "
|
|
317
|
+
f"Run a full rebuild (/graphify .) to be safe, or pass force=True "
|
|
318
|
+
f"only if you have verified the reduction is legitimate.",
|
|
319
|
+
file=_sys.stderr,
|
|
320
|
+
)
|
|
321
|
+
return False
|
|
322
|
+
|
|
323
|
+
node_community = _node_community_map(communities)
|
|
324
|
+
_labels: dict[int, str] = {int(k): v for k, v in (community_labels or {}).items()}
|
|
325
|
+
try:
|
|
326
|
+
data = json_graph.node_link_data(G, edges="links")
|
|
327
|
+
except TypeError:
|
|
328
|
+
data = json_graph.node_link_data(G)
|
|
329
|
+
|
|
330
|
+
def _json_sort_key(item: dict) -> str:
|
|
331
|
+
return json.dumps(item, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
|
332
|
+
|
|
333
|
+
for node in data["nodes"]:
|
|
334
|
+
cid = node_community.get(node["id"])
|
|
335
|
+
node["community"] = cid
|
|
336
|
+
if cid is not None and _labels:
|
|
337
|
+
node["community_name"] = _labels.get(cid, f"Community {cid}")
|
|
338
|
+
node["norm_label"] = _strip_diacritics(node.get("label", "")).lower()
|
|
339
|
+
for link in data["links"]:
|
|
340
|
+
if "confidence_score" not in link:
|
|
341
|
+
conf = link.get("confidence", "EXTRACTED")
|
|
342
|
+
link["confidence_score"] = _CONFIDENCE_SCORE_DEFAULTS.get(conf, 1.0)
|
|
343
|
+
# Restore original edge direction. Undirected NetworkX storage may
|
|
344
|
+
# canonicalize endpoint order, flipping `calls` and other directional
|
|
345
|
+
# edges in graph.json. The build path stashes the true endpoints in
|
|
346
|
+
# _src/_tgt for exactly this purpose (#563).
|
|
347
|
+
true_src = link.pop("_src", None)
|
|
348
|
+
true_tgt = link.pop("_tgt", None)
|
|
349
|
+
if true_src is not None and true_tgt is not None:
|
|
350
|
+
link["source"] = true_src
|
|
351
|
+
link["target"] = true_tgt
|
|
352
|
+
# Canonicalize the key order WITHIN each node/link dict. node_link_data always
|
|
353
|
+
# appends the node key (`id`) at the end, so a node whose `id` was an inline
|
|
354
|
+
# attribute on a cold build (position varies) lands last after a read-rebuild
|
|
355
|
+
# (build_from_json consumes `id` as the pure node key). The values are
|
|
356
|
+
# identical either way, but the field order churns, so a byte-diff of two
|
|
357
|
+
# equivalent graph.json files is noisy and any position-sensitive consumer
|
|
358
|
+
# sees a spurious change on every round-trip. Emit a stable order — the
|
|
359
|
+
# identity keys first, then the remaining keys sorted — so the serialized
|
|
360
|
+
# form is invariant regardless of how the attribute was stored in memory.
|
|
361
|
+
def _canonical(item: dict, lead: tuple[str, ...]) -> dict:
|
|
362
|
+
leading = [k for k in lead if k in item]
|
|
363
|
+
rest = sorted(k for k in item if k not in leading)
|
|
364
|
+
return {k: item[k] for k in (*leading, *rest)}
|
|
365
|
+
|
|
366
|
+
data["nodes"] = [_canonical(n, ("id", "label")) for n in data["nodes"]]
|
|
367
|
+
data["links"] = [_canonical(link, ("source", "target", "relation")) for link in data["links"]]
|
|
368
|
+
data["nodes"].sort(key=_json_sort_key)
|
|
369
|
+
data["links"].sort(key=_json_sort_key)
|
|
370
|
+
if "hyperedges" not in getattr(G, "graph", {}):
|
|
371
|
+
# Hardening (#2485): a graph with NO hyperedges key at all was built by
|
|
372
|
+
# a path that never engaged hyperedge metadata — distinct from an
|
|
373
|
+
# intentional empty set ([], which build_from_json now stores
|
|
374
|
+
# explicitly after a full-wipeout revalidation). If the file on disk
|
|
375
|
+
# already holds a non-empty set, emptying it without a trace is silent
|
|
376
|
+
# data loss; warn loudly so the wipeout is attributable. We still write
|
|
377
|
+
# the graph's truth rather than preserving the stale set — resurrecting
|
|
378
|
+
# hyperedges whose members may no longer exist would reintroduce the
|
|
379
|
+
# dangling-member shape #1916 removed.
|
|
380
|
+
_prev_hyperedges = None
|
|
381
|
+
try:
|
|
382
|
+
if existing_path.exists():
|
|
383
|
+
from graphify.security import check_graph_file_size_cap
|
|
384
|
+
check_graph_file_size_cap(existing_path)
|
|
385
|
+
_prev = json.loads(existing_path.read_text(encoding="utf-8"))
|
|
386
|
+
if isinstance(_prev, dict):
|
|
387
|
+
_prev_hyperedges = _prev.get("hyperedges")
|
|
388
|
+
except Exception:
|
|
389
|
+
_prev_hyperedges = None
|
|
390
|
+
if _prev_hyperedges:
|
|
391
|
+
print(
|
|
392
|
+
f"[graphify] WARNING: graph carries no hyperedge metadata but "
|
|
393
|
+
f"{existing_path} already holds {len(_prev_hyperedges)} "
|
|
394
|
+
f"hyperedge(s); writing an empty set. Rebuild from the original "
|
|
395
|
+
f"extraction if this is unexpected.",
|
|
396
|
+
file=sys.stderr,
|
|
397
|
+
)
|
|
398
|
+
hyperedges = sorted(getattr(G, "graph", {}).get("hyperedges", []), key=_json_sort_key)
|
|
399
|
+
if isinstance(data.get("graph"), dict) and "hyperedges" in data["graph"]:
|
|
400
|
+
data["graph"]["hyperedges"] = hyperedges
|
|
401
|
+
data["hyperedges"] = hyperedges
|
|
402
|
+
# Fallback provenance comes from the repo the graph is being written INTO
|
|
403
|
+
# (output_path lives in <target>/graphify-out/), never the shell's cwd —
|
|
404
|
+
# the same cwd-anchoring mistake #2316 fixed for `update`.
|
|
405
|
+
commit = built_at_commit if built_at_commit is not None else _git_head(Path(output_path).resolve().parent)
|
|
406
|
+
if commit:
|
|
407
|
+
data["built_at_commit"] = commit
|
|
408
|
+
from graphify.paths import write_json_atomic
|
|
409
|
+
# Atomic write: a crash/ENOSPC mid-write must not truncate a good graph.json.
|
|
410
|
+
write_json_atomic(output_path, data, indent=2)
|
|
411
|
+
return True
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def prune_dangling_edges(graph_data: dict) -> tuple[dict, int]:
|
|
415
|
+
"""Remove edges whose source or target node is not in the node set.
|
|
416
|
+
|
|
417
|
+
Returns the cleaned graph_data dict and the number of pruned edges.
|
|
418
|
+
"""
|
|
419
|
+
node_ids = {n["id"] for n in graph_data["nodes"]}
|
|
420
|
+
links_key = "links" if "links" in graph_data else "edges"
|
|
421
|
+
before = len(graph_data[links_key])
|
|
422
|
+
graph_data[links_key] = [
|
|
423
|
+
e for e in graph_data[links_key]
|
|
424
|
+
if e["source"] in node_ids and e["target"] in node_ids
|
|
425
|
+
]
|
|
426
|
+
return graph_data, before - len(graph_data[links_key])
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _cypher_escape(s: str) -> str:
|
|
430
|
+
"""Escape a string for safe embedding in a Cypher single-quoted literal.
|
|
431
|
+
|
|
432
|
+
Handles all characters that could prematurely terminate the literal or
|
|
433
|
+
inject control sequences:
|
|
434
|
+
- `\\` and `'` (literal terminators)
|
|
435
|
+
- newlines/CRs (would break the per-line statement framing)
|
|
436
|
+
- NUL/control bytes (defensive — Neo4j errors on raw NULs)
|
|
437
|
+
|
|
438
|
+
Also strips any leading/trailing whitespace that would let an attacker
|
|
439
|
+
break the `;`-terminated statement boundary used by `cypher-shell`.
|
|
440
|
+
Closing `}` and `)` are NOT special inside a single-quoted Cypher string,
|
|
441
|
+
so escaping the quote and backslash correctly is sufficient (a `}` inside
|
|
442
|
+
a properly-closed `'...'` literal is just a character) — but we previously
|
|
443
|
+
missed `\\n` / `\\r` which DO let a payload break out of the statement
|
|
444
|
+
line and inject a fresh MATCH/DELETE on the following line. See F-008.
|
|
445
|
+
"""
|
|
446
|
+
# First normalise: drop NUL and other C0 control chars except tab.
|
|
447
|
+
s = "".join(ch for ch in s if ch >= " " or ch == "\t")
|
|
448
|
+
return (
|
|
449
|
+
s.replace("\\", "\\\\")
|
|
450
|
+
.replace("'", "\\'")
|
|
451
|
+
.replace("\n", "\\n")
|
|
452
|
+
.replace("\r", "\\r")
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
# Restrict identifier-position values (labels and relationship types are NOT
|
|
457
|
+
# quoted in Cypher and so cannot be safely escaped — they must be allowlisted).
|
|
458
|
+
_CYPHER_IDENT_RE = re.compile(r"[^A-Za-z0-9_]")
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _cypher_label(raw: str, fallback: str) -> str:
|
|
462
|
+
"""Sanitise a value used in identifier position (node label / rel type).
|
|
463
|
+
|
|
464
|
+
Cypher does not provide a way to escape `:Foo` label syntax, so we must
|
|
465
|
+
strip everything except `[A-Za-z0-9_]` and require the result to start
|
|
466
|
+
with a letter; otherwise we fall back to a safe constant.
|
|
467
|
+
"""
|
|
468
|
+
cleaned = _CYPHER_IDENT_RE.sub("", raw or "")
|
|
469
|
+
if not cleaned or not cleaned[0].isalpha():
|
|
470
|
+
return fallback
|
|
471
|
+
return cleaned
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def to_cypher(G: nx.Graph, output_path: str) -> None:
|
|
475
|
+
lines = ["// Neo4j Cypher import - generated by /graphify", ""]
|
|
476
|
+
for node_id, data in G.nodes(data=True):
|
|
477
|
+
label = _cypher_escape(data.get("label", node_id))
|
|
478
|
+
node_id_esc = _cypher_escape(node_id)
|
|
479
|
+
ftype = _cypher_label(
|
|
480
|
+
(data.get("file_type", "unknown") or "unknown").capitalize(),
|
|
481
|
+
"Entity",
|
|
482
|
+
)
|
|
483
|
+
lines.append(f"MERGE (n:{ftype} {{id: '{node_id_esc}', label: '{label}'}});")
|
|
484
|
+
lines.append("")
|
|
485
|
+
for u, v, data in G.edges(data=True):
|
|
486
|
+
rel = _cypher_label(
|
|
487
|
+
(data.get("relation", "RELATES_TO") or "RELATES_TO").upper(),
|
|
488
|
+
"RELATES_TO",
|
|
489
|
+
)
|
|
490
|
+
conf = _cypher_escape(data.get("confidence", "EXTRACTED"))
|
|
491
|
+
u_esc = _cypher_escape(u)
|
|
492
|
+
v_esc = _cypher_escape(v)
|
|
493
|
+
lines.append(
|
|
494
|
+
f"MATCH (a {{id: '{u_esc}'}}), (b {{id: '{v_esc}'}}) "
|
|
495
|
+
f"MERGE (a)-[:{rel} {{confidence: '{conf}'}}]->(b);"
|
|
496
|
+
)
|
|
497
|
+
with open(output_path, "w", encoding="utf-8") as f: # nosec
|
|
498
|
+
f.write("\n".join(lines))
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
# Keep backward-compatible alias - skill.md calls generate_html
|
|
502
|
+
generate_html = to_html
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
# Characters XML 1.0 cannot carry: the C0 controls except tab, LF and CR.
|
|
506
|
+
_XML_ILLEGAL_RE = re.compile("[\x00-\x08\x0b\x0c\x0e-\x1f]")
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def _strip_xml_illegal(s: str) -> str:
|
|
510
|
+
"""Drop characters XML 1.0 cannot represent, leaving tab/LF/CR intact.
|
|
511
|
+
|
|
512
|
+
``nx.write_graphml`` raises ``ValueError("All strings must be XML
|
|
513
|
+
compatible: Unicode or ASCII, no NULL bytes or control characters")`` on any
|
|
514
|
+
of them and aborts the whole export over a single label. Labels arrive
|
|
515
|
+
unfiltered from the corpus, so this is ordinary content rather than hostile
|
|
516
|
+
input: an ANSI escape in a markdown heading pasted from a terminal capture,
|
|
517
|
+
or the form feed some Python/Emacs sources use as a section separator
|
|
518
|
+
(#2897).
|
|
519
|
+
"""
|
|
520
|
+
return _XML_ILLEGAL_RE.sub("", s)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
# C0 controls and DEL, folded to a space when building a filename stem. Windows
|
|
524
|
+
# rejects them in a path outright with OSError EINVAL, so one of them in a label
|
|
525
|
+
# aborted a whole Obsidian vault export; POSIX would accept the name but leave a
|
|
526
|
+
# note nothing can comfortably open (#2897).
|
|
527
|
+
_CONTROL_TO_SPACE_RE = re.compile("[\x00-\x1f\x7f]")
|
|
528
|
+
|
|
529
|
+
|
|
530
|
+
def _cap_filename(s: str, limit: int = 200) -> str:
|
|
531
|
+
"""Cap a filename stem to ``limit`` UTF-8 bytes so it stays under the 255-byte
|
|
532
|
+
filesystem limit even after the ``.md`` extension and dedup suffix are added
|
|
533
|
+
(#1094). The cap is on BYTES, not chars, because a label of multibyte
|
|
534
|
+
characters (CJK, accented) can exceed 255 bytes well under 255 chars. When
|
|
535
|
+
truncation happens, an 8-char hash of the full label is appended so two
|
|
536
|
+
distinct labels sharing a long prefix produce distinct, deterministic
|
|
537
|
+
filenames instead of colliding."""
|
|
538
|
+
b = s.encode("utf-8")
|
|
539
|
+
if len(b) <= limit:
|
|
540
|
+
return s
|
|
541
|
+
digest = hashlib.sha1(s.encode("utf-8")).hexdigest()[:8] # nosec - not security
|
|
542
|
+
keep = limit - 9 # "_" + 8 hex chars
|
|
543
|
+
truncated = b[:keep].decode("utf-8", "ignore") # "ignore" drops a split trailing char
|
|
544
|
+
return f"{truncated}_{digest}"
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
# A frontmatter tag entry in graphify's own namespace, e.g. " - graphify/document".
|
|
548
|
+
_GRAPHIFY_TAG_RE = re.compile(r"^\s*-\s+graphify/\S")
|
|
549
|
+
|
|
550
|
+
# Frontmatter sits at the very top of a note; reading this much is enough to see
|
|
551
|
+
# the whole block without pulling a large note into memory.
|
|
552
|
+
_NOTE_FRONTMATTER_PROBE_BYTES = 4096
|
|
553
|
+
|
|
554
|
+
# Community notes carry no frontmatter; graphify identifies its own by the
|
|
555
|
+
# Dataview query it writes into every one of them.
|
|
556
|
+
_COMMUNITY_QUERY_MARKER = "FROM #community/"
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _is_graphify_note(path: Path) -> bool:
|
|
560
|
+
"""Whether a vault note carries graphify's own frontmatter signature.
|
|
561
|
+
|
|
562
|
+
Every note graphify writes opens with a YAML frontmatter block tagging it in
|
|
563
|
+
the ``graphify/`` namespace::
|
|
564
|
+
|
|
565
|
+
---
|
|
566
|
+
source_file: "d0.md"
|
|
567
|
+
tags:
|
|
568
|
+
- graphify/document
|
|
569
|
+
- graphify/EXTRACTED
|
|
570
|
+
---
|
|
571
|
+
|
|
572
|
+
Only that block is inspected, and only a tag entry inside it counts — a
|
|
573
|
+
user's note that merely mentions graphify in its prose is not adopted.
|
|
574
|
+
|
|
575
|
+
Community overview notes are recognised separately: they carry no
|
|
576
|
+
frontmatter at all, so they are identified by graphify's own filename prefix
|
|
577
|
+
together with the Dataview query it writes into the body. Requiring both
|
|
578
|
+
keeps a user's own ``_COMMUNITY_*.md`` from being adopted on the name alone.
|
|
579
|
+
"""
|
|
580
|
+
try:
|
|
581
|
+
with path.open("r", encoding="utf-8", errors="replace") as fh:
|
|
582
|
+
head = fh.read(_NOTE_FRONTMATTER_PROBE_BYTES)
|
|
583
|
+
except OSError:
|
|
584
|
+
return False
|
|
585
|
+
if path.name.startswith(_COMMUNITY_PREFIX) and _COMMUNITY_QUERY_MARKER in head:
|
|
586
|
+
return True
|
|
587
|
+
if not head.startswith("---"):
|
|
588
|
+
return False
|
|
589
|
+
for line in head.splitlines()[1:]:
|
|
590
|
+
if line.strip() == "---":
|
|
591
|
+
return False # frontmatter closed without a graphify tag
|
|
592
|
+
if _GRAPHIFY_TAG_RE.match(line):
|
|
593
|
+
return True
|
|
594
|
+
return False
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
def _adopt_pre_manifest_notes(out: Path) -> set[str]:
|
|
598
|
+
"""Names of notes in *out* that graphify itself wrote before manifests existed.
|
|
599
|
+
|
|
600
|
+
Deliberately limited to top-level ``*.md``: those are the only files graphify
|
|
601
|
+
can identify as its own from their content. ``.obsidian/graph.json`` is NOT
|
|
602
|
+
adopted — graphify writes one, but so does Obsidian, and with no manifest
|
|
603
|
+
there is no way to tell whose it is. Leaving it unowned keeps the
|
|
604
|
+
conservative behaviour for the one file where guessing wrong would cost the
|
|
605
|
+
user their own vault configuration.
|
|
606
|
+
"""
|
|
607
|
+
try:
|
|
608
|
+
candidates = sorted(out.glob("*.md"))
|
|
609
|
+
except OSError:
|
|
610
|
+
return set()
|
|
611
|
+
return {p.name for p in candidates if _is_graphify_note(p)}
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def _obsidian_safe_stem(label: str, limit: int = 200) -> str:
|
|
615
|
+
"""Filename stem for an Obsidian note / canvas card from a node label.
|
|
616
|
+
|
|
617
|
+
Strips filesystem-unsafe characters, a trailing ``.md``-family extension
|
|
618
|
+
(so ``CLAUDE.md`` does not become ``CLAUDE.md.md``), and a leading ``.`` —
|
|
619
|
+
Obsidian hides every note whose name starts with a dot, so ``.env.md``
|
|
620
|
+
would be written but invisible in the UI (#2205). The ``dot-`` prefix keeps
|
|
621
|
+
the name recognizable; H1 / frontmatter still carry the true label.
|
|
622
|
+
"""
|
|
623
|
+
cleaned = re.sub(
|
|
624
|
+
r'[\\/*?:"<>|#^[\]]',
|
|
625
|
+
"",
|
|
626
|
+
# CR/LF were already folded to spaces here; every other C0 control now
|
|
627
|
+
# goes the same way. They are not merely awkward in a filename — Windows
|
|
628
|
+
# rejects them outright, so a single one aborted the whole vault export
|
|
629
|
+
# rather than spoiling one note (#2897).
|
|
630
|
+
_CONTROL_TO_SPACE_RE.sub(" ", label),
|
|
631
|
+
).strip()
|
|
632
|
+
cleaned = re.sub(r"\.(md|mdx|qmd|markdown)$", "", cleaned, flags=re.IGNORECASE)
|
|
633
|
+
# Obsidian treats a leading-dot filename as a hidden file (#2205). Only
|
|
634
|
+
# prefix when something nameable remains after the dots: an all-dots label
|
|
635
|
+
# like "..." would otherwise become the meaningless stem "dot-" instead of
|
|
636
|
+
# falling through to the "unnamed" guard below (#1409).
|
|
637
|
+
if cleaned.startswith(".") and re.search(r"\w", cleaned.lstrip("."), flags=re.UNICODE):
|
|
638
|
+
cleaned = "dot-" + cleaned.lstrip(".")
|
|
639
|
+
# A stem of only punctuation (e.g. "@", "*", "#") survives the unsafe-char
|
|
640
|
+
# strip above but is empty once a downstream tool re-slugs on word chars
|
|
641
|
+
# (e.g. qmd's handelize() reduces "@" -> "" and raises, aborting the whole
|
|
642
|
+
# `qmd update`). Require at least one word char; else fall back so we never
|
|
643
|
+
# emit a "@.md"-style filename. (#1409)
|
|
644
|
+
if not re.search(r"\w", cleaned, flags=re.UNICODE):
|
|
645
|
+
return "unnamed"
|
|
646
|
+
return _cap_filename(cleaned, limit)
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
# Room _dedup_node_filenames / the community loop need for a collision suffix
|
|
650
|
+
# ("_1" … "_9999") appended AFTER the stem was capped. The suffix is technically
|
|
651
|
+
# unbounded, but 5 chars ("_" + 4 digits) covers ~10k identical stems, far past
|
|
652
|
+
# anything real; sizing it to 3 digits let a 1000th collision overrun MAX_PATH.
|
|
653
|
+
_DEDUP_SUFFIX_RESERVE = 5
|
|
654
|
+
|
|
655
|
+
# Prefix the community overview notes carry ("_COMMUNITY_Backend.md").
|
|
656
|
+
_COMMUNITY_PREFIX = "_COMMUNITY_"
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def _dedup_node_filenames(G: nx.Graph, safe_name) -> dict[str, str]:
|
|
660
|
+
"""Map each node_id to a unique note filename, appending a numeric suffix on
|
|
661
|
+
collision. The collision set is keyed on the lowercased name so two labels
|
|
662
|
+
differing only by case (e.g. "References" vs "references") still get distinct
|
|
663
|
+
filenames - on case-insensitive filesystems (macOS/APFS, Windows/NTFS) they
|
|
664
|
+
would otherwise resolve to one path and silently overwrite each other on disk.
|
|
665
|
+
The suffixed candidate is itself re-checked, so a generated "base_1" never
|
|
666
|
+
silently overwrites a node whose literal label is already "base_1"."""
|
|
667
|
+
node_filenames: dict[str, str] = {}
|
|
668
|
+
used: set[str] = set()
|
|
669
|
+
for node_id, data in G.nodes(data=True):
|
|
670
|
+
base = safe_name(data.get("label", node_id))
|
|
671
|
+
candidate = base
|
|
672
|
+
n = 1
|
|
673
|
+
while candidate.lower() in used:
|
|
674
|
+
candidate = f"{base}_{n}"
|
|
675
|
+
n += 1
|
|
676
|
+
used.add(candidate.lower())
|
|
677
|
+
node_filenames[node_id] = candidate
|
|
678
|
+
return node_filenames
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def to_obsidian(
|
|
682
|
+
G: nx.Graph,
|
|
683
|
+
communities: dict[int, list[str]],
|
|
684
|
+
output_dir: str,
|
|
685
|
+
community_labels: dict[int, str] | None = None,
|
|
686
|
+
cohesion: dict[int, float] | None = None,
|
|
687
|
+
) -> int:
|
|
688
|
+
"""Export graph as an Obsidian vault - one .md file per node with [[wikilinks]],
|
|
689
|
+
plus one _COMMUNITY_name.md overview note per community (sorted to top by underscore prefix).
|
|
690
|
+
|
|
691
|
+
Open the output directory as a vault in Obsidian to get an interactive
|
|
692
|
+
graph view with community colors and full-text search over node metadata.
|
|
693
|
+
|
|
694
|
+
Returns the number of node notes + community notes written.
|
|
695
|
+
"""
|
|
696
|
+
out = Path(output_dir)
|
|
697
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
698
|
+
|
|
699
|
+
# #1506: when the export target is an existing Obsidian vault (a user pointed
|
|
700
|
+
# --obsidian-dir at one), we must not clobber the user's own notes or their
|
|
701
|
+
# .obsidian/ config. Track the files graphify owns in a manifest; a pre-existing
|
|
702
|
+
# file NOT in the manifest is the user's and is never overwritten.
|
|
703
|
+
_manifest_path = out / ".graphify_obsidian_manifest.json"
|
|
704
|
+
try:
|
|
705
|
+
_owned: set[str] = set(json.loads(_manifest_path.read_text(encoding="utf-8")).get("files", []))
|
|
706
|
+
_manifest_existed = True
|
|
707
|
+
except (OSError, ValueError):
|
|
708
|
+
_owned = set()
|
|
709
|
+
_manifest_existed = False
|
|
710
|
+
if not _manifest_existed:
|
|
711
|
+
# A vault written before the manifest existed has no record of what
|
|
712
|
+
# graphify owns, so every note it wrote last time reads as the user's and
|
|
713
|
+
# is skipped. The re-export then writes fresh notes BESIDE the stale ones
|
|
714
|
+
# and the vault carries two generations, with a warning claiming graphify
|
|
715
|
+
# "did not create" files it did (#2863). Adopt the notes that carry
|
|
716
|
+
# graphify's own frontmatter, once, so the manifest starts out honest.
|
|
717
|
+
_owned |= _adopt_pre_manifest_notes(out)
|
|
718
|
+
_written: list[str] = []
|
|
719
|
+
_skipped: list[str] = []
|
|
720
|
+
|
|
721
|
+
def _owned_write(rel_name: str, content: str) -> bool:
|
|
722
|
+
"""Write a graphify-owned file, refusing to overwrite a pre-existing file
|
|
723
|
+
graphify didn't create. Returns True if written."""
|
|
724
|
+
target = out / rel_name
|
|
725
|
+
if target.exists() and rel_name not in _owned:
|
|
726
|
+
_skipped.append(rel_name)
|
|
727
|
+
return False
|
|
728
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
729
|
+
write_text_atomic(target, content)
|
|
730
|
+
_written.append(rel_name)
|
|
731
|
+
return True
|
|
732
|
+
|
|
733
|
+
node_community = _node_community_map(communities)
|
|
734
|
+
|
|
735
|
+
# Cap stems against THIS vault's path, not just NAME_MAX: on Windows the
|
|
736
|
+
# 200-byte default plus an ordinary vault directory overruns MAX_PATH and
|
|
737
|
+
# every note write raises FileNotFoundError (#2655). No-op on POSIX.
|
|
738
|
+
_stem_limit = stem_filename_budget(out, reserve=_DEDUP_SUFFIX_RESERVE)
|
|
739
|
+
|
|
740
|
+
# Map node_id → safe filename so wikilinks stay consistent.
|
|
741
|
+
# Deduplicate: if two nodes produce the same filename, append a numeric suffix.
|
|
742
|
+
node_filename = _dedup_node_filenames(
|
|
743
|
+
G, lambda label: _obsidian_safe_stem(label, _stem_limit)
|
|
744
|
+
)
|
|
745
|
+
|
|
746
|
+
# Helper: compute dominant confidence for a node across all its edges
|
|
747
|
+
def _dominant_confidence(node_id: str) -> str:
|
|
748
|
+
confs = []
|
|
749
|
+
for u, v, edata in G.edges(node_id, data=True):
|
|
750
|
+
confs.append(edata.get("confidence", "EXTRACTED"))
|
|
751
|
+
if not confs:
|
|
752
|
+
return "EXTRACTED"
|
|
753
|
+
return Counter(confs).most_common(1)[0][0]
|
|
754
|
+
|
|
755
|
+
# Map file_type → graphify tag
|
|
756
|
+
_FTYPE_TAG = {
|
|
757
|
+
"code": "graphify/code",
|
|
758
|
+
"document": "graphify/document",
|
|
759
|
+
"paper": "graphify/paper",
|
|
760
|
+
"image": "graphify/image",
|
|
761
|
+
}
|
|
762
|
+
|
|
763
|
+
# Write one .md file per node
|
|
764
|
+
node_notes_written = 0
|
|
765
|
+
for node_id, data in G.nodes(data=True):
|
|
766
|
+
label = data.get("label", node_id)
|
|
767
|
+
cid = node_community.get(node_id)
|
|
768
|
+
community_name = (
|
|
769
|
+
community_labels.get(cid, f"Community {cid}")
|
|
770
|
+
if community_labels and cid is not None
|
|
771
|
+
else f"Community {cid}"
|
|
772
|
+
)
|
|
773
|
+
|
|
774
|
+
# Build tags for this node
|
|
775
|
+
ftype = data.get("file_type", "")
|
|
776
|
+
ftype_tag = _FTYPE_TAG.get(ftype, f"graphify/{ftype}" if ftype else "graphify/document")
|
|
777
|
+
dom_conf = _dominant_confidence(node_id)
|
|
778
|
+
conf_tag = f"graphify/{dom_conf}"
|
|
779
|
+
comm_tag = f"community/{_obsidian_tag(community_name)}"
|
|
780
|
+
node_tags = [ftype_tag, conf_tag, comm_tag]
|
|
781
|
+
|
|
782
|
+
lines: list[str] = []
|
|
783
|
+
|
|
784
|
+
# YAML frontmatter - readable in Obsidian's properties panel.
|
|
785
|
+
# All scalars pass through _yaml_str so a hostile source_file or
|
|
786
|
+
# community label cannot break out and inject sibling keys (F-009).
|
|
787
|
+
lines += [
|
|
788
|
+
"---",
|
|
789
|
+
f'source_file: "{_yaml_str(data.get("source_file", ""))}"',
|
|
790
|
+
f'type: "{_yaml_str(ftype)}"',
|
|
791
|
+
f'community: "{_yaml_str(community_name)}"',
|
|
792
|
+
]
|
|
793
|
+
if data.get("source_location"):
|
|
794
|
+
lines.append(f'location: "{_yaml_str(str(data["source_location"]))}"')
|
|
795
|
+
# Add tags list to frontmatter
|
|
796
|
+
lines.append("tags:")
|
|
797
|
+
for tag in node_tags:
|
|
798
|
+
lines.append(f" - {tag}")
|
|
799
|
+
lines += ["---", "", f"# {label}", ""]
|
|
800
|
+
|
|
801
|
+
# Outgoing edges as wikilinks
|
|
802
|
+
neighbors = list(G.neighbors(node_id))
|
|
803
|
+
if neighbors:
|
|
804
|
+
lines.append("## Connections")
|
|
805
|
+
for neighbor in sorted(neighbors, key=lambda n: G.nodes[n].get("label", n)):
|
|
806
|
+
edata = edge_data(G, node_id, neighbor)
|
|
807
|
+
neighbor_label = node_filename[neighbor]
|
|
808
|
+
relation = edata.get("relation", "")
|
|
809
|
+
confidence = edata.get("confidence", "EXTRACTED")
|
|
810
|
+
lines.append(f"- [[{neighbor_label}]] - `{relation}` [{confidence}]")
|
|
811
|
+
lines.append("")
|
|
812
|
+
|
|
813
|
+
# Inline tags at bottom of note body (for Obsidian tag panel)
|
|
814
|
+
inline_tags = " ".join(f"#{t}" for t in node_tags)
|
|
815
|
+
lines.append(inline_tags)
|
|
816
|
+
|
|
817
|
+
fname = node_filename[node_id] + ".md"
|
|
818
|
+
if _owned_write(fname, "\n".join(lines)):
|
|
819
|
+
node_notes_written += 1
|
|
820
|
+
|
|
821
|
+
# Write one _COMMUNITY_name.md overview note per community
|
|
822
|
+
# Build inter-community edge counts for "Connections to other communities"
|
|
823
|
+
inter_community_edges: dict[int, dict[int, int]] = {}
|
|
824
|
+
for cid in communities:
|
|
825
|
+
inter_community_edges[cid] = {}
|
|
826
|
+
for u, v in G.edges():
|
|
827
|
+
cu = node_community.get(u)
|
|
828
|
+
cv = node_community.get(v)
|
|
829
|
+
if cu is not None and cv is not None and cu != cv:
|
|
830
|
+
inter_community_edges.setdefault(cu, {})
|
|
831
|
+
inter_community_edges.setdefault(cv, {})
|
|
832
|
+
inter_community_edges[cu][cv] = inter_community_edges[cu].get(cv, 0) + 1
|
|
833
|
+
inter_community_edges[cv][cu] = inter_community_edges[cv].get(cu, 0) + 1
|
|
834
|
+
|
|
835
|
+
# Precompute per-node community reach (number of distinct communities a node connects to)
|
|
836
|
+
def _community_reach(node_id: str) -> int:
|
|
837
|
+
neighbor_cids = {
|
|
838
|
+
node_community[nb]
|
|
839
|
+
for nb in G.neighbors(node_id)
|
|
840
|
+
if nb in node_community and node_community[nb] != node_community.get(node_id)
|
|
841
|
+
}
|
|
842
|
+
return len(neighbor_cids)
|
|
843
|
+
|
|
844
|
+
def _community_name(cid) -> str:
|
|
845
|
+
return (
|
|
846
|
+
community_labels.get(cid, f"Community {cid}")
|
|
847
|
+
if community_labels and cid is not None
|
|
848
|
+
else f"Community {cid}"
|
|
849
|
+
)
|
|
850
|
+
|
|
851
|
+
# One case-folded-deduped filename per community, computed once so the note we
|
|
852
|
+
# write and every [[_COMMUNITY_...]] cross-reference resolve to the same file.
|
|
853
|
+
# Two community labels differing only by case (e.g. LLM labels "API" vs "Api")
|
|
854
|
+
# would otherwise overwrite each other on case-insensitive filesystems - and
|
|
855
|
+
# this path had no dedup at all, so even same-case duplicate labels collided.
|
|
856
|
+
community_filename: dict = {}
|
|
857
|
+
used_community: set[str] = set()
|
|
858
|
+
# The community stem carries the "_COMMUNITY_" prefix on top of the dedup
|
|
859
|
+
# suffix, so it gets that much less of the MAX_PATH window (#2655).
|
|
860
|
+
_community_stem_limit = stem_filename_budget(
|
|
861
|
+
out, reserve=_DEDUP_SUFFIX_RESERVE + len(_COMMUNITY_PREFIX)
|
|
862
|
+
)
|
|
863
|
+
for cid in communities:
|
|
864
|
+
base = f"{_COMMUNITY_PREFIX}{_obsidian_safe_stem(_community_name(cid), _community_stem_limit)}"
|
|
865
|
+
candidate = base
|
|
866
|
+
n = 1
|
|
867
|
+
while candidate.lower() in used_community:
|
|
868
|
+
candidate = f"{base}_{n}"
|
|
869
|
+
n += 1
|
|
870
|
+
used_community.add(candidate.lower())
|
|
871
|
+
community_filename[cid] = candidate
|
|
872
|
+
|
|
873
|
+
community_notes_written = 0
|
|
874
|
+
for cid, all_members in communities.items():
|
|
875
|
+
community_name = _community_name(cid)
|
|
876
|
+
# A community's member list can contain ids with no backing node in G
|
|
877
|
+
# (e.g. pruned nodes, stale community assignments from a prior run, or
|
|
878
|
+
# synthesized/merge-artifact ids). Dereferencing those via G.nodes[n] or
|
|
879
|
+
# node_filename[n] raises KeyError and aborts the whole vault export, so
|
|
880
|
+
# skip dangling members rather than crashing (issue #1236).
|
|
881
|
+
members = [m for m in all_members if m in G and m in node_filename]
|
|
882
|
+
n_members = len(members)
|
|
883
|
+
coh_value = cohesion.get(cid) if cohesion else None
|
|
884
|
+
|
|
885
|
+
lines: list[str] = []
|
|
886
|
+
|
|
887
|
+
# YAML frontmatter
|
|
888
|
+
lines.append("---")
|
|
889
|
+
lines.append("type: community")
|
|
890
|
+
if coh_value is not None:
|
|
891
|
+
lines.append(f"cohesion: {coh_value:.2f}")
|
|
892
|
+
lines.append(f"members: {n_members}")
|
|
893
|
+
lines.append("---")
|
|
894
|
+
lines.append("")
|
|
895
|
+
lines.append(f"# {community_name}")
|
|
896
|
+
lines.append("")
|
|
897
|
+
|
|
898
|
+
# Cohesion + member count summary
|
|
899
|
+
if coh_value is not None:
|
|
900
|
+
cohesion_desc = (
|
|
901
|
+
"tightly connected" if coh_value >= 0.7
|
|
902
|
+
else "moderately connected" if coh_value >= 0.4
|
|
903
|
+
else "loosely connected"
|
|
904
|
+
)
|
|
905
|
+
lines.append(f"**Cohesion:** {coh_value:.2f} - {cohesion_desc}")
|
|
906
|
+
lines.append(f"**Members:** {n_members} nodes")
|
|
907
|
+
lines.append("")
|
|
908
|
+
|
|
909
|
+
# Members section
|
|
910
|
+
lines.append("## Members")
|
|
911
|
+
for node_id in sorted(members, key=lambda n: G.nodes[n].get("label", n)):
|
|
912
|
+
data = G.nodes[node_id]
|
|
913
|
+
node_label = node_filename[node_id]
|
|
914
|
+
ftype = data.get("file_type", "")
|
|
915
|
+
source = data.get("source_file", "")
|
|
916
|
+
entry = f"- [[{node_label}]]"
|
|
917
|
+
if ftype:
|
|
918
|
+
entry += f" - {ftype}"
|
|
919
|
+
if source:
|
|
920
|
+
entry += f" - {source}"
|
|
921
|
+
lines.append(entry)
|
|
922
|
+
lines.append("")
|
|
923
|
+
|
|
924
|
+
# Dataview live query (improvement 2)
|
|
925
|
+
comm_tag_name = _obsidian_tag(community_name)
|
|
926
|
+
lines.append("## Live Query (requires Dataview plugin)")
|
|
927
|
+
lines.append("")
|
|
928
|
+
lines.append("```dataview")
|
|
929
|
+
lines.append(f"TABLE source_file, type FROM #community/{comm_tag_name}")
|
|
930
|
+
lines.append("SORT file.name ASC")
|
|
931
|
+
lines.append("```")
|
|
932
|
+
lines.append("")
|
|
933
|
+
|
|
934
|
+
# Connections to other communities
|
|
935
|
+
cross = inter_community_edges.get(cid, {})
|
|
936
|
+
if cross:
|
|
937
|
+
lines.append("## Connections to other communities")
|
|
938
|
+
for other_cid, edge_count in sorted(cross.items(), key=lambda x: -x[1]):
|
|
939
|
+
other_fname = community_filename.get(other_cid) or (
|
|
940
|
+
f"{_COMMUNITY_PREFIX}"
|
|
941
|
+
f"{_obsidian_safe_stem(_community_name(other_cid), _community_stem_limit)}"
|
|
942
|
+
)
|
|
943
|
+
lines.append(f"- {edge_count} edge{'s' if edge_count != 1 else ''} to [[{other_fname}]]")
|
|
944
|
+
lines.append("")
|
|
945
|
+
|
|
946
|
+
# Top bridge nodes - highest degree nodes that connect to other communities
|
|
947
|
+
bridge_nodes = [
|
|
948
|
+
(node_id, G.degree(node_id), _community_reach(node_id))
|
|
949
|
+
for node_id in members
|
|
950
|
+
if _community_reach(node_id) > 0
|
|
951
|
+
]
|
|
952
|
+
bridge_nodes.sort(key=lambda x: (-x[2], -x[1]))
|
|
953
|
+
top_bridges = bridge_nodes[:5]
|
|
954
|
+
if top_bridges:
|
|
955
|
+
lines.append("## Top bridge nodes")
|
|
956
|
+
for node_id, degree, reach in top_bridges:
|
|
957
|
+
node_label = node_filename[node_id]
|
|
958
|
+
lines.append(
|
|
959
|
+
f"- [[{node_label}]] - degree {degree}, connects to {reach} "
|
|
960
|
+
f"{'community' if reach == 1 else 'communities'}"
|
|
961
|
+
)
|
|
962
|
+
|
|
963
|
+
fname = community_filename[cid] + ".md"
|
|
964
|
+
if _owned_write(fname, "\n".join(lines)):
|
|
965
|
+
community_notes_written += 1
|
|
966
|
+
|
|
967
|
+
# Improvement 4: write .obsidian/graph.json to color nodes by community in graph
|
|
968
|
+
# view — but never clobber an existing .obsidian/graph.json graphify doesn't own
|
|
969
|
+
# (the user's graph-view settings live there). _owned_write handles that and
|
|
970
|
+
# creates the .obsidian/ dir only when it actually writes.
|
|
971
|
+
graph_config = {
|
|
972
|
+
"colorGroups": [
|
|
973
|
+
{
|
|
974
|
+
# Same sanitizer as the note tags (#2862): built from the raw
|
|
975
|
+
# label, the canvas colour group queried a tag that no note
|
|
976
|
+
# carries whenever the label held non-ASCII or punctuation.
|
|
977
|
+
"query": f"tag:#community/{_obsidian_tag(label)}",
|
|
978
|
+
"color": {"a": 1, "rgb": int(COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)].lstrip('#'), 16)}
|
|
979
|
+
}
|
|
980
|
+
for cid, label in sorted((community_labels or {}).items())
|
|
981
|
+
]
|
|
982
|
+
}
|
|
983
|
+
_owned_write(".obsidian/graph.json", json.dumps(graph_config, indent=2))
|
|
984
|
+
|
|
985
|
+
# #1896: prune notes for nodes that dropped out of the graph. Only files the
|
|
986
|
+
# manifest says graphify owns are candidates, and anything written or skipped
|
|
987
|
+
# this run is excluded — so a user's own note is never touched (foreign files
|
|
988
|
+
# land in _skipped, never _owned). Guard each path to stay inside the vault in
|
|
989
|
+
# case a corrupt/hostile manifest contains `../` entries.
|
|
990
|
+
stale = _owned - set(_written) - set(_skipped)
|
|
991
|
+
pruned = 0
|
|
992
|
+
for rel_name in sorted(stale):
|
|
993
|
+
target = (out / rel_name).resolve()
|
|
994
|
+
if out.resolve() not in target.parents:
|
|
995
|
+
continue
|
|
996
|
+
try:
|
|
997
|
+
target.unlink(missing_ok=True)
|
|
998
|
+
pruned += 1
|
|
999
|
+
except OSError:
|
|
1000
|
+
pass
|
|
1001
|
+
if pruned:
|
|
1002
|
+
print(
|
|
1003
|
+
f"[graphify] pruned {pruned} note(s) for nodes no longer in the graph",
|
|
1004
|
+
file=sys.stderr,
|
|
1005
|
+
)
|
|
1006
|
+
|
|
1007
|
+
# Persist the manifest of files graphify owns, so a re-run can safely update its
|
|
1008
|
+
# own notes while still refusing to touch the user's. Warn (once, aggregated)
|
|
1009
|
+
# about anything skipped to avoid clobbering a pre-existing file.
|
|
1010
|
+
try:
|
|
1011
|
+
_manifest_path.write_text(json.dumps({"files": sorted(set(_written))}, indent=2), encoding="utf-8")
|
|
1012
|
+
except OSError:
|
|
1013
|
+
pass
|
|
1014
|
+
if _skipped:
|
|
1015
|
+
shown = ", ".join(_skipped[:5]) + (f" (+{len(_skipped) - 5} more)" if len(_skipped) > 5 else "")
|
|
1016
|
+
print(
|
|
1017
|
+
f"[graphify] WARNING: skipped {len(_skipped)} pre-existing file(s) graphify "
|
|
1018
|
+
f"did not create, to avoid overwriting your notes: {shown}. "
|
|
1019
|
+
f"Export into an empty directory (or the default graphify-out/obsidian) "
|
|
1020
|
+
f"to get the full vault.",
|
|
1021
|
+
file=sys.stderr,
|
|
1022
|
+
)
|
|
1023
|
+
|
|
1024
|
+
return node_notes_written + community_notes_written
|
|
1025
|
+
|
|
1026
|
+
|
|
1027
|
+
def to_canvas(
|
|
1028
|
+
G: nx.Graph,
|
|
1029
|
+
communities: dict[int, list[str]],
|
|
1030
|
+
output_path: str,
|
|
1031
|
+
community_labels: dict[int, str] | None = None,
|
|
1032
|
+
node_filenames: dict[str, str] | None = None,
|
|
1033
|
+
) -> None:
|
|
1034
|
+
"""Export graph as an Obsidian Canvas file - communities as groups, nodes as cards.
|
|
1035
|
+
|
|
1036
|
+
Generates a structured layout: communities arranged in a grid, nodes within
|
|
1037
|
+
each community arranged in rows. Edges shown between connected nodes.
|
|
1038
|
+
Opens in Obsidian as an infinite canvas with community groupings visible.
|
|
1039
|
+
"""
|
|
1040
|
+
# Obsidian canvas color codes (cycle through for communities)
|
|
1041
|
+
CANVAS_COLORS = ["1", "2", "3", "4", "5", "6"] # red, orange, yellow, green, cyan, purple
|
|
1042
|
+
|
|
1043
|
+
# Build node_filenames if not provided (same dedup logic as to_obsidian).
|
|
1044
|
+
# The CLI calls to_canvas without passing the map, so it must derive the
|
|
1045
|
+
# SAME stem budget to keep card links pointing at the notes to_obsidian
|
|
1046
|
+
# wrote — hence budgeting against the canvas's own directory, which is the
|
|
1047
|
+
# vault directory (#2655).
|
|
1048
|
+
_stem_limit = stem_filename_budget(
|
|
1049
|
+
Path(output_path).parent, reserve=_DEDUP_SUFFIX_RESERVE
|
|
1050
|
+
)
|
|
1051
|
+
if node_filenames is None:
|
|
1052
|
+
node_filenames = _dedup_node_filenames(
|
|
1053
|
+
G, lambda label: _obsidian_safe_stem(label, _stem_limit)
|
|
1054
|
+
)
|
|
1055
|
+
|
|
1056
|
+
# Fallback: with no community data (e.g. --no-cluster builds or a missing
|
|
1057
|
+
# analysis sidecar) the grid below produces nothing and the canvas is written
|
|
1058
|
+
# as an empty 32-byte shell on an otherwise populated graph. Emit every node
|
|
1059
|
+
# into one synthetic community so the canvas always reflects the graph (#1324).
|
|
1060
|
+
if not communities and G.number_of_nodes() > 0:
|
|
1061
|
+
communities = {0: [str(n) for n in G.nodes()]}
|
|
1062
|
+
|
|
1063
|
+
num_communities = len(communities)
|
|
1064
|
+
cols = math.ceil(math.sqrt(num_communities)) if num_communities > 0 else 1
|
|
1065
|
+
rows = math.ceil(num_communities / cols) if num_communities > 0 else 1
|
|
1066
|
+
|
|
1067
|
+
canvas_nodes: list[dict] = []
|
|
1068
|
+
canvas_edges: list[dict] = []
|
|
1069
|
+
|
|
1070
|
+
# Lay out communities in a grid
|
|
1071
|
+
gap = 80
|
|
1072
|
+
group_x_offsets: list[int] = []
|
|
1073
|
+
group_y_offsets: list[int] = []
|
|
1074
|
+
|
|
1075
|
+
# Precompute group sizes so we can calculate offsets.
|
|
1076
|
+
# inner_cols is the per-community grid width; the box dimensions AND the node
|
|
1077
|
+
# placement loop below both derive from it, so the cards always fill the box
|
|
1078
|
+
# instead of wrapping into a narrow strip inside an oversized box.
|
|
1079
|
+
sorted_cids = sorted(communities.keys())
|
|
1080
|
+
group_sizes: dict[int, tuple[int, int]] = {}
|
|
1081
|
+
group_cols: dict[int, int] = {}
|
|
1082
|
+
for cid in sorted_cids:
|
|
1083
|
+
# Skip dangling community members with no backing node / filename, so box
|
|
1084
|
+
# sizing matches the cards actually laid out and `G.nodes[m]` never
|
|
1085
|
+
# KeyErrors below — mirrors the to_obsidian guard (#1236).
|
|
1086
|
+
members = [m for m in communities[cid] if m in G and m in node_filenames]
|
|
1087
|
+
n = len(members)
|
|
1088
|
+
inner_cols = max(1, math.ceil(math.sqrt(n)))
|
|
1089
|
+
w = max(600, 220 * inner_cols)
|
|
1090
|
+
h = max(400, 100 * math.ceil(n / inner_cols) + 120)
|
|
1091
|
+
group_sizes[cid] = (w, h)
|
|
1092
|
+
group_cols[cid] = inner_cols
|
|
1093
|
+
|
|
1094
|
+
# Compute cumulative row heights and col widths for grid placement
|
|
1095
|
+
# Each grid cell uses the max width/height in its col/row
|
|
1096
|
+
col_widths: list[int] = []
|
|
1097
|
+
row_heights: list[int] = []
|
|
1098
|
+
for col_idx in range(cols):
|
|
1099
|
+
max_w = 0
|
|
1100
|
+
for row_idx in range(rows):
|
|
1101
|
+
linear = row_idx * cols + col_idx
|
|
1102
|
+
if linear < len(sorted_cids):
|
|
1103
|
+
cid = sorted_cids[linear]
|
|
1104
|
+
w, _ = group_sizes[cid]
|
|
1105
|
+
max_w = max(max_w, w)
|
|
1106
|
+
col_widths.append(max_w)
|
|
1107
|
+
|
|
1108
|
+
for row_idx in range(rows):
|
|
1109
|
+
max_h = 0
|
|
1110
|
+
for col_idx in range(cols):
|
|
1111
|
+
linear = row_idx * cols + col_idx
|
|
1112
|
+
if linear < len(sorted_cids):
|
|
1113
|
+
cid = sorted_cids[linear]
|
|
1114
|
+
_, h = group_sizes[cid]
|
|
1115
|
+
max_h = max(max_h, h)
|
|
1116
|
+
row_heights.append(max_h)
|
|
1117
|
+
|
|
1118
|
+
# Map from cid → (group_x, group_y, group_w, group_h)
|
|
1119
|
+
group_layout: dict[int, tuple[int, int, int, int]] = {}
|
|
1120
|
+
for idx, cid in enumerate(sorted_cids):
|
|
1121
|
+
col_idx = idx % cols
|
|
1122
|
+
row_idx = idx // cols
|
|
1123
|
+
gx = sum(col_widths[:col_idx]) + col_idx * gap
|
|
1124
|
+
gy = sum(row_heights[:row_idx]) + row_idx * gap
|
|
1125
|
+
gw, gh = group_sizes[cid]
|
|
1126
|
+
group_layout[cid] = (gx, gy, gw, gh)
|
|
1127
|
+
|
|
1128
|
+
# Build set of all node_ids in canvas for edge filtering
|
|
1129
|
+
all_canvas_nodes: set[str] = set()
|
|
1130
|
+
for members in communities.values():
|
|
1131
|
+
all_canvas_nodes.update(members)
|
|
1132
|
+
|
|
1133
|
+
# Generate group and node canvas entries
|
|
1134
|
+
for idx, cid in enumerate(sorted_cids):
|
|
1135
|
+
members = communities[cid]
|
|
1136
|
+
community_name = (
|
|
1137
|
+
community_labels.get(cid, f"Community {cid}")
|
|
1138
|
+
if community_labels and cid is not None
|
|
1139
|
+
else f"Community {cid}"
|
|
1140
|
+
)
|
|
1141
|
+
gx, gy, gw, gh = group_layout[cid]
|
|
1142
|
+
canvas_color = CANVAS_COLORS[idx % len(CANVAS_COLORS)]
|
|
1143
|
+
|
|
1144
|
+
# Group node
|
|
1145
|
+
canvas_nodes.append({
|
|
1146
|
+
"id": f"g{cid}",
|
|
1147
|
+
"type": "group",
|
|
1148
|
+
"label": community_name,
|
|
1149
|
+
"x": gx,
|
|
1150
|
+
"y": gy,
|
|
1151
|
+
"width": gw,
|
|
1152
|
+
"height": gh,
|
|
1153
|
+
"color": canvas_color,
|
|
1154
|
+
})
|
|
1155
|
+
|
|
1156
|
+
# Node cards inside the group - laid out in the same ceil(sqrt(n))-column
|
|
1157
|
+
# grid the box was sized for (group_cols[cid]), so cards fill the box.
|
|
1158
|
+
inner_cols = group_cols[cid]
|
|
1159
|
+
# Same dangling-member guard as the sizing loop and to_obsidian (#1236):
|
|
1160
|
+
# a community id absent from G / node_filenames would KeyError the sort.
|
|
1161
|
+
members = [m for m in members if m in G and m in node_filenames]
|
|
1162
|
+
sorted_members = sorted(members, key=lambda n: G.nodes[n].get("label", n))
|
|
1163
|
+
for m_idx, node_id in enumerate(sorted_members):
|
|
1164
|
+
col = m_idx % inner_cols
|
|
1165
|
+
row = m_idx // inner_cols
|
|
1166
|
+
nx_x = gx + 20 + col * (180 + 20)
|
|
1167
|
+
nx_y = gy + 80 + row * (60 + 20)
|
|
1168
|
+
fname = node_filenames.get(
|
|
1169
|
+
node_id,
|
|
1170
|
+
_obsidian_safe_stem(G.nodes[node_id].get("label", node_id), _stem_limit),
|
|
1171
|
+
)
|
|
1172
|
+
canvas_nodes.append({
|
|
1173
|
+
"id": f"n_{node_id}",
|
|
1174
|
+
"type": "file",
|
|
1175
|
+
"file": f"{fname}.md",
|
|
1176
|
+
"x": nx_x,
|
|
1177
|
+
"y": nx_y,
|
|
1178
|
+
"width": 180,
|
|
1179
|
+
"height": 60,
|
|
1180
|
+
})
|
|
1181
|
+
|
|
1182
|
+
# Generate edges - only between nodes both in canvas, cap at 200 highest-weight
|
|
1183
|
+
all_edges_weighted: list[tuple[float, str, str, str]] = []
|
|
1184
|
+
for u, v, edata in G.edges(data=True):
|
|
1185
|
+
if u in all_canvas_nodes and v in all_canvas_nodes:
|
|
1186
|
+
weight = edata.get("weight", 1.0)
|
|
1187
|
+
relation = edata.get("relation", "")
|
|
1188
|
+
conf = edata.get("confidence", "EXTRACTED")
|
|
1189
|
+
label = f"{relation} [{conf}]" if relation else f"[{conf}]"
|
|
1190
|
+
all_edges_weighted.append((weight, u, v, label))
|
|
1191
|
+
|
|
1192
|
+
all_edges_weighted.sort(key=lambda x: -x[0])
|
|
1193
|
+
for weight, u, v, label in all_edges_weighted[:200]:
|
|
1194
|
+
canvas_edges.append({
|
|
1195
|
+
"id": f"e_{u}_{v}",
|
|
1196
|
+
"fromNode": f"n_{u}",
|
|
1197
|
+
"toNode": f"n_{v}",
|
|
1198
|
+
"label": label,
|
|
1199
|
+
})
|
|
1200
|
+
|
|
1201
|
+
canvas_data = {"nodes": canvas_nodes, "edges": canvas_edges}
|
|
1202
|
+
write_json_atomic(output_path, canvas_data, indent=2)
|
|
1203
|
+
|
|
1204
|
+
|
|
1205
|
+
def to_graphml(
|
|
1206
|
+
G: nx.Graph,
|
|
1207
|
+
communities: dict[int, list[str]],
|
|
1208
|
+
output_path: str,
|
|
1209
|
+
) -> None:
|
|
1210
|
+
"""Export graph as GraphML - opens in Gephi, yEd, and any GraphML-compatible tool.
|
|
1211
|
+
|
|
1212
|
+
Community IDs are written as a node attribute so Gephi can colour by community.
|
|
1213
|
+
Edge confidence (EXTRACTED/INFERRED/AMBIGUOUS) is preserved as an edge attribute.
|
|
1214
|
+
"""
|
|
1215
|
+
H = G.copy()
|
|
1216
|
+
node_community = _node_community_map(communities)
|
|
1217
|
+
for node_id in H.nodes():
|
|
1218
|
+
H.nodes[node_id]["community"] = node_community.get(node_id, -1)
|
|
1219
|
+
# Drop internal markers (e.g. the AST-provenance "_origin" tag, #1116, and
|
|
1220
|
+
# the "_src"/"_tgt" direction markers) — they are persistence/runtime details,
|
|
1221
|
+
# not graph data, and should not leak into the exported file.
|
|
1222
|
+
for _, attrs in H.nodes(data=True):
|
|
1223
|
+
for k in [k for k in attrs if k.startswith("_")]:
|
|
1224
|
+
del attrs[k]
|
|
1225
|
+
for _, _, attrs in H.edges(data=True):
|
|
1226
|
+
for k in [k for k in attrs if k.startswith("_")]:
|
|
1227
|
+
del attrs[k]
|
|
1228
|
+
# nx.write_graphml only accepts scalar attribute values: None raises, and a
|
|
1229
|
+
# dict/list value (e.g. a per-node `metadata` dict, or the graph-level
|
|
1230
|
+
# `hyperedges` list set by attach_hyperedges()) raises
|
|
1231
|
+
# "GraphML does not support type <class 'dict'/'list'> as data values" (#1831).
|
|
1232
|
+
# Coerce None -> "" and non-scalars -> a JSON string, across all three scopes.
|
|
1233
|
+
def _graphml_safe(val):
|
|
1234
|
+
if val is None:
|
|
1235
|
+
return ""
|
|
1236
|
+
if isinstance(val, bool) or isinstance(val, (int, float)):
|
|
1237
|
+
return val # GraphML-native scalars pass through unchanged
|
|
1238
|
+
if isinstance(val, str):
|
|
1239
|
+
# Scalar, but still has to be XML-representable — see
|
|
1240
|
+
# _strip_xml_illegal. This is the line that turns "one label carried
|
|
1241
|
+
# an ANSI escape" from a lost export into a lost escape character.
|
|
1242
|
+
return _strip_xml_illegal(val)
|
|
1243
|
+
try:
|
|
1244
|
+
return _strip_xml_illegal(json.dumps(val, default=str, sort_keys=True))
|
|
1245
|
+
except (TypeError, ValueError):
|
|
1246
|
+
return _strip_xml_illegal(str(val))
|
|
1247
|
+
|
|
1248
|
+
# Node IDs become the `id` attribute of every <node> and edge endpoint, so
|
|
1249
|
+
# they must be XML-representable too. Normalised ids never carry a control
|
|
1250
|
+
# character, but a caller can hand us a hand-built graph, and a crash here
|
|
1251
|
+
# loses the export just as completely as one in the values.
|
|
1252
|
+
_id_remap = {n: _strip_xml_illegal(n) for n in H.nodes if isinstance(n, str)}
|
|
1253
|
+
_id_remap = {k: v for k, v in _id_remap.items() if k != v}
|
|
1254
|
+
if _id_remap:
|
|
1255
|
+
H = nx.relabel_nodes(H, _id_remap, copy=True)
|
|
1256
|
+
|
|
1257
|
+
for key, val in list(H.graph.items()):
|
|
1258
|
+
H.graph[key] = _graphml_safe(val)
|
|
1259
|
+
for node_id in H.nodes():
|
|
1260
|
+
for key, val in list(H.nodes[node_id].items()):
|
|
1261
|
+
H.nodes[node_id][key] = _graphml_safe(val)
|
|
1262
|
+
for u, v in H.edges():
|
|
1263
|
+
for key, val in list(H.edges[u, v].items()):
|
|
1264
|
+
H.edges[u, v][key] = _graphml_safe(val)
|
|
1265
|
+
|
|
1266
|
+
# Write atomically: a mid-serialization error otherwise leaves a 0-byte
|
|
1267
|
+
# .graphml on disk that downstream tooling mistakes for a completed export
|
|
1268
|
+
# (#1831). Write to a sibling temp file, then replace on success.
|
|
1269
|
+
out = Path(output_path)
|
|
1270
|
+
tmp = out.with_name(out.name + ".tmp")
|
|
1271
|
+
try:
|
|
1272
|
+
nx.write_graphml(H, str(tmp))
|
|
1273
|
+
os.replace(str(tmp), str(out))
|
|
1274
|
+
finally:
|
|
1275
|
+
if tmp.exists():
|
|
1276
|
+
try:
|
|
1277
|
+
tmp.unlink()
|
|
1278
|
+
except OSError:
|
|
1279
|
+
pass
|
|
1280
|
+
|
|
1281
|
+
|
|
1282
|
+
def to_svg(
|
|
1283
|
+
G: nx.Graph,
|
|
1284
|
+
communities: dict[int, list[str]],
|
|
1285
|
+
output_path: str,
|
|
1286
|
+
community_labels: dict[int, str] | None = None,
|
|
1287
|
+
figsize: tuple[int, int] = (20, 14),
|
|
1288
|
+
) -> None:
|
|
1289
|
+
"""Export graph as an SVG file using matplotlib + spring layout.
|
|
1290
|
+
|
|
1291
|
+
Lightweight and embeddable - works in Obsidian notes, Notion, GitHub READMEs,
|
|
1292
|
+
and any markdown renderer. No JavaScript required.
|
|
1293
|
+
|
|
1294
|
+
Node size scales with degree. Community colors match the HTML output.
|
|
1295
|
+
"""
|
|
1296
|
+
try:
|
|
1297
|
+
import matplotlib
|
|
1298
|
+
matplotlib.use("Agg")
|
|
1299
|
+
import matplotlib.pyplot as plt
|
|
1300
|
+
import matplotlib.patches as mpatches
|
|
1301
|
+
except ImportError as e:
|
|
1302
|
+
raise ImportError("matplotlib not installed. Run: pip install matplotlib") from e
|
|
1303
|
+
|
|
1304
|
+
node_community = _node_community_map(communities)
|
|
1305
|
+
|
|
1306
|
+
fig, ax = plt.subplots(figsize=figsize, facecolor="#1a1a2e")
|
|
1307
|
+
ax.set_facecolor("#1a1a2e")
|
|
1308
|
+
ax.axis("off")
|
|
1309
|
+
|
|
1310
|
+
pos = nx.spring_layout(G, seed=42, k=2.0 / (G.number_of_nodes() ** 0.5 + 1))
|
|
1311
|
+
|
|
1312
|
+
degree = dict(G.degree())
|
|
1313
|
+
max_deg = max(degree.values(), default=1) or 1
|
|
1314
|
+
|
|
1315
|
+
node_colors = [COMMUNITY_COLORS[node_community.get(n, 0) % len(COMMUNITY_COLORS)] for n in G.nodes()]
|
|
1316
|
+
node_sizes = [300 + 1200 * (degree.get(n, 1) / max_deg) for n in G.nodes()]
|
|
1317
|
+
|
|
1318
|
+
# Draw edges - dashed for non-EXTRACTED
|
|
1319
|
+
for u, v, data in G.edges(data=True):
|
|
1320
|
+
conf = data.get("confidence", "EXTRACTED")
|
|
1321
|
+
style = "solid" if conf == "EXTRACTED" else "dashed"
|
|
1322
|
+
alpha = 0.6 if conf == "EXTRACTED" else 0.3
|
|
1323
|
+
x0, y0 = pos[u]
|
|
1324
|
+
x1, y1 = pos[v]
|
|
1325
|
+
ax.plot([x0, x1], [y0, y1], color="#aaaaaa", linewidth=0.8,
|
|
1326
|
+
linestyle=style, alpha=alpha, zorder=1)
|
|
1327
|
+
|
|
1328
|
+
nx.draw_networkx_nodes(G, pos, ax=ax, node_color=node_colors,
|
|
1329
|
+
node_size=node_sizes, alpha=0.9)
|
|
1330
|
+
nx.draw_networkx_labels(G, pos, ax=ax,
|
|
1331
|
+
labels={n: G.nodes[n].get("label", n) for n in G.nodes()},
|
|
1332
|
+
font_size=7, font_color="white")
|
|
1333
|
+
|
|
1334
|
+
# Legend
|
|
1335
|
+
if community_labels:
|
|
1336
|
+
patches = [
|
|
1337
|
+
mpatches.Patch(
|
|
1338
|
+
color=COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)],
|
|
1339
|
+
label=f"{label} ({len(communities.get(cid, []))})",
|
|
1340
|
+
)
|
|
1341
|
+
for cid, label in sorted(community_labels.items())
|
|
1342
|
+
]
|
|
1343
|
+
ax.legend(handles=patches, loc="upper left", framealpha=0.7,
|
|
1344
|
+
facecolor="#2a2a4e", labelcolor="white", fontsize=8)
|
|
1345
|
+
|
|
1346
|
+
plt.tight_layout()
|
|
1347
|
+
plt.savefig(output_path, format="svg", bbox_inches="tight",
|
|
1348
|
+
facecolor=fig.get_facecolor())
|
|
1349
|
+
plt.close(fig)
|