graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,1877 @@
|
|
|
1
|
+
"""Maps a GroundedUnderstanding into archify's real architecture IR, then
|
|
2
|
+
shells out to the `archify` CLI to validate and render it.
|
|
3
|
+
|
|
4
|
+
The componentType classification and grid row/col assignment are the two
|
|
5
|
+
things the original plan sketch didn't account for (plan.md §04). Both are
|
|
6
|
+
heuristics here, deliberately simple - a real implementation should let
|
|
7
|
+
Synthesize's own LLM pass override _classify_component with better judgment
|
|
8
|
+
when grounding evidence (e.g. Graphify's file_type/source_file) disagrees.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import re
|
|
15
|
+
import shlex
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
import warnings
|
|
19
|
+
from collections import Counter
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Literal
|
|
22
|
+
|
|
23
|
+
from ..models import GroundedUnderstanding
|
|
24
|
+
from ..synthesize.llm_backend import LLMBackend
|
|
25
|
+
from . import archify_repair
|
|
26
|
+
from .archify_ir import ArchitectureIR, Component, Connection, GridLayout, GuidedView, Meta
|
|
27
|
+
|
|
28
|
+
# Archify is part of the Graphitect wheel. It intentionally has no npm runtime
|
|
29
|
+
# dependencies, so the bundled sources run directly with Node 18+ and never
|
|
30
|
+
# trigger a network install.
|
|
31
|
+
_BUNDLED_ARCHIFY_PATH = (
|
|
32
|
+
Path(__file__).resolve().parents[1] / "_vendor" / "archify" / "bin" / "archify.mjs"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
_TYPE_KEYWORDS: dict[str, tuple[str, ...]] = {
|
|
36
|
+
"frontend": ("component", "page", "view", ".tsx", ".jsx", "ui/", "frontend/"),
|
|
37
|
+
"database": ("table", "repository", "db", "postgres", "sqlite", "supabase", ".sql"),
|
|
38
|
+
"security": ("auth", "jwt", "oauth", "gate", "verifier"),
|
|
39
|
+
"messagebus": ("queue", "sqs", "kafka", "pubsub", "event bus"),
|
|
40
|
+
"cloud": ("s3", "cdn", "vercel", "cloudfront", "lambda", "cloud run", "k3s", "ec2"),
|
|
41
|
+
"external": ("hook", "human", "external", "third-party", "llm_client", "portfolio_site"),
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _classify_component(node: dict) -> str:
|
|
46
|
+
"""First-pass heuristic classifier - see module docstring."""
|
|
47
|
+
haystack = " ".join(
|
|
48
|
+
str(node.get(k, "")) for k in ("id", "label", "source_file", "file_type")
|
|
49
|
+
).lower()
|
|
50
|
+
for archify_type, keywords in _TYPE_KEYWORDS.items():
|
|
51
|
+
if any(kw in haystack for kw in keywords):
|
|
52
|
+
return archify_type
|
|
53
|
+
return "backend" # default: most graph nodes in a service repo are backend logic
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _assign_grid(nodes: list[dict], edges: list[dict], cols: int = 5) -> dict[str, tuple[int, int]]:
|
|
57
|
+
"""Row = longest-path depth from a source node (topological layering).
|
|
58
|
+
Col = stable order within a row, by first-seen index. Deterministic, not
|
|
59
|
+
pretty - a real layout pass would minimize edge crossings within a row.
|
|
60
|
+
"""
|
|
61
|
+
node_ids = [n["id"] for n in nodes]
|
|
62
|
+
incoming: dict[str, set[str]] = {nid: set() for nid in node_ids}
|
|
63
|
+
for e in edges:
|
|
64
|
+
src, dst = e.get("from") or e.get("source"), e.get("to") or e.get("target")
|
|
65
|
+
if src in incoming and dst in incoming:
|
|
66
|
+
incoming[dst].add(src)
|
|
67
|
+
|
|
68
|
+
depth: dict[str, int] = {}
|
|
69
|
+
|
|
70
|
+
def _depth_of(nid: str, _seen: frozenset[str] = frozenset()) -> int:
|
|
71
|
+
if nid in depth:
|
|
72
|
+
return depth[nid]
|
|
73
|
+
if nid in _seen: # cycle guard
|
|
74
|
+
return 0
|
|
75
|
+
parents = incoming.get(nid, set())
|
|
76
|
+
d = 0 if not parents else 1 + max(_depth_of(p, _seen | {nid}) for p in parents)
|
|
77
|
+
depth[nid] = d
|
|
78
|
+
return d
|
|
79
|
+
|
|
80
|
+
for nid in node_ids:
|
|
81
|
+
_depth_of(nid)
|
|
82
|
+
|
|
83
|
+
by_depth: dict[int, list[str]] = {}
|
|
84
|
+
for nid in node_ids:
|
|
85
|
+
by_depth.setdefault(depth[nid], []).append(nid)
|
|
86
|
+
|
|
87
|
+
# A depth with more than `cols` members used to wrap via `col % cols`,
|
|
88
|
+
# which silently placed multiple nodes in the exact same (row, col) cell
|
|
89
|
+
# once a depth exceeded `cols` nodes - confirmed live against FundersAI's
|
|
90
|
+
# real graph, where depth-0 alone had more than 5 nodes and archify's
|
|
91
|
+
# layout validator rejected the resulting "c0"/"c10" overlap. Each depth
|
|
92
|
+
# now claims as many grid rows as it needs (ceil(count / cols)), and the
|
|
93
|
+
# next depth's rows start after all of those - no two nodes ever share
|
|
94
|
+
# a cell.
|
|
95
|
+
positions: dict[str, tuple[int, int]] = {}
|
|
96
|
+
next_row = 0
|
|
97
|
+
for d in sorted(by_depth):
|
|
98
|
+
ids_at_depth = by_depth[d]
|
|
99
|
+
rows_needed = -(-len(ids_at_depth) // cols) # ceil division
|
|
100
|
+
for i, nid in enumerate(ids_at_depth):
|
|
101
|
+
positions[nid] = (next_row + i // cols, i % cols)
|
|
102
|
+
next_row += rows_needed
|
|
103
|
+
return positions
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _estimate_box_size(labels: list[str]) -> tuple[int, int]:
|
|
107
|
+
"""A uniform box size wide enough to fit the longest label, rather than
|
|
108
|
+
archify's own 120px grid default which is only right for short labels.
|
|
109
|
+
|
|
110
|
+
~7px/char is a rough estimate for archify's default UI font at its
|
|
111
|
+
default size - not measured against the actual rendered font metrics
|
|
112
|
+
(that would need a real text-measurement pass, out of scope here), but
|
|
113
|
+
confirmed live to fix the specific "label wider than component" failures
|
|
114
|
+
this heuristic was built to address, on a real 12-node graph with labels
|
|
115
|
+
up to "Portfolio Description Agent" (~178px measured by archify itself).
|
|
116
|
+
Uniform rather than per-component sizing: keeps every box in a shared
|
|
117
|
+
grid column the same width, avoiding the inter-column overlaps a
|
|
118
|
+
variable-width box would otherwise risk.
|
|
119
|
+
"""
|
|
120
|
+
longest = max((len(label) for label in labels), default=0)
|
|
121
|
+
width = max(120, longest * 7 + 40)
|
|
122
|
+
return width, 60
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# archify's own authoring-contract.md recommends "6-12 primary components" -
|
|
126
|
+
# above that, a diagram either fails archify's showcase layout checks or is
|
|
127
|
+
# just unreadable regardless. Mirrors Graphify's own graph.json/export html
|
|
128
|
+
# fallback to an aggregated community view once a raw graph gets too big to
|
|
129
|
+
# render meaningfully (confirmed live during Phase 1 against a 7,885-node
|
|
130
|
+
# real graph), except the trigger here is "readable diagram" (~15 boxes),
|
|
131
|
+
# not "would this crash a force-directed graph renderer" (Graphify's own
|
|
132
|
+
# threshold there is 5,000 nodes - a completely different concern).
|
|
133
|
+
_MAX_RAW_COMPONENTS = 15
|
|
134
|
+
|
|
135
|
+
# Keep the default report readable while retaining a separate complete
|
|
136
|
+
# relationship explorer. The cap is deliberately above the 6-12 primary
|
|
137
|
+
# components Archify recommends, so the overview can still give every
|
|
138
|
+
# connected component one representative relationship on a 15-box graph.
|
|
139
|
+
_OVERVIEW_MAX_CONNECTIONS = 18
|
|
140
|
+
|
|
141
|
+
# Purely mechanical spacing retries render() falls back to when there's no
|
|
142
|
+
# LLM repair_backend - confirmed live (12 Sep 2026) that a genuinely small
|
|
143
|
+
# 5-node graph can still fail archify's validator on tight label spacing
|
|
144
|
+
# alone. Confirmed NOT to help on its own, though: archify's default
|
|
145
|
+
# connector-label placement turned out to sit at a FIXED offset from the
|
|
146
|
+
# FROM component regardless of how much room widening gapY actually opens
|
|
147
|
+
# up (the exact same overlapping label rect at every multiplier tried) - so
|
|
148
|
+
# this is combined with _apply_suggested_label_fixes below, not relied on
|
|
149
|
+
# alone.
|
|
150
|
+
_SPACING_RETRY_MULTIPLIERS = (1.0, 1.75, 2.5)
|
|
151
|
+
|
|
152
|
+
# How many times render()'s no-repair-backend path retries the mechanical
|
|
153
|
+
# label-fix at a single spacing level before giving up and escalating.
|
|
154
|
+
# Confirmed live (12 Sep 2026) on a real dense large graph (FundersAI,
|
|
155
|
+
# aggregated to 15 boxes / ~35 connections): fixing one label overlap
|
|
156
|
+
# reveals or shifts another - archify's validator doesn't necessarily
|
|
157
|
+
# surface every issue in one pass - so this can take several rounds to
|
|
158
|
+
# fully settle. Each fix only ever pins a connection's labelAt to a fixed
|
|
159
|
+
# point (never undoes an earlier fix), so this is guaranteed to converge
|
|
160
|
+
# within at most one round per connection that ever needs fixing, not
|
|
161
|
+
# oscillate - and each round is a cheap local subprocess call, no LLM
|
|
162
|
+
# involved, so a generous budget costs nothing but a little wall time.
|
|
163
|
+
_MAX_LABEL_FIX_ROUNDS = 20
|
|
164
|
+
|
|
165
|
+
# archify's `layout/constraint` (label-overlap) diagnostics carry no
|
|
166
|
+
# structured `subject`/`evidence` identifying which connection is at fault
|
|
167
|
+
# (confirmed live, 12 Sep 2026 - unlike `clean-flow/edge-through-node`,
|
|
168
|
+
# which does) - only the free-text message names the label text and the
|
|
169
|
+
# component it overlaps, and often a concrete "Suggested fix: labelAt
|
|
170
|
+
# [x, y]". These parse that message well enough to apply the fix directly,
|
|
171
|
+
# no LLM needed - archify's own validator already computed the exact
|
|
172
|
+
# answer, this just has to read it.
|
|
173
|
+
_LABEL_OVERLAP_RE = re.compile(r'Label "([^"]+)" overlaps component "([^"]+)"')
|
|
174
|
+
_LABEL_AT_RE = re.compile(r"labelAt \[(-?[\d.]+),\s*(-?[\d.]+)\]")
|
|
175
|
+
_GENERIC_COMMUNITY_RE = re.compile(
|
|
176
|
+
r"^(?:community|cluster)\s*[-_#]?\s*\d+$", re.IGNORECASE
|
|
177
|
+
)
|
|
178
|
+
_TEST_PATH_PARTS = {"test", "tests", "__tests__", "spec", "specs", "fixtures"}
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _normalized_source_file(node: dict) -> str:
|
|
182
|
+
return str(node.get("source_file") or "").replace("\\", "/").strip("/")
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _is_test_node(node: dict) -> bool:
|
|
186
|
+
path = _normalized_source_file(node).lower()
|
|
187
|
+
if not path:
|
|
188
|
+
return False
|
|
189
|
+
parts = path.split("/")
|
|
190
|
+
filename = parts[-1]
|
|
191
|
+
return any(part in _TEST_PATH_PARTS for part in parts) or filename.startswith(
|
|
192
|
+
("test_", "test-", ".test.", ".spec.")
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _dominant_source_file(nodes: list[dict]) -> str:
|
|
197
|
+
production = [node for node in nodes if not _is_test_node(node)]
|
|
198
|
+
evidence = production or nodes
|
|
199
|
+
paths = [_normalized_source_file(node) for node in evidence]
|
|
200
|
+
paths = [path for path in paths if path]
|
|
201
|
+
return Counter(paths).most_common(1)[0][0] if paths else ""
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _humanize_identifier(value: str) -> str:
|
|
205
|
+
value = re.sub(r"\.(?:py|tsx?|jsx?|json|ya?ml)$", "", value, flags=re.IGNORECASE)
|
|
206
|
+
words = re.sub(r"[_-]+", " ", value).strip().split()
|
|
207
|
+
acronyms = {
|
|
208
|
+
"adk": "ADK",
|
|
209
|
+
"ai": "AI",
|
|
210
|
+
"api": "API",
|
|
211
|
+
"ats": "ATS",
|
|
212
|
+
"db": "DB",
|
|
213
|
+
"http": "HTTP",
|
|
214
|
+
"llm": "LLM",
|
|
215
|
+
"ui": "UI",
|
|
216
|
+
}
|
|
217
|
+
return " ".join(acronyms.get(word.lower(), word.capitalize()) for word in words)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _has_meaningful_community_label(label: str | None, key: str) -> bool:
|
|
221
|
+
if not label:
|
|
222
|
+
return False
|
|
223
|
+
cleaned = str(label).strip()
|
|
224
|
+
return bool(cleaned) and cleaned.lower() != key.lower() and not _GENERIC_COMMUNITY_RE.fullmatch(
|
|
225
|
+
cleaned
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _label_suffix(nodes: list[dict]) -> str:
|
|
230
|
+
"""A concise, source-grounded tie-breaker for duplicate labels."""
|
|
231
|
+
filename = _dominant_source_file(nodes).rsplit("/", 1)[-1]
|
|
232
|
+
stem = re.sub(r"\.[^.]+$", "", filename).lower()
|
|
233
|
+
if stem not in {"", "__init__", "app", "index", "main", "page", "route"}:
|
|
234
|
+
return _humanize_identifier(stem)
|
|
235
|
+
|
|
236
|
+
for node in nodes:
|
|
237
|
+
symbol = re.sub(r"\(.*", "", str(node.get("label") or "")).strip()
|
|
238
|
+
if symbol and len(symbol) <= 40:
|
|
239
|
+
return _humanize_identifier(symbol)
|
|
240
|
+
return "component"
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _disambiguate_component_labels(
|
|
244
|
+
label_of: dict[str, str], members_by_key: dict[str, list[dict]], keys: list[str]
|
|
245
|
+
) -> None:
|
|
246
|
+
"""Keep rendered component labels unique without inventing an LLM name."""
|
|
247
|
+
keys_by_label: dict[str, list[str]] = {}
|
|
248
|
+
for key in keys:
|
|
249
|
+
keys_by_label.setdefault(label_of[key], []).append(key)
|
|
250
|
+
|
|
251
|
+
existing = set(label_of.values())
|
|
252
|
+
for label, duplicate_keys in keys_by_label.items():
|
|
253
|
+
if len(duplicate_keys) < 2:
|
|
254
|
+
continue
|
|
255
|
+
|
|
256
|
+
suffixes = [_label_suffix(members_by_key.get(key, [])) for key in duplicate_keys]
|
|
257
|
+
duplicate_suffixes = Counter(suffixes)
|
|
258
|
+
for index, (key, suffix) in enumerate(zip(duplicate_keys, suffixes), start=1):
|
|
259
|
+
if duplicate_suffixes[suffix] > 1:
|
|
260
|
+
suffix = f"component {index}"
|
|
261
|
+
candidate = f"{label} ({suffix})"
|
|
262
|
+
while candidate in existing:
|
|
263
|
+
index += 1
|
|
264
|
+
candidate = f"{label} (component {index})"
|
|
265
|
+
label_of[key] = candidate
|
|
266
|
+
existing.add(candidate)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _derive_community_label(nodes: list[dict], key: str) -> str:
|
|
270
|
+
"""Name a Graphify community from repository evidence, without an LLM."""
|
|
271
|
+
production = [node for node in nodes if not _is_test_node(node)]
|
|
272
|
+
evidence = production or nodes
|
|
273
|
+
paths = [_normalized_source_file(node) for node in evidence]
|
|
274
|
+
paths = [path for path in paths if path]
|
|
275
|
+
dominant_path = _dominant_source_file(evidence)
|
|
276
|
+
filename = dominant_path.rsplit("/", 1)[-1] if dominant_path else ""
|
|
277
|
+
stem = re.sub(r"\.[^.]+$", "", filename).lower()
|
|
278
|
+
corpus = " ".join(
|
|
279
|
+
str(node.get(field) or "") for node in evidence for field in ("id", "label")
|
|
280
|
+
).lower()
|
|
281
|
+
|
|
282
|
+
if stem == "freelance_graph" or "talentos // studio" in corpus:
|
|
283
|
+
return "Studio Pipeline"
|
|
284
|
+
if stem == "graph" and ("pipeline" in corpus or "talentos // careers" in corpus):
|
|
285
|
+
return "Careers Pipeline"
|
|
286
|
+
|
|
287
|
+
if "firestore" in dominant_path.lower() or stem == "firestore_store":
|
|
288
|
+
budget_score = sum(
|
|
289
|
+
corpus.count(term)
|
|
290
|
+
for term in ("budget", "run_slot", "claim_run", "reserve", "settle", "release")
|
|
291
|
+
)
|
|
292
|
+
record_score = sum(
|
|
293
|
+
corpus.count(term)
|
|
294
|
+
for term in ("application", "materials", "job", "lead", "client", "profile", "listing")
|
|
295
|
+
)
|
|
296
|
+
if budget_score >= 2 and budget_score > record_score:
|
|
297
|
+
return "Firestore Budgets & Runs"
|
|
298
|
+
if record_score:
|
|
299
|
+
return "Firestore Records"
|
|
300
|
+
return "Firestore"
|
|
301
|
+
|
|
302
|
+
if stem in {"notify", "notification", "notifications"}:
|
|
303
|
+
return "Notifications"
|
|
304
|
+
|
|
305
|
+
frontend_paths = [path.lower() for path in paths if "/frontend/" in f"/{path.lower()}/"]
|
|
306
|
+
if paths and len(frontend_paths) * 2 >= len(paths):
|
|
307
|
+
if stem == "package":
|
|
308
|
+
return "Frontend Dependencies"
|
|
309
|
+
if any("/app/api/" in f"/{path}" for path in frontend_paths) or stem in {
|
|
310
|
+
"cloud-run",
|
|
311
|
+
"auth-server",
|
|
312
|
+
}:
|
|
313
|
+
return "Frontend API"
|
|
314
|
+
# A community may merely import authentication helpers. Prefer the
|
|
315
|
+
# dominant file's role so a generic UI component does not become auth.
|
|
316
|
+
if "auth" in stem or "firebase" in stem:
|
|
317
|
+
return "Frontend Authentication"
|
|
318
|
+
if stem == "types":
|
|
319
|
+
return "Frontend Domain Models"
|
|
320
|
+
if "dashboard" in stem:
|
|
321
|
+
return _humanize_identifier(stem)
|
|
322
|
+
return "Frontend UI"
|
|
323
|
+
|
|
324
|
+
source_paths = [path for path in paths if "/sources/" in f"/{path.lower()}/"]
|
|
325
|
+
if source_paths and len(source_paths) * 2 >= len(paths):
|
|
326
|
+
source_names = {
|
|
327
|
+
"aggregators": "Job Aggregators",
|
|
328
|
+
"ats_boards": "ATS Sources",
|
|
329
|
+
"company_portals": "Company Sources",
|
|
330
|
+
"freelance_boards": "Freelance Sources",
|
|
331
|
+
"profile_sources": "Profile Sources",
|
|
332
|
+
"text_utils": "Source Utilities",
|
|
333
|
+
}
|
|
334
|
+
return source_names.get(stem, "Sources")
|
|
335
|
+
|
|
336
|
+
named_files = {
|
|
337
|
+
"agent": "AI Agents",
|
|
338
|
+
"board_scout": "Board Discovery",
|
|
339
|
+
"matching": "Job Matching",
|
|
340
|
+
"models": "Domain Models",
|
|
341
|
+
"pipeline": "Evaluation Pipeline",
|
|
342
|
+
"resume_render": "Resume Generation",
|
|
343
|
+
"review_groups": "Review Workflows",
|
|
344
|
+
"run_progress": "Run Progress",
|
|
345
|
+
"schemas": "Data Contracts",
|
|
346
|
+
"telemetry": "Observability",
|
|
347
|
+
}
|
|
348
|
+
if stem in named_files:
|
|
349
|
+
return named_files[stem]
|
|
350
|
+
if stem not in {"", "__init__", "app", "index", "main", "page", "route"}:
|
|
351
|
+
return _humanize_identifier(stem)
|
|
352
|
+
|
|
353
|
+
labels = [str(node.get("label") or "").strip() for node in evidence]
|
|
354
|
+
labels = [label for label in labels if label and len(label) <= 48 and not label.startswith(".")]
|
|
355
|
+
return _humanize_identifier(labels[0]) if labels else key
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _apply_suggested_label_fixes(ir: ArchitectureIR, diagnostics: list[dict]) -> ArchitectureIR | None:
|
|
359
|
+
"""Best-effort: returns a copy of `ir` with archify's own suggested
|
|
360
|
+
`labelAt` applied to whichever connection each parseable
|
|
361
|
+
`layout/constraint` diagnostic names, or None if no diagnostic could be
|
|
362
|
+
matched to a real connection (nothing to retry with). A diagnostic that
|
|
363
|
+
doesn't parse, or whose named label doesn't match any real connection,
|
|
364
|
+
is skipped rather than treated as fatal - the caller's own retry loop
|
|
365
|
+
still has other spacing attempts left either way.
|
|
366
|
+
|
|
367
|
+
Matches purely by label text, NOT by requiring the overlapped
|
|
368
|
+
component to be one of the connection's own endpoints - confirmed live
|
|
369
|
+
(12 Sep 2026) that this requirement made the fix silently never apply
|
|
370
|
+
at all on a real large graph: a rerouted connection's label can default
|
|
371
|
+
to a position anywhere along its (now much longer) multi-waypoint path,
|
|
372
|
+
landing on and overlapping a totally unrelated third component that
|
|
373
|
+
isn't its `from`/`to` at all - the earlier version's endpoint check
|
|
374
|
+
rejected exactly the one real match available, and the same diagnostic
|
|
375
|
+
kept recurring identically across every retry round because the actual
|
|
376
|
+
offending connection was never touched.
|
|
377
|
+
|
|
378
|
+
Also skips connections that already have `label_at` set. Label text
|
|
379
|
+
alone isn't a reliable unique key either - `_aggregate_by_community`
|
|
380
|
+
generates generic labels like "3 connections" that multiple different
|
|
381
|
+
community pairs can share verbatim, confirmed live on a real graph
|
|
382
|
+
(three separate connections all labeled "12 connections"). Without this
|
|
383
|
+
check, once the first match got fixed, every later round kept
|
|
384
|
+
"fixing" that SAME already-fixed connection again (its label still
|
|
385
|
+
equalled label_text) instead of moving on to the next real offender
|
|
386
|
+
sharing that label - the retry loop looked like it was making no
|
|
387
|
+
progress at all when it was actually just stuck re-patching one
|
|
388
|
+
connection repeatedly.
|
|
389
|
+
"""
|
|
390
|
+
connections = list(ir.connections)
|
|
391
|
+
changed = False
|
|
392
|
+
for diag in diagnostics:
|
|
393
|
+
if diag.get("code") != "layout/constraint":
|
|
394
|
+
continue
|
|
395
|
+
message = diag.get("message", "")
|
|
396
|
+
overlap_match = _LABEL_OVERLAP_RE.search(message)
|
|
397
|
+
labelat_match = _LABEL_AT_RE.search(message)
|
|
398
|
+
if not overlap_match or not labelat_match:
|
|
399
|
+
continue
|
|
400
|
+
label_text, _component_id = overlap_match.groups()
|
|
401
|
+
x, y = (float(v) for v in labelat_match.groups())
|
|
402
|
+
for i, conn in enumerate(connections):
|
|
403
|
+
if conn.label == label_text and conn.label_at is None:
|
|
404
|
+
connections[i] = conn.model_copy(update={"label_at": (x, y)})
|
|
405
|
+
changed = True
|
|
406
|
+
break
|
|
407
|
+
if not changed:
|
|
408
|
+
return None
|
|
409
|
+
return ir.model_copy(update={"connections": connections})
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _group_by_community(
|
|
413
|
+
nodes: list[dict], community_labels: dict[str, str]
|
|
414
|
+
) -> tuple[dict[str, str], dict[str, str], list[str]]:
|
|
415
|
+
"""The grouping + overflow-folding step of _aggregate_by_community,
|
|
416
|
+
extracted so node_id_remap() can compute the exact same raw-node-id ->
|
|
417
|
+
diagram-component-id mapping used for the actual rendered diagram,
|
|
418
|
+
without duplicating (and risking drifting from) the aggregation logic
|
|
419
|
+
itself. See _aggregate_by_community's docstring for the *why*.
|
|
420
|
+
|
|
421
|
+
Returns (community_of, label_of, ordered_component_ids).
|
|
422
|
+
"""
|
|
423
|
+
community_of: dict[str, str] = {}
|
|
424
|
+
label_of: dict[str, str] = {}
|
|
425
|
+
member_count: dict[str, int] = {}
|
|
426
|
+
members_by_key: dict[str, list[dict]] = {}
|
|
427
|
+
for n in nodes:
|
|
428
|
+
cid = n.get("community")
|
|
429
|
+
key = f"c{cid}" if cid is not None else f"solo_{n['id']}"
|
|
430
|
+
community_of[n["id"]] = key
|
|
431
|
+
members_by_key.setdefault(key, []).append(n)
|
|
432
|
+
member_count[key] = member_count.get(key, 0) + 1
|
|
433
|
+
|
|
434
|
+
for key, members in members_by_key.items():
|
|
435
|
+
cid = members[0].get("community")
|
|
436
|
+
supplied = community_labels.get(str(cid)) if cid is not None else None
|
|
437
|
+
if _has_meaningful_community_label(supplied, key):
|
|
438
|
+
label_of[key] = str(supplied).strip()
|
|
439
|
+
elif cid is None:
|
|
440
|
+
label_of[key] = members[0].get("label", members[0]["id"])
|
|
441
|
+
else:
|
|
442
|
+
label_of[key] = _derive_community_label(members, key)
|
|
443
|
+
|
|
444
|
+
all_keys = list(dict.fromkeys(community_of.values())) # stable order, de-duplicated
|
|
445
|
+
if len(all_keys) > _MAX_RAW_COMPONENTS:
|
|
446
|
+
# Tests often contain more symbols than the production subsystem they
|
|
447
|
+
# exercise. Rank by production members first so test-only communities
|
|
448
|
+
# cannot displace the architecture users are trying to understand.
|
|
449
|
+
production_count = {
|
|
450
|
+
key: sum(not _is_test_node(node) for node in members_by_key[key])
|
|
451
|
+
for key in all_keys
|
|
452
|
+
}
|
|
453
|
+
ranked = sorted(
|
|
454
|
+
all_keys,
|
|
455
|
+
key=lambda key: (production_count[key], member_count[key]),
|
|
456
|
+
reverse=True,
|
|
457
|
+
)
|
|
458
|
+
kept_order = ranked[: _MAX_RAW_COMPONENTS - 1]
|
|
459
|
+
kept = set(kept_order)
|
|
460
|
+
overflow_count = sum(member_count[k] for k in all_keys if k not in kept)
|
|
461
|
+
for nid, key in community_of.items():
|
|
462
|
+
if key not in kept:
|
|
463
|
+
community_of[nid] = "other_components"
|
|
464
|
+
label_of["other_components"] = f"Other components ({overflow_count})"
|
|
465
|
+
all_keys = kept_order + ["other_components"]
|
|
466
|
+
|
|
467
|
+
_disambiguate_component_labels(label_of, members_by_key, all_keys)
|
|
468
|
+
return community_of, label_of, all_keys
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def _aggregate_by_community(
|
|
472
|
+
nodes: list[dict], edges: list[dict], community_labels: dict[str, str]
|
|
473
|
+
) -> tuple[list[dict], list[dict]]:
|
|
474
|
+
"""Collapse to one box per Graphify community instead of one box per
|
|
475
|
+
raw AST node, when there are too many of the latter for a readable
|
|
476
|
+
diagram. Real Graphify communities, not an LLM's invented summary -
|
|
477
|
+
exactly Naman's point (12 Sep 2026): a large codebase's full graph
|
|
478
|
+
either won't render or renders unreadably, so fall back to Graphify's
|
|
479
|
+
own community-detection output, which was built to solve exactly this
|
|
480
|
+
problem, rather than asking a model to guess at a smaller structure.
|
|
481
|
+
|
|
482
|
+
Nodes without a `community` field (e.g. non-Graphify grounding) each
|
|
483
|
+
become their own single-node "community" - this aggregation only
|
|
484
|
+
meaningfully reduces node count when real community data is present.
|
|
485
|
+
|
|
486
|
+
A real codebase can have far more communities than _MAX_RAW_COMPONENTS
|
|
487
|
+
allows for a single diagram - confirmed live against FundersAI's actual
|
|
488
|
+
graph.json: 548 communities, not the 2-3 a small unit-test graph has.
|
|
489
|
+
One box per community isn't enough on its own; if there are still too
|
|
490
|
+
many after that first collapse, keep only the largest (by member count,
|
|
491
|
+
a proxy for architectural significance - the same intuition behind
|
|
492
|
+
Graphify's own "god nodes" ranking) and fold everything else into a
|
|
493
|
+
single "Other components" box, rather than emitting hundreds of boxes
|
|
494
|
+
archify will just reject anyway.
|
|
495
|
+
"""
|
|
496
|
+
community_of, label_of, all_keys = _group_by_community(nodes, community_labels)
|
|
497
|
+
members_by_component: dict[str, list[dict]] = {key: [] for key in all_keys}
|
|
498
|
+
for node in nodes:
|
|
499
|
+
component_id = community_of[node["id"]]
|
|
500
|
+
members_by_component.setdefault(component_id, []).append(node)
|
|
501
|
+
agg_nodes = [
|
|
502
|
+
{
|
|
503
|
+
"id": key,
|
|
504
|
+
"label": label_of[key] or key,
|
|
505
|
+
"file_type": "community",
|
|
506
|
+
"source_file": _dominant_source_file(members_by_component[key]),
|
|
507
|
+
}
|
|
508
|
+
for key in all_keys
|
|
509
|
+
]
|
|
510
|
+
|
|
511
|
+
cross_edge_counts: dict[tuple[str, str], int] = {}
|
|
512
|
+
for e in edges:
|
|
513
|
+
src, dst = e.get("from") or e.get("source"), e.get("to") or e.get("target")
|
|
514
|
+
c_src, c_dst = community_of.get(src), community_of.get(dst)
|
|
515
|
+
if not c_src or not c_dst or c_src == c_dst:
|
|
516
|
+
continue # drop intra-community edges - they're implied by sharing a box
|
|
517
|
+
pair = tuple(sorted((c_src, c_dst)))
|
|
518
|
+
cross_edge_counts[pair] = cross_edge_counts.get(pair, 0) + 1
|
|
519
|
+
|
|
520
|
+
agg_edges = [
|
|
521
|
+
{
|
|
522
|
+
"from": a,
|
|
523
|
+
"to": b,
|
|
524
|
+
"label": f"{count} connection" + ("s" if count != 1 else ""),
|
|
525
|
+
}
|
|
526
|
+
for (a, b), count in cross_edge_counts.items()
|
|
527
|
+
]
|
|
528
|
+
return agg_nodes, agg_edges
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def _edge_endpoints(edge: dict) -> tuple[str, str]:
|
|
532
|
+
return (
|
|
533
|
+
str(edge.get("from") or edge.get("source") or ""),
|
|
534
|
+
str(edge.get("to") or edge.get("target") or ""),
|
|
535
|
+
)
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def _edge_strength(edge: dict) -> int:
|
|
539
|
+
"""Return a deterministic importance proxy for a structural edge.
|
|
540
|
+
|
|
541
|
+
Community aggregation turns many raw imports into a label such as
|
|
542
|
+
``"12 connections"``. Prefer that measured count for the overview;
|
|
543
|
+
preserve the original edge order only as a final stable tie-breaker.
|
|
544
|
+
"""
|
|
545
|
+
for key in ("weight", "count"):
|
|
546
|
+
try:
|
|
547
|
+
value = int(edge.get(key, 0))
|
|
548
|
+
except (TypeError, ValueError):
|
|
549
|
+
value = 0
|
|
550
|
+
if value > 0:
|
|
551
|
+
return value
|
|
552
|
+
match = re.match(r"\s*(\d+)\s+connections?\b", str(edge.get("label") or ""), re.IGNORECASE)
|
|
553
|
+
return int(match.group(1)) if match else 1
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def _select_overview_edges(
|
|
557
|
+
nodes: list[dict], edges: list[dict], *, max_connections: int = _OVERVIEW_MAX_CONNECTIONS
|
|
558
|
+
) -> list[dict]:
|
|
559
|
+
"""Keep a compact, deterministic structural subset for the default view.
|
|
560
|
+
|
|
561
|
+
The complete graph remains available in the Full graph mode. Here, first
|
|
562
|
+
choose each component's strongest incident relationship so no connected
|
|
563
|
+
subsystem disappears, then fill the remaining slots by aggregate edge
|
|
564
|
+
strength. This is evidence-preserving filtering, not an inferred summary.
|
|
565
|
+
"""
|
|
566
|
+
if len(edges) <= max_connections:
|
|
567
|
+
return list(edges)
|
|
568
|
+
|
|
569
|
+
def sort_key(index: int) -> tuple[int, str, str, str, int]:
|
|
570
|
+
source, target = _edge_endpoints(edges[index])
|
|
571
|
+
first, second = sorted((source, target))
|
|
572
|
+
return (-_edge_strength(edges[index]), first, second, str(edges[index].get("label") or ""), index)
|
|
573
|
+
|
|
574
|
+
ranked = sorted(range(len(edges)), key=sort_key)
|
|
575
|
+
selected: set[int] = set()
|
|
576
|
+
for node_id in (str(node.get("id") or "") for node in nodes):
|
|
577
|
+
if not node_id or len(selected) >= max_connections:
|
|
578
|
+
break
|
|
579
|
+
candidate = next(
|
|
580
|
+
(
|
|
581
|
+
index
|
|
582
|
+
for index in ranked
|
|
583
|
+
if node_id in _edge_endpoints(edges[index]) and index not in selected
|
|
584
|
+
),
|
|
585
|
+
None,
|
|
586
|
+
)
|
|
587
|
+
if candidate is not None:
|
|
588
|
+
selected.add(candidate)
|
|
589
|
+
|
|
590
|
+
for index in ranked:
|
|
591
|
+
if len(selected) >= max_connections:
|
|
592
|
+
break
|
|
593
|
+
selected.add(index)
|
|
594
|
+
|
|
595
|
+
return [edges[index] for index in ranked if index in selected]
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
# archify's own grid math (archify/renderers/architecture/grid.mjs,
|
|
599
|
+
# resolveComponentPos/DEFAULT_GRID - read directly from its source, not
|
|
600
|
+
# inferred from rendered output): pos = origin + [col, row] * (cellSize +
|
|
601
|
+
# gap). graphitect never overrides origin, and never used to set gapX
|
|
602
|
+
# explicitly either (silently inheriting archify's own default) - both are
|
|
603
|
+
# made explicit constants here so _route_around_obstacles's geometry is
|
|
604
|
+
# guaranteed to match what archify will actually render, not an assumed
|
|
605
|
+
# implicit default that could drift.
|
|
606
|
+
_GRID_ORIGIN = (40, 80)
|
|
607
|
+
_GRID_GAP_X = 30
|
|
608
|
+
_GRID_COLUMNS = 5
|
|
609
|
+
_VIEWBOX_MARGIN = 40
|
|
610
|
+
_VIEWBOX_LEGEND_RESERVE = 68 # 40px renderer margin + Archify's 28px legend rail
|
|
611
|
+
|
|
612
|
+
# Workflow views deliberately use a short, direct path. A codebase-wide
|
|
613
|
+
# dependency graph is valuable as the Overview/Full rollup, but it is not a
|
|
614
|
+
# presentation path and becomes jumpy when replayed one node at a time.
|
|
615
|
+
_WORKFLOW_MAX_PATH_NODES = 6
|
|
616
|
+
_WORKFLOW_MAX_BRANCHES = 2
|
|
617
|
+
_WORKFLOW_RELATION_ORDER = {
|
|
618
|
+
"calls": 0,
|
|
619
|
+
"invokes": 0,
|
|
620
|
+
"uses": 1,
|
|
621
|
+
"imports": 2,
|
|
622
|
+
"depends_on": 2,
|
|
623
|
+
"contains": 3,
|
|
624
|
+
"method": 4,
|
|
625
|
+
"inherits": 5,
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
# A sequence diagram has a stricter evidence threshold than the workflow
|
|
629
|
+
# story. It is only useful when Graphify observed an ordered chain of actual
|
|
630
|
+
# calls; imports, generic uses, and community links are not request evidence.
|
|
631
|
+
_SEQUENCE_MAX_PARTICIPANTS = 6
|
|
632
|
+
_SEQUENCE_RELATIONS = frozenset({"calls", "invokes"})
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def _route_around_obstacles(
|
|
636
|
+
connections: list[Connection],
|
|
637
|
+
positions: dict[str, tuple[int, int]],
|
|
638
|
+
*,
|
|
639
|
+
box_width: int,
|
|
640
|
+
box_height: int,
|
|
641
|
+
cell_w: int,
|
|
642
|
+
cell_h: int,
|
|
643
|
+
gap_x: int,
|
|
644
|
+
gap_y: int,
|
|
645
|
+
cols: int,
|
|
646
|
+
) -> list[Connection]:
|
|
647
|
+
"""Deterministic obstacle-avoiding routing for connections that would
|
|
648
|
+
otherwise have to cross through an unrelated component's box - the real
|
|
649
|
+
fix for `clean-flow/edge-through-node` on large aggregated diagrams
|
|
650
|
+
without an LLM (Naman, 12 Sep 2026: "at least the graph should work"
|
|
651
|
+
for big, complex codebases with no key configured).
|
|
652
|
+
|
|
653
|
+
The grid layout here has no crossing-avoidance of its own. An earlier
|
|
654
|
+
version of this function decided whether to reroute a connection purely
|
|
655
|
+
by row/column distance ("2+ rows apart always gets a detour") -
|
|
656
|
+
confirmed live to reroute far more connections than actually had
|
|
657
|
+
anything in their way, producing a needlessly busy diagram of long
|
|
658
|
+
side-lane detours where archify's own short default route would have
|
|
659
|
+
been perfectly safe. `is_blocked()` below checks the real geometry
|
|
660
|
+
instead: only reroute when some other component's rectangle actually
|
|
661
|
+
overlaps the bounding box between the two endpoints. Rerouted
|
|
662
|
+
connections travel ONLY through space that is provably always empty,
|
|
663
|
+
regardless of how many components exist:
|
|
664
|
+
|
|
665
|
+
- The horizontal strip in the gap between any two adjacent rows (from
|
|
666
|
+
one row's bottom edge to the next row's top edge) contains no
|
|
667
|
+
component at ANY column, since every component in a row shares that
|
|
668
|
+
row's y-range - a horizontal move confined to that strip can never
|
|
669
|
+
cross a box.
|
|
670
|
+
- A vertical "lane" positioned past the last column contains no
|
|
671
|
+
component at any row, for the same reason - a vertical move confined
|
|
672
|
+
to that lane can never cross one either.
|
|
673
|
+
|
|
674
|
+
So a connection more than one row apart exits its own box vertically
|
|
675
|
+
(safe - it's leaving through its own column, stopping before the next
|
|
676
|
+
row's box starts), travels the empty inter-row strip out to the lane,
|
|
677
|
+
travels down/up the empty lane, then reverses through the target's own
|
|
678
|
+
inter-row strip into its box - never once crossing box-body height in
|
|
679
|
+
another column. A same-row, far-column connection is simpler: dip into
|
|
680
|
+
the empty strip below (or above) that row and back up, without needing
|
|
681
|
+
the lane at all - no other row lies between two components in the same
|
|
682
|
+
row. (An earlier version of this router connected box centers directly
|
|
683
|
+
through box-body height and was confirmed live to still cross other
|
|
684
|
+
components - this waypoint choice is the actual fix, not a refinement
|
|
685
|
+
of it.) Short, local connections (adjacent row and/or column) are left
|
|
686
|
+
on archify's own default routing, already confirmed to work at that
|
|
687
|
+
scale.
|
|
688
|
+
"""
|
|
689
|
+
ox, oy = _GRID_ORIGIN
|
|
690
|
+
step_x = cell_w + gap_x
|
|
691
|
+
step_y = cell_h + gap_y
|
|
692
|
+
# Safely past the last *occupied* column, rather than after the configured
|
|
693
|
+
# grid width. A 5-column grid can have only four used columns; using the
|
|
694
|
+
# phantom fifth column made routes needlessly wide and, before meta.viewBox
|
|
695
|
+
# was authored, put them outside Archify's component-derived canvas.
|
|
696
|
+
rightmost_col = max((col for _row, col in positions.values()), default=cols - 1)
|
|
697
|
+
lane_x = ox + (rightmost_col + 1) * step_x + _VIEWBOX_MARGIN
|
|
698
|
+
|
|
699
|
+
def rect_of(row: int, col: int) -> tuple[float, float, float, float]:
|
|
700
|
+
x = ox + col * step_x
|
|
701
|
+
y = oy + row * step_y
|
|
702
|
+
return x, y, x + box_width, y + box_height
|
|
703
|
+
|
|
704
|
+
def gap_below(row: int) -> float:
|
|
705
|
+
return oy + row * step_y + box_height + gap_y / 2
|
|
706
|
+
|
|
707
|
+
def gap_above(row: int) -> float:
|
|
708
|
+
return gap_below(row - 1)
|
|
709
|
+
|
|
710
|
+
def x_mid(col: int) -> float:
|
|
711
|
+
return ox + col * step_x + box_width / 2
|
|
712
|
+
|
|
713
|
+
def is_blocked(from_pos: tuple[int, int], to_pos: tuple[int, int]) -> bool:
|
|
714
|
+
"""Conservative check: does any OTHER component's rectangle overlap
|
|
715
|
+
the bounding box spanned by the two endpoints? Confirmed live (12
|
|
716
|
+
Sep 2026) that a pure row-distance heuristic ("2+ rows apart always
|
|
717
|
+
needs a detour") reroutes far more connections than actually have
|
|
718
|
+
anything in their way, producing a needlessly cluttered diagram of
|
|
719
|
+
long side-lane detours where a short default route would have been
|
|
720
|
+
perfectly safe. Nothing else even overlapping the rectangle between
|
|
721
|
+
two components means there is nothing for any reasonable route
|
|
722
|
+
between them to cross - safe to leave on archify's own (simpler,
|
|
723
|
+
shorter-looking) default routing.
|
|
724
|
+
"""
|
|
725
|
+
from_rect, to_rect = rect_of(*from_pos), rect_of(*to_pos)
|
|
726
|
+
bx1 = min(from_rect[0], to_rect[0])
|
|
727
|
+
by1 = min(from_rect[1], to_rect[1])
|
|
728
|
+
bx2 = max(from_rect[2], to_rect[2])
|
|
729
|
+
by2 = max(from_rect[3], to_rect[3])
|
|
730
|
+
for pos in positions.values():
|
|
731
|
+
if pos == from_pos or pos == to_pos:
|
|
732
|
+
continue
|
|
733
|
+
rx1, ry1, rx2, ry2 = rect_of(*pos)
|
|
734
|
+
if rx1 < bx2 and rx2 > bx1 and ry1 < by2 and ry2 > by1:
|
|
735
|
+
return True
|
|
736
|
+
return False
|
|
737
|
+
|
|
738
|
+
# Confirmed live (12 Sep 2026), twice: every connection crossing the
|
|
739
|
+
# SAME row-boundary used the exact same y at first (gap_below(row)
|
|
740
|
+
# depends only on `row`, not on which connection) - their lines
|
|
741
|
+
# coincided into what looked like one flat "highway" with labels
|
|
742
|
+
# floating on it with no distinguishable line to trace. A first fix
|
|
743
|
+
# (a fixed per-connection step, applied greedily as each connection was
|
|
744
|
+
# visited) was also confirmed live to fall short on a real dense graph:
|
|
745
|
+
# 8px apart reads as one solid band at normal zoom, and a boundary with
|
|
746
|
+
# more connections than the step's clamp allows starts recolliding
|
|
747
|
+
# anyway. The real fix needs to know, before assigning any offset, how
|
|
748
|
+
# many connections will actually share each boundary - hence two passes:
|
|
749
|
+
# first tally how many connections use each boundary, then space that
|
|
750
|
+
# boundary's connections evenly across the FULL safe band (not a fixed
|
|
751
|
+
# step), so N connections sharing a corridor are always maximally and
|
|
752
|
+
# evenly separated regardless of how large N is.
|
|
753
|
+
boundary_counts: dict[int, int] = {}
|
|
754
|
+
|
|
755
|
+
def count_boundary(key: int) -> None:
|
|
756
|
+
boundary_counts[key] = boundary_counts.get(key, 0) + 1
|
|
757
|
+
|
|
758
|
+
decisions: list[
|
|
759
|
+
tuple[Connection, tuple[int, int] | None, tuple[int, int] | None, tuple[int, ...]]
|
|
760
|
+
] = []
|
|
761
|
+
for conn in connections:
|
|
762
|
+
from_pos = positions.get(conn.from_)
|
|
763
|
+
to_pos = positions.get(conn.to)
|
|
764
|
+
if from_pos is None or to_pos is None or not is_blocked(from_pos, to_pos):
|
|
765
|
+
decisions.append((conn, from_pos, to_pos, ()))
|
|
766
|
+
continue
|
|
767
|
+
|
|
768
|
+
from_row, _ = from_pos
|
|
769
|
+
to_row, _ = to_pos
|
|
770
|
+
row_gap = to_row - from_row
|
|
771
|
+
if row_gap == 0:
|
|
772
|
+
boundary_keys = (from_row,)
|
|
773
|
+
elif abs(row_gap) == 1:
|
|
774
|
+
boundary_keys = (min(from_row, to_row),)
|
|
775
|
+
elif row_gap > 0:
|
|
776
|
+
boundary_keys = (from_row, to_row - 1) # gap_above(to_row) == gap_below(to_row - 1)
|
|
777
|
+
else:
|
|
778
|
+
boundary_keys = (from_row - 1, to_row)
|
|
779
|
+
for key in boundary_keys:
|
|
780
|
+
count_boundary(key)
|
|
781
|
+
decisions.append((conn, from_pos, to_pos, boundary_keys))
|
|
782
|
+
|
|
783
|
+
stagger_max = max(0.0, gap_y / 2 - 12)
|
|
784
|
+
boundary_seen: dict[int, int] = {}
|
|
785
|
+
|
|
786
|
+
def next_offset(key: int) -> float:
|
|
787
|
+
count = boundary_counts.get(key, 1)
|
|
788
|
+
index = boundary_seen.get(key, 0)
|
|
789
|
+
boundary_seen[key] = index + 1
|
|
790
|
+
if count <= 1:
|
|
791
|
+
return 0.0
|
|
792
|
+
# Evenly spaced from -stagger_max to +stagger_max across all
|
|
793
|
+
# `count` connections sharing this boundary - the more that share
|
|
794
|
+
# it, the closer together they sit, but they never collide and
|
|
795
|
+
# never leave the safe band, however many there are.
|
|
796
|
+
return -stagger_max + index * (2 * stagger_max) / (count - 1)
|
|
797
|
+
|
|
798
|
+
routed: list[Connection] = []
|
|
799
|
+
lane_offset = 0
|
|
800
|
+
for conn, from_pos, to_pos, boundary_keys in decisions:
|
|
801
|
+
if not boundary_keys:
|
|
802
|
+
routed.append(conn) # unknown endpoint, or nothing in the way - archify's default is fine
|
|
803
|
+
continue
|
|
804
|
+
|
|
805
|
+
from_row, from_col = from_pos
|
|
806
|
+
to_row, to_col = to_pos
|
|
807
|
+
row_gap = to_row - from_row
|
|
808
|
+
|
|
809
|
+
if row_gap == 0:
|
|
810
|
+
# Same row: no other ROW lies between them, so a simple dip
|
|
811
|
+
# into this row's own inter-row gap strip and back is enough -
|
|
812
|
+
# the lane isn't needed.
|
|
813
|
+
y = gap_below(from_row) + next_offset(boundary_keys[0])
|
|
814
|
+
routed.append(
|
|
815
|
+
conn.model_copy(
|
|
816
|
+
update={
|
|
817
|
+
"from_side": "bottom",
|
|
818
|
+
"to_side": "bottom",
|
|
819
|
+
"via": [(x_mid(from_col), y), (x_mid(to_col), y)],
|
|
820
|
+
}
|
|
821
|
+
)
|
|
822
|
+
)
|
|
823
|
+
continue
|
|
824
|
+
|
|
825
|
+
if abs(row_gap) == 1:
|
|
826
|
+
# Adjacent rows: exactly one shared gap strip lies directly
|
|
827
|
+
# between them - still no need for the lane, just use it.
|
|
828
|
+
y = gap_below(boundary_keys[0]) + next_offset(boundary_keys[0])
|
|
829
|
+
from_side = "bottom" if row_gap > 0 else "top"
|
|
830
|
+
to_side = "top" if row_gap > 0 else "bottom"
|
|
831
|
+
routed.append(
|
|
832
|
+
conn.model_copy(
|
|
833
|
+
update={
|
|
834
|
+
"from_side": from_side,
|
|
835
|
+
"to_side": to_side,
|
|
836
|
+
"via": [(x_mid(from_col), y), (x_mid(to_col), y)],
|
|
837
|
+
}
|
|
838
|
+
)
|
|
839
|
+
)
|
|
840
|
+
continue
|
|
841
|
+
|
|
842
|
+
# 2+ rows apart: no single gap strip touches both endpoints, so
|
|
843
|
+
# bridge between the two different strips via the empty side lane.
|
|
844
|
+
this_lane_x = lane_x + lane_offset
|
|
845
|
+
# Stagger parallel lane routes so a diagram with many of them
|
|
846
|
+
# doesn't stack every route on the exact same vertical line.
|
|
847
|
+
lane_offset = (lane_offset + 20) % 100
|
|
848
|
+
from_boundary, to_boundary = boundary_keys
|
|
849
|
+
|
|
850
|
+
if row_gap > 0:
|
|
851
|
+
from_side, to_side = "bottom", "top"
|
|
852
|
+
from_gap_y = gap_below(from_row) + next_offset(from_boundary)
|
|
853
|
+
to_gap_y = gap_above(to_row) + next_offset(to_boundary)
|
|
854
|
+
else:
|
|
855
|
+
from_side, to_side = "top", "bottom"
|
|
856
|
+
from_gap_y = gap_above(from_row) + next_offset(from_boundary)
|
|
857
|
+
to_gap_y = gap_below(to_row) + next_offset(to_boundary)
|
|
858
|
+
|
|
859
|
+
routed.append(
|
|
860
|
+
conn.model_copy(
|
|
861
|
+
update={
|
|
862
|
+
"from_side": from_side,
|
|
863
|
+
"to_side": to_side,
|
|
864
|
+
"via": [
|
|
865
|
+
(x_mid(from_col), from_gap_y),
|
|
866
|
+
(this_lane_x, from_gap_y),
|
|
867
|
+
(this_lane_x, to_gap_y),
|
|
868
|
+
(x_mid(to_col), to_gap_y),
|
|
869
|
+
],
|
|
870
|
+
}
|
|
871
|
+
)
|
|
872
|
+
)
|
|
873
|
+
return routed
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
def _view_box_for_routes(
|
|
877
|
+
positions: dict[str, tuple[int, int]],
|
|
878
|
+
connections: list[Connection],
|
|
879
|
+
*,
|
|
880
|
+
box_width: int,
|
|
881
|
+
box_height: int,
|
|
882
|
+
cell_w: int,
|
|
883
|
+
cell_h: int,
|
|
884
|
+
gap_x: int,
|
|
885
|
+
gap_y: int,
|
|
886
|
+
) -> tuple[int, int]:
|
|
887
|
+
"""Fit components, explicit detours, connection labels, and the legend.
|
|
888
|
+
|
|
889
|
+
Archify's automatic architecture viewBox intentionally measures component
|
|
890
|
+
and boundary boxes only. Graphitect also authors explicit ``via`` points,
|
|
891
|
+
so its canvas must include those points or a valid side lane is visibly
|
|
892
|
+
cropped. The constants mirror Archify's architecture renderer layout.
|
|
893
|
+
"""
|
|
894
|
+
ox, oy = _GRID_ORIGIN
|
|
895
|
+
step_x = cell_w + gap_x
|
|
896
|
+
step_y = cell_h + gap_y
|
|
897
|
+
max_component_x = max(
|
|
898
|
+
(ox + col * step_x + box_width for _row, col in positions.values()), default=ox + box_width
|
|
899
|
+
)
|
|
900
|
+
max_component_y = max(
|
|
901
|
+
(oy + row * step_y + box_height for row, _col in positions.values()), default=oy + box_height
|
|
902
|
+
)
|
|
903
|
+
route_points = [point for connection in connections for point in (connection.via or [])]
|
|
904
|
+
max_route_x = max((point[0] for point in route_points), default=max_component_x)
|
|
905
|
+
max_route_y = max((point[1] for point in route_points), default=max_component_y)
|
|
906
|
+
label_half_width = max(
|
|
907
|
+
(max(30.0, len(connection.label or "") * 4.8 + 10.0) / 2 for connection in connections),
|
|
908
|
+
default=0.0,
|
|
909
|
+
)
|
|
910
|
+
right_padding = max(float(_VIEWBOX_MARGIN), label_half_width + 14.0)
|
|
911
|
+
return (
|
|
912
|
+
max(320, int(max(max_component_x, max_route_x) + right_padding + 0.999)),
|
|
913
|
+
max(240, int(max(max_component_y, max_route_y) + _VIEWBOX_LEGEND_RESERVE + 0.999)),
|
|
914
|
+
)
|
|
915
|
+
|
|
916
|
+
|
|
917
|
+
def node_id_remap(understanding: GroundedUnderstanding) -> dict[str, str]:
|
|
918
|
+
"""Maps every raw Graphify node id to the diagram component id it
|
|
919
|
+
actually renders as - itself, unchanged, when node count is at or under
|
|
920
|
+
_MAX_RAW_COMPONENTS (to_architecture_ir never aggregates in that case),
|
|
921
|
+
or its community/"other_components" box id when it does.
|
|
922
|
+
|
|
923
|
+
A doc section's `related_node_ids` are set once during Synthesize using
|
|
924
|
+
real raw node ids - correct for citing evidence, but once aggregation
|
|
925
|
+
kicks in on a large graph those exact ids no longer exist as elements in
|
|
926
|
+
the rendered SVG, so a node-ref pill's hover-highlight silently finds
|
|
927
|
+
nothing (a known gap flagged in plan.md, fixed by having doc_compiler
|
|
928
|
+
look elements up via this remap rather than the raw id directly).
|
|
929
|
+
"""
|
|
930
|
+
nodes = understanding.nodes
|
|
931
|
+
if len(nodes) <= _MAX_RAW_COMPONENTS:
|
|
932
|
+
return {n["id"]: n["id"] for n in nodes if "id" in n}
|
|
933
|
+
community_of, _, _ = _group_by_community(nodes, understanding.community_labels)
|
|
934
|
+
return community_of
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
def _story_view_id(heading: str, index: int) -> str:
|
|
938
|
+
"""Stable, schema-safe chapter id derived from an existing doc heading."""
|
|
939
|
+
slug = re.sub(r"[^a-z0-9]+", "-", heading.lower()).strip("-")
|
|
940
|
+
return slug or f"chapter-{index + 1}"
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
def _component_adjacency(component_ids: set[str], edges: list[dict]) -> dict[str, set[str]]:
|
|
944
|
+
"""Undirected adjacency for a walk over exact authored relationships.
|
|
945
|
+
|
|
946
|
+
The story viewer separately identifies each hop as forward or reverse.
|
|
947
|
+
This helper only decides whether two consecutive focus nodes have a real
|
|
948
|
+
relationship at all; it never invents a transitive hop from proximity.
|
|
949
|
+
"""
|
|
950
|
+
adjacency = {component_id: set() for component_id in component_ids}
|
|
951
|
+
for edge in edges:
|
|
952
|
+
source, target = _edge_endpoints(edge)
|
|
953
|
+
if source in adjacency and target in adjacency:
|
|
954
|
+
adjacency[source].add(target)
|
|
955
|
+
adjacency[target].add(source)
|
|
956
|
+
return adjacency
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
def _continuous_story_path(
|
|
960
|
+
adjacency: dict[str, set[str]],
|
|
961
|
+
component_ids: set[str],
|
|
962
|
+
*,
|
|
963
|
+
required_id: str | None = None,
|
|
964
|
+
max_nodes: int = 5,
|
|
965
|
+
) -> list[str]:
|
|
966
|
+
"""Pick a bounded simple path whose every adjacent pair is observed.
|
|
967
|
+
|
|
968
|
+
`meta.views.focus` is also Archify's Story Beat order. Supplying a hub
|
|
969
|
+
followed by unrelated neighbours made the viewer honestly report
|
|
970
|
+
grouped/no-direct-link transitions, which reads as a jump. A path keeps
|
|
971
|
+
every beat on a real connection and leaves the viewer to truthfully mark
|
|
972
|
+
that connection's direction.
|
|
973
|
+
"""
|
|
974
|
+
eligible = set(component_ids)
|
|
975
|
+
if len(eligible) > 1:
|
|
976
|
+
eligible.discard("other_components")
|
|
977
|
+
if not eligible:
|
|
978
|
+
return []
|
|
979
|
+
if required_id is not None and required_id not in eligible:
|
|
980
|
+
return [required_id] if required_id in component_ids else []
|
|
981
|
+
|
|
982
|
+
best: tuple[str, ...] = ()
|
|
983
|
+
|
|
984
|
+
def consider(path: list[str]) -> None:
|
|
985
|
+
nonlocal best
|
|
986
|
+
if required_id is not None and required_id not in path:
|
|
987
|
+
return
|
|
988
|
+
candidate = tuple(path)
|
|
989
|
+
if len(candidate) > len(best) or (len(candidate) == len(best) and candidate < best):
|
|
990
|
+
best = candidate
|
|
991
|
+
|
|
992
|
+
def visit(current: str, path: list[str]) -> None:
|
|
993
|
+
consider(path)
|
|
994
|
+
if len(path) == max_nodes:
|
|
995
|
+
return
|
|
996
|
+
for neighbour in sorted(adjacency.get(current, set()) & eligible):
|
|
997
|
+
if neighbour not in path:
|
|
998
|
+
visit(neighbour, [*path, neighbour])
|
|
999
|
+
|
|
1000
|
+
for start in sorted(eligible):
|
|
1001
|
+
visit(start, [start])
|
|
1002
|
+
return list(best)
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
def _story_views(
|
|
1006
|
+
understanding: GroundedUnderstanding,
|
|
1007
|
+
components: list[Component],
|
|
1008
|
+
all_edges: list[dict],
|
|
1009
|
+
) -> list[GuidedView]:
|
|
1010
|
+
"""Create guided chapters without inferring an execution path.
|
|
1011
|
+
|
|
1012
|
+
LLM-backed reports focus repository components explicitly cited by a doc
|
|
1013
|
+
section. Diagram-only reports use connected walks around prominent
|
|
1014
|
+
components. In both cases, adjacent Story beats share an actual graph
|
|
1015
|
+
relationship rather than jumping among a hub's unrelated neighbours.
|
|
1016
|
+
"""
|
|
1017
|
+
component_ids = {component.id for component in components}
|
|
1018
|
+
remap = node_id_remap(understanding)
|
|
1019
|
+
adjacency = _component_adjacency(component_ids, all_edges)
|
|
1020
|
+
views: list[GuidedView] = []
|
|
1021
|
+
seen_paths: set[frozenset[str]] = set()
|
|
1022
|
+
|
|
1023
|
+
for index, section in enumerate(understanding.doc):
|
|
1024
|
+
cited_components = set(
|
|
1025
|
+
dict.fromkeys(
|
|
1026
|
+
remap.get(node_id, node_id)
|
|
1027
|
+
for node_id in section.related_node_ids
|
|
1028
|
+
if remap.get(node_id, node_id) in component_ids
|
|
1029
|
+
)
|
|
1030
|
+
)
|
|
1031
|
+
path = _continuous_story_path(adjacency, cited_components)
|
|
1032
|
+
path_key = frozenset(path)
|
|
1033
|
+
if not path or path_key in seen_paths:
|
|
1034
|
+
continue
|
|
1035
|
+
seen_paths.add(path_key)
|
|
1036
|
+
views.append(
|
|
1037
|
+
GuidedView(
|
|
1038
|
+
id=_story_view_id(section.heading, index),
|
|
1039
|
+
label=section.heading,
|
|
1040
|
+
focus=path,
|
|
1041
|
+
note=f"Follows directly observed relationships cited in the {section.heading} explanation.",
|
|
1042
|
+
)
|
|
1043
|
+
)
|
|
1044
|
+
if len(views) == 5:
|
|
1045
|
+
return views
|
|
1046
|
+
|
|
1047
|
+
if views:
|
|
1048
|
+
return views
|
|
1049
|
+
|
|
1050
|
+
# The aggregation catch-all can touch almost every community. It is useful
|
|
1051
|
+
# in the map but a bad story anchor because its focus would dim nothing.
|
|
1052
|
+
# Prefer named components whenever any are available.
|
|
1053
|
+
anchors = [component for component in components if component.id != "other_components"] or components
|
|
1054
|
+
component_by_id = {component.id: component for component in components}
|
|
1055
|
+
ranked = sorted(
|
|
1056
|
+
anchors,
|
|
1057
|
+
key=lambda component: (-len(adjacency[component.id]), component.label.lower()),
|
|
1058
|
+
)
|
|
1059
|
+
for index, component in enumerate(ranked[:3]):
|
|
1060
|
+
path = _continuous_story_path(
|
|
1061
|
+
adjacency,
|
|
1062
|
+
set(component_by_id),
|
|
1063
|
+
required_id=component.id,
|
|
1064
|
+
)[:5]
|
|
1065
|
+
path_key = frozenset(path)
|
|
1066
|
+
if not path or path_key in seen_paths:
|
|
1067
|
+
continue
|
|
1068
|
+
seen_paths.add(path_key)
|
|
1069
|
+
views.append(
|
|
1070
|
+
GuidedView(
|
|
1071
|
+
id=f"component-{index + 1}",
|
|
1072
|
+
label=component.label,
|
|
1073
|
+
focus=path,
|
|
1074
|
+
note="Follows a connected path through directly observed architecture relationships.",
|
|
1075
|
+
)
|
|
1076
|
+
)
|
|
1077
|
+
return views
|
|
1078
|
+
|
|
1079
|
+
|
|
1080
|
+
def _workflow_relation_key(edge: dict) -> tuple[int, str, str, str]:
|
|
1081
|
+
"""Keep workflow selection deterministic while preferring executable links."""
|
|
1082
|
+
source, target = _edge_endpoints(edge)
|
|
1083
|
+
relation = str(edge.get("relation") or edge.get("label") or "relates to").lower()
|
|
1084
|
+
return (_WORKFLOW_RELATION_ORDER.get(relation, 99), relation, source, target)
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
def _workflow_source_label(node: dict) -> str:
|
|
1088
|
+
"""A compact, code-grounded subtitle for a workflow node."""
|
|
1089
|
+
source_file = _normalized_source_file(node)
|
|
1090
|
+
if not source_file:
|
|
1091
|
+
return "Observed source component"
|
|
1092
|
+
return source_file.rsplit("/", 1)[-1]
|
|
1093
|
+
|
|
1094
|
+
|
|
1095
|
+
def _workflow_display_label(node: dict) -> str:
|
|
1096
|
+
"""Turn a raw code symbol into a readable, bounded diagram label."""
|
|
1097
|
+
raw = str(node.get("label") or node.get("id") or "")
|
|
1098
|
+
# A source helper named ``function_ollama_*`` is not, by itself, proof
|
|
1099
|
+
# that the deployed workflow uses Ollama. Keep the exact symbol in the
|
|
1100
|
+
# hover card and label the diagram block by its code-level responsibility.
|
|
1101
|
+
if raw.casefold().startswith("function_ollama_"):
|
|
1102
|
+
return "LLM Provider Adapter"
|
|
1103
|
+
raw = re.sub(r"^function_", "", raw, flags=re.IGNORECASE)
|
|
1104
|
+
words = re.sub(r"([a-z0-9])([A-Z])", r"\1 \2", raw)
|
|
1105
|
+
words = re.sub(r"[^A-Za-z0-9]+", " ", words).strip().split()
|
|
1106
|
+
label_words: list[str] = []
|
|
1107
|
+
for word in words:
|
|
1108
|
+
candidate = " ".join([*label_words, word])
|
|
1109
|
+
if label_words and len(candidate) > 20:
|
|
1110
|
+
break
|
|
1111
|
+
label_words.append(word)
|
|
1112
|
+
return " ".join(label_words).title() or "Observed Component"
|
|
1113
|
+
|
|
1114
|
+
|
|
1115
|
+
def workflow_hover_details(understanding: GroundedUnderstanding) -> dict[str, dict[str, str]]:
|
|
1116
|
+
"""Return source-grounded hover details for the selected workflow path.
|
|
1117
|
+
|
|
1118
|
+
These details are embedded only in Graphitect's report iframe. The
|
|
1119
|
+
delivered Archify artifact remains byte-for-byte unchanged and validates
|
|
1120
|
+
against Archify's workflow schema on its own.
|
|
1121
|
+
"""
|
|
1122
|
+
path, path_edges = _workflow_path(understanding)
|
|
1123
|
+
node_by_id = {str(node["id"]): node for node in understanding.nodes if node.get("id")}
|
|
1124
|
+
details: dict[str, dict[str, str]] = {}
|
|
1125
|
+
for index, node_id in enumerate(path):
|
|
1126
|
+
node = node_by_id[node_id]
|
|
1127
|
+
raw_symbol = str(node.get("label") or node_id)
|
|
1128
|
+
source_file = _normalized_source_file(node)
|
|
1129
|
+
if index == 0:
|
|
1130
|
+
summary = "Entry in the selected direct code path."
|
|
1131
|
+
elif index == len(path) - 1:
|
|
1132
|
+
summary = "Endpoint of the selected direct code path."
|
|
1133
|
+
else:
|
|
1134
|
+
incoming = str(path_edges[index - 1].get("relation") or "relationship")
|
|
1135
|
+
outgoing = str(path_edges[index].get("relation") or "relationship")
|
|
1136
|
+
summary = (
|
|
1137
|
+
f"Reached through a direct {incoming} relationship and continues "
|
|
1138
|
+
f"through a direct {outgoing} relationship."
|
|
1139
|
+
)
|
|
1140
|
+
if raw_symbol.casefold().startswith("function_ollama_"):
|
|
1141
|
+
summary = (
|
|
1142
|
+
"Code-level LLM provider adapter. Its historical source name is not "
|
|
1143
|
+
"evidence that a particular provider is used."
|
|
1144
|
+
)
|
|
1145
|
+
detail = {"symbol": raw_symbol, "summary": summary}
|
|
1146
|
+
if source_file:
|
|
1147
|
+
detail["source"] = source_file
|
|
1148
|
+
details[f"step-{index + 1}"] = detail
|
|
1149
|
+
return details
|
|
1150
|
+
|
|
1151
|
+
|
|
1152
|
+
def _workflow_node_width(node: dict) -> int:
|
|
1153
|
+
"""Leave enough room for exact code symbols without making a giant card."""
|
|
1154
|
+
label = _workflow_display_label(node)
|
|
1155
|
+
return min(220, max(132, len(label) * 8 + 32))
|
|
1156
|
+
|
|
1157
|
+
|
|
1158
|
+
def _workflow_seed_ids(understanding: GroundedUnderstanding, node_ids: set[str]) -> list[str]:
|
|
1159
|
+
"""Prioritise components the explanation calls a workflow, if available."""
|
|
1160
|
+
workflow_sections = [
|
|
1161
|
+
section for section in understanding.doc if section.heading.lower() == "key workflows"
|
|
1162
|
+
]
|
|
1163
|
+
ordered_sections = [*workflow_sections, *understanding.doc]
|
|
1164
|
+
seeds = [
|
|
1165
|
+
node_id
|
|
1166
|
+
for section in ordered_sections
|
|
1167
|
+
for node_id in section.related_node_ids
|
|
1168
|
+
if node_id in node_ids
|
|
1169
|
+
]
|
|
1170
|
+
return list(dict.fromkeys(seeds))
|
|
1171
|
+
|
|
1172
|
+
|
|
1173
|
+
def _choose_workflow_edge(
|
|
1174
|
+
candidates: list[dict], seen_nodes: set[str], *, incoming: bool
|
|
1175
|
+
) -> dict | None:
|
|
1176
|
+
valid = []
|
|
1177
|
+
for edge in candidates:
|
|
1178
|
+
source, target = _edge_endpoints(edge)
|
|
1179
|
+
next_node = source if incoming else target
|
|
1180
|
+
if next_node and next_node not in seen_nodes:
|
|
1181
|
+
valid.append(edge)
|
|
1182
|
+
return min(valid, key=_workflow_relation_key) if valid else None
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def _workflow_path(understanding: GroundedUnderstanding) -> tuple[list[str], list[dict]]:
|
|
1186
|
+
"""Find one short connected code path without inferring runtime behavior.
|
|
1187
|
+
|
|
1188
|
+
The path follows only direct Graphify edges. It is intentionally not a
|
|
1189
|
+
global longest path: a six-node slice can be read in presentation mode,
|
|
1190
|
+
while a repository-wide path would be both unstable and misleading.
|
|
1191
|
+
"""
|
|
1192
|
+
node_by_id = {str(node.get("id") or ""): node for node in understanding.nodes}
|
|
1193
|
+
node_by_id = {
|
|
1194
|
+
node_id: node for node_id, node in node_by_id.items() if node_id and not _is_test_node(node)
|
|
1195
|
+
}
|
|
1196
|
+
if not node_by_id:
|
|
1197
|
+
raise ValueError("workflow story needs at least one non-test code node")
|
|
1198
|
+
|
|
1199
|
+
incoming: dict[str, list[dict]] = {node_id: [] for node_id in node_by_id}
|
|
1200
|
+
outgoing: dict[str, list[dict]] = {node_id: [] for node_id in node_by_id}
|
|
1201
|
+
for edge in understanding.edges:
|
|
1202
|
+
source, target = _edge_endpoints(edge)
|
|
1203
|
+
if source in node_by_id and target in node_by_id and source != target:
|
|
1204
|
+
outgoing[source].append(edge)
|
|
1205
|
+
incoming[target].append(edge)
|
|
1206
|
+
for edges in [*incoming.values(), *outgoing.values()]:
|
|
1207
|
+
edges.sort(key=_workflow_relation_key)
|
|
1208
|
+
|
|
1209
|
+
cited_seeds = _workflow_seed_ids(understanding, set(node_by_id))
|
|
1210
|
+
ranked_nodes = sorted(
|
|
1211
|
+
node_by_id,
|
|
1212
|
+
key=lambda node_id: (
|
|
1213
|
+
-(len(incoming[node_id]) + len(outgoing[node_id])),
|
|
1214
|
+
node_id,
|
|
1215
|
+
),
|
|
1216
|
+
)
|
|
1217
|
+
seeds = list(dict.fromkeys([*cited_seeds, *ranked_nodes]))
|
|
1218
|
+
best_path: list[str] = []
|
|
1219
|
+
best_edges: list[dict] = []
|
|
1220
|
+
best_score: tuple[int, int, int] = (-1, -1, -1)
|
|
1221
|
+
|
|
1222
|
+
for seed in seeds:
|
|
1223
|
+
path = [seed]
|
|
1224
|
+
path_edges: list[dict] = []
|
|
1225
|
+
seen = {seed}
|
|
1226
|
+
|
|
1227
|
+
# A caller of a cited component makes a better beginning than the
|
|
1228
|
+
# component itself, when Graphify observed one. Then extend forward.
|
|
1229
|
+
first = _choose_workflow_edge(incoming[seed], seen, incoming=True)
|
|
1230
|
+
if first is not None:
|
|
1231
|
+
source, _ = _edge_endpoints(first)
|
|
1232
|
+
path.insert(0, source)
|
|
1233
|
+
path_edges.insert(0, first)
|
|
1234
|
+
seen.add(source)
|
|
1235
|
+
|
|
1236
|
+
while len(path) < _WORKFLOW_MAX_PATH_NODES:
|
|
1237
|
+
next_edge = _choose_workflow_edge(outgoing[path[-1]], seen, incoming=False)
|
|
1238
|
+
if next_edge is None:
|
|
1239
|
+
break
|
|
1240
|
+
_, target = _edge_endpoints(next_edge)
|
|
1241
|
+
path.append(target)
|
|
1242
|
+
path_edges.append(next_edge)
|
|
1243
|
+
seen.add(target)
|
|
1244
|
+
|
|
1245
|
+
# A sink-only cited component may still have a useful direct caller.
|
|
1246
|
+
if len(path) == 1:
|
|
1247
|
+
next_edge = _choose_workflow_edge(outgoing[seed], seen, incoming=False)
|
|
1248
|
+
if next_edge is not None:
|
|
1249
|
+
_, target = _edge_endpoints(next_edge)
|
|
1250
|
+
path.append(target)
|
|
1251
|
+
path_edges.append(next_edge)
|
|
1252
|
+
|
|
1253
|
+
relation_quality = -sum(_workflow_relation_key(edge)[0] for edge in path_edges)
|
|
1254
|
+
score = (
|
|
1255
|
+
sum(node_id in cited_seeds for node_id in path),
|
|
1256
|
+
len(path),
|
|
1257
|
+
relation_quality,
|
|
1258
|
+
)
|
|
1259
|
+
if score > best_score:
|
|
1260
|
+
best_path, best_edges, best_score = path, path_edges, score
|
|
1261
|
+
|
|
1262
|
+
if len(best_path) >= 2:
|
|
1263
|
+
return best_path, best_edges
|
|
1264
|
+
|
|
1265
|
+
# A graph with no cited anchors or executable chain still deserves a
|
|
1266
|
+
# workflow-like view if Graphify observed any relationship at all.
|
|
1267
|
+
for source in ranked_nodes:
|
|
1268
|
+
if outgoing[source]:
|
|
1269
|
+
edge = outgoing[source][0]
|
|
1270
|
+
_, target = _edge_endpoints(edge)
|
|
1271
|
+
return [source, target], [edge]
|
|
1272
|
+
raise ValueError("workflow story needs at least one direct code relationship")
|
|
1273
|
+
|
|
1274
|
+
|
|
1275
|
+
def to_workflow_spec(understanding: GroundedUnderstanding, title: str) -> dict:
|
|
1276
|
+
"""Build a compact Archify workflow artifact from direct graph evidence.
|
|
1277
|
+
|
|
1278
|
+
This is intentionally a separate diagram type from the architecture
|
|
1279
|
+
rollup. Its main path is a continuous, bounded sequence of observed
|
|
1280
|
+
edges; it never turns community aggregates or transitive guesses into a
|
|
1281
|
+
narrated flow.
|
|
1282
|
+
"""
|
|
1283
|
+
if understanding.diagram_kind != "architecture":
|
|
1284
|
+
raise ValueError(
|
|
1285
|
+
f"to_workflow_spec only handles diagram_kind='architecture', got "
|
|
1286
|
+
f"{understanding.diagram_kind!r}"
|
|
1287
|
+
)
|
|
1288
|
+
|
|
1289
|
+
path, path_edges = _workflow_path(understanding)
|
|
1290
|
+
node_by_id = {str(node["id"]): node for node in understanding.nodes if node.get("id")}
|
|
1291
|
+
path_set = set(path)
|
|
1292
|
+
node_ids = {node_id: f"step-{index + 1}" for index, node_id in enumerate(path)}
|
|
1293
|
+
last_col = 5 if len(path) > 1 else 0
|
|
1294
|
+
|
|
1295
|
+
def column(index: int) -> int:
|
|
1296
|
+
return round(index * last_col / (len(path) - 1)) if len(path) > 1 else 0
|
|
1297
|
+
|
|
1298
|
+
nodes = [
|
|
1299
|
+
{
|
|
1300
|
+
"id": node_ids[node_id],
|
|
1301
|
+
"lane": "path",
|
|
1302
|
+
"col": column(index),
|
|
1303
|
+
"type": _classify_component(node_by_id[node_id]),
|
|
1304
|
+
"label": _workflow_display_label(node_by_id[node_id]),
|
|
1305
|
+
"sublabel": _workflow_source_label(node_by_id[node_id]),
|
|
1306
|
+
"width": _workflow_node_width(node_by_id[node_id]),
|
|
1307
|
+
}
|
|
1308
|
+
for index, node_id in enumerate(path)
|
|
1309
|
+
]
|
|
1310
|
+
edges = [
|
|
1311
|
+
{
|
|
1312
|
+
"id": f"path-{index}",
|
|
1313
|
+
"from": node_ids[source],
|
|
1314
|
+
"to": node_ids[target],
|
|
1315
|
+
"label": str(edge.get("relation") or "observed relationship"),
|
|
1316
|
+
"variant": "emphasis" if index == 1 else "default",
|
|
1317
|
+
"role": "main",
|
|
1318
|
+
}
|
|
1319
|
+
for index, (edge, source, target) in enumerate(
|
|
1320
|
+
zip(path_edges, path, path[1:]), start=1
|
|
1321
|
+
)
|
|
1322
|
+
]
|
|
1323
|
+
|
|
1324
|
+
# Add two genuine, directly linked supporting nodes at most. They give
|
|
1325
|
+
# the workflow a second lane without turning it back into a dense map.
|
|
1326
|
+
branch_count = 0
|
|
1327
|
+
for index, node_id in enumerate(path):
|
|
1328
|
+
if branch_count == _WORKFLOW_MAX_BRANCHES:
|
|
1329
|
+
break
|
|
1330
|
+
for edge in sorted(understanding.edges, key=_workflow_relation_key):
|
|
1331
|
+
source, target = _edge_endpoints(edge)
|
|
1332
|
+
other = target if source == node_id else source if target == node_id else ""
|
|
1333
|
+
if not other or other in path_set or other not in node_by_id:
|
|
1334
|
+
continue
|
|
1335
|
+
if _is_test_node(node_by_id[other]) or not _normalized_source_file(node_by_id[other]):
|
|
1336
|
+
continue
|
|
1337
|
+
branch_count += 1
|
|
1338
|
+
branch_id = f"dependency-{branch_count}"
|
|
1339
|
+
nodes.append(
|
|
1340
|
+
{
|
|
1341
|
+
"id": branch_id,
|
|
1342
|
+
"lane": "dependencies",
|
|
1343
|
+
"col": column(index),
|
|
1344
|
+
"type": _classify_component(node_by_id[other]),
|
|
1345
|
+
"label": _workflow_display_label(node_by_id[other]),
|
|
1346
|
+
"sublabel": _workflow_source_label(node_by_id[other]),
|
|
1347
|
+
"width": _workflow_node_width(node_by_id[other]),
|
|
1348
|
+
}
|
|
1349
|
+
)
|
|
1350
|
+
edges.append(
|
|
1351
|
+
{
|
|
1352
|
+
"id": f"branch-{branch_count}",
|
|
1353
|
+
"from": node_ids[node_id] if source == node_id else branch_id,
|
|
1354
|
+
"to": branch_id if source == node_id else node_ids[node_id],
|
|
1355
|
+
"label": str(edge.get("relation") or "observed relationship"),
|
|
1356
|
+
"variant": "dashed",
|
|
1357
|
+
"role": "branch",
|
|
1358
|
+
}
|
|
1359
|
+
)
|
|
1360
|
+
break
|
|
1361
|
+
|
|
1362
|
+
main_path = [node_ids[node_id] for node_id in path]
|
|
1363
|
+
phases = [
|
|
1364
|
+
{"id": "entry", "label": "Entry", "fromCol": 0, "toCol": min(1, last_col)},
|
|
1365
|
+
{
|
|
1366
|
+
"id": "trace",
|
|
1367
|
+
"label": "Observed code path",
|
|
1368
|
+
"fromCol": min(2, last_col),
|
|
1369
|
+
"toCol": min(3, last_col),
|
|
1370
|
+
"variant": "emphasis",
|
|
1371
|
+
},
|
|
1372
|
+
{
|
|
1373
|
+
"id": "endpoint",
|
|
1374
|
+
"label": "Endpoint",
|
|
1375
|
+
"fromCol": min(4, last_col),
|
|
1376
|
+
"toCol": last_col,
|
|
1377
|
+
"variant": "dashed",
|
|
1378
|
+
},
|
|
1379
|
+
]
|
|
1380
|
+
return {
|
|
1381
|
+
"schema_version": 2,
|
|
1382
|
+
"diagram_type": "workflow",
|
|
1383
|
+
"meta": {
|
|
1384
|
+
"title": f"{title} · observed code path",
|
|
1385
|
+
"animation": "trace",
|
|
1386
|
+
"visual_preset": "signal-flow",
|
|
1387
|
+
"quality_profile": "showcase",
|
|
1388
|
+
"views": [
|
|
1389
|
+
{
|
|
1390
|
+
"id": "traced-path",
|
|
1391
|
+
"label": "Traced code path",
|
|
1392
|
+
"focus": main_path,
|
|
1393
|
+
"note": "Follow each direct Graphify relationship in order; no transitive links are added.",
|
|
1394
|
+
}
|
|
1395
|
+
],
|
|
1396
|
+
},
|
|
1397
|
+
"lanes": [
|
|
1398
|
+
{"id": "path", "label": "Observed code path"},
|
|
1399
|
+
{"id": "dependencies", "label": "Direct dependencies"},
|
|
1400
|
+
],
|
|
1401
|
+
"phases": phases,
|
|
1402
|
+
"mainPath": main_path,
|
|
1403
|
+
"nodes": nodes,
|
|
1404
|
+
"edges": edges,
|
|
1405
|
+
"cards": [
|
|
1406
|
+
{
|
|
1407
|
+
"dot": "cyan",
|
|
1408
|
+
"title": "Evidence boundary",
|
|
1409
|
+
"items": [
|
|
1410
|
+
"Every arrow is a direct Graphify relationship.",
|
|
1411
|
+
"Node subtitles identify the observed source file.",
|
|
1412
|
+
],
|
|
1413
|
+
}
|
|
1414
|
+
],
|
|
1415
|
+
}
|
|
1416
|
+
|
|
1417
|
+
|
|
1418
|
+
def render_workflow_story(
|
|
1419
|
+
understanding: GroundedUnderstanding,
|
|
1420
|
+
title: str,
|
|
1421
|
+
out_path: Path,
|
|
1422
|
+
*,
|
|
1423
|
+
archify_bin: str | None = None,
|
|
1424
|
+
auto_install: bool = True,
|
|
1425
|
+
) -> Path:
|
|
1426
|
+
"""Render the compact, presentation-safe workflow with bundled Archify."""
|
|
1427
|
+
bin_cmd = _resolve_archify_bin(archify_bin, auto_install=auto_install)
|
|
1428
|
+
spec_path = out_path.with_suffix(".workflow.json")
|
|
1429
|
+
spec_path.write_text(json.dumps(to_workflow_spec(understanding, title), indent=2), encoding="utf-8")
|
|
1430
|
+
report = archify_repair.run_deliver_json(
|
|
1431
|
+
bin_cmd, spec_path, out_path, diagram_type="workflow"
|
|
1432
|
+
)
|
|
1433
|
+
if report.get("ok"):
|
|
1434
|
+
return out_path
|
|
1435
|
+
raise subprocess.CalledProcessError(
|
|
1436
|
+
1,
|
|
1437
|
+
[*bin_cmd, "deliver", "workflow", str(spec_path), str(out_path)],
|
|
1438
|
+
output=json.dumps(report),
|
|
1439
|
+
)
|
|
1440
|
+
|
|
1441
|
+
|
|
1442
|
+
def _sequence_relation(edge: dict) -> str:
|
|
1443
|
+
return str(edge.get("relation") or edge.get("label") or "").strip().lower()
|
|
1444
|
+
|
|
1445
|
+
|
|
1446
|
+
def _is_observed_static_call(edge: dict) -> bool:
|
|
1447
|
+
"""Require extracted call evidence when Graphify provides confidence.
|
|
1448
|
+
|
|
1449
|
+
Graphify can add inferred symbol-resolution links to its graph. They are
|
|
1450
|
+
useful in the architecture explorer, but a sequence must not turn them
|
|
1451
|
+
into an apparently observed call trace.
|
|
1452
|
+
"""
|
|
1453
|
+
if _sequence_relation(edge) not in _SEQUENCE_RELATIONS:
|
|
1454
|
+
return False
|
|
1455
|
+
confidence = str(edge.get("confidence") or "").strip().upper()
|
|
1456
|
+
context = str(edge.get("context") or "").strip().lower()
|
|
1457
|
+
return (not confidence or confidence == "EXTRACTED") and (not context or context == "call")
|
|
1458
|
+
|
|
1459
|
+
|
|
1460
|
+
def _sequence_path(understanding: GroundedUnderstanding) -> tuple[list[str], list[dict]]:
|
|
1461
|
+
"""Find one bounded, ordered chain of direct static call edges.
|
|
1462
|
+
|
|
1463
|
+
This deliberately does *not* infer a runtime request path. A sequence is
|
|
1464
|
+
offered only when Graphify found at least two consecutive ``calls`` or
|
|
1465
|
+
``invokes`` relationships between non-test symbols.
|
|
1466
|
+
"""
|
|
1467
|
+
node_by_id = {str(node.get("id") or ""): node for node in understanding.nodes}
|
|
1468
|
+
node_by_id = {
|
|
1469
|
+
node_id: node for node_id, node in node_by_id.items() if node_id and not _is_test_node(node)
|
|
1470
|
+
}
|
|
1471
|
+
if not node_by_id:
|
|
1472
|
+
raise ValueError("sequence trace needs non-test code nodes")
|
|
1473
|
+
|
|
1474
|
+
outgoing: dict[str, list[dict]] = {node_id: [] for node_id in node_by_id}
|
|
1475
|
+
for edge in understanding.edges:
|
|
1476
|
+
source, target = _edge_endpoints(edge)
|
|
1477
|
+
if (
|
|
1478
|
+
source in node_by_id
|
|
1479
|
+
and target in node_by_id
|
|
1480
|
+
and source != target
|
|
1481
|
+
and _is_observed_static_call(edge)
|
|
1482
|
+
):
|
|
1483
|
+
outgoing[source].append(edge)
|
|
1484
|
+
for edges in outgoing.values():
|
|
1485
|
+
edges.sort(key=_workflow_relation_key)
|
|
1486
|
+
|
|
1487
|
+
cited_seeds = set(_workflow_seed_ids(understanding, set(node_by_id)))
|
|
1488
|
+
starts = sorted(
|
|
1489
|
+
node_by_id,
|
|
1490
|
+
key=lambda node_id: (
|
|
1491
|
+
node_id not in cited_seeds,
|
|
1492
|
+
-(len(outgoing[node_id])),
|
|
1493
|
+
node_id,
|
|
1494
|
+
),
|
|
1495
|
+
)
|
|
1496
|
+
best_path: list[str] = []
|
|
1497
|
+
best_edges: list[dict] = []
|
|
1498
|
+
best_score = (-1, -1, -1)
|
|
1499
|
+
|
|
1500
|
+
for start in starts:
|
|
1501
|
+
path = [start]
|
|
1502
|
+
path_edges: list[dict] = []
|
|
1503
|
+
seen = {start}
|
|
1504
|
+
while len(path) < _SEQUENCE_MAX_PARTICIPANTS:
|
|
1505
|
+
next_edge = _choose_workflow_edge(outgoing[path[-1]], seen, incoming=False)
|
|
1506
|
+
if next_edge is None:
|
|
1507
|
+
break
|
|
1508
|
+
_, target = _edge_endpoints(next_edge)
|
|
1509
|
+
path.append(target)
|
|
1510
|
+
path_edges.append(next_edge)
|
|
1511
|
+
seen.add(target)
|
|
1512
|
+
|
|
1513
|
+
if len(path_edges) < 2:
|
|
1514
|
+
continue
|
|
1515
|
+
score = (
|
|
1516
|
+
sum(node_id in cited_seeds for node_id in path),
|
|
1517
|
+
len(path),
|
|
1518
|
+
-sum(_workflow_relation_key(edge)[0] for edge in path_edges),
|
|
1519
|
+
)
|
|
1520
|
+
if score > best_score:
|
|
1521
|
+
best_path, best_edges, best_score = path, path_edges, score
|
|
1522
|
+
|
|
1523
|
+
if len(best_edges) < 2:
|
|
1524
|
+
raise ValueError(
|
|
1525
|
+
"sequence trace needs two consecutive direct Graphify calls or invocations"
|
|
1526
|
+
)
|
|
1527
|
+
return best_path, best_edges
|
|
1528
|
+
|
|
1529
|
+
|
|
1530
|
+
def to_sequence_spec(understanding: GroundedUnderstanding, title: str) -> dict:
|
|
1531
|
+
"""Build an evidence-gated Archify sequence from direct static calls.
|
|
1532
|
+
|
|
1533
|
+
The explicit subtitle and card prevent readers from mistaking source
|
|
1534
|
+
analysis for request telemetry, timing data, or a captured trace.
|
|
1535
|
+
"""
|
|
1536
|
+
if understanding.diagram_kind != "architecture":
|
|
1537
|
+
raise ValueError(
|
|
1538
|
+
f"to_sequence_spec only handles diagram_kind='architecture', got "
|
|
1539
|
+
f"{understanding.diagram_kind!r}"
|
|
1540
|
+
)
|
|
1541
|
+
|
|
1542
|
+
path, path_edges = _sequence_path(understanding)
|
|
1543
|
+
node_by_id = {str(node["id"]): node for node in understanding.nodes if node.get("id")}
|
|
1544
|
+
participant_ids = {node_id: f"step-{index + 1}" for index, node_id in enumerate(path)}
|
|
1545
|
+
message_start_y = 170
|
|
1546
|
+
message_gap = 58
|
|
1547
|
+
view_box_height = max(480, message_start_y + len(path_edges) * message_gap + 100)
|
|
1548
|
+
view_box_width = max(720, len(path) * 180)
|
|
1549
|
+
|
|
1550
|
+
messages = [
|
|
1551
|
+
{
|
|
1552
|
+
"id": f"call-{index}",
|
|
1553
|
+
"from": participant_ids[source],
|
|
1554
|
+
"to": participant_ids[target],
|
|
1555
|
+
"y": message_start_y + (index - 1) * message_gap,
|
|
1556
|
+
"label": _sequence_relation(edge),
|
|
1557
|
+
"variant": "emphasis",
|
|
1558
|
+
}
|
|
1559
|
+
for index, (edge, source, target) in enumerate(
|
|
1560
|
+
zip(path_edges, path, path[1:]), start=1
|
|
1561
|
+
)
|
|
1562
|
+
]
|
|
1563
|
+
return {
|
|
1564
|
+
"schema_version": 1,
|
|
1565
|
+
"diagram_type": "sequence",
|
|
1566
|
+
"meta": {
|
|
1567
|
+
"title": f"{title} · observed static call sequence",
|
|
1568
|
+
"subtitle": "Direct Graphify calls/invocations; not runtime telemetry.",
|
|
1569
|
+
"viewBox": [view_box_width, view_box_height],
|
|
1570
|
+
"column_fit": "spread",
|
|
1571
|
+
"animation": "trace",
|
|
1572
|
+
"visual_preset": "signal-flow",
|
|
1573
|
+
"quality_profile": "showcase",
|
|
1574
|
+
"legend": {
|
|
1575
|
+
"mode": "auto",
|
|
1576
|
+
"entries": {"emphasis": {"label": "direct static call"}},
|
|
1577
|
+
},
|
|
1578
|
+
"views": [
|
|
1579
|
+
{
|
|
1580
|
+
"id": "observed-static-call-sequence",
|
|
1581
|
+
"label": "Observed call chain",
|
|
1582
|
+
"focus": [participant_ids[node_id] for node_id in path],
|
|
1583
|
+
"note": "Each arrow is a direct extracted Graphify calls/invokes relationship.",
|
|
1584
|
+
}
|
|
1585
|
+
],
|
|
1586
|
+
},
|
|
1587
|
+
"participants": [
|
|
1588
|
+
{
|
|
1589
|
+
"id": participant_ids[node_id],
|
|
1590
|
+
"type": _classify_component(node_by_id[node_id]),
|
|
1591
|
+
"label": _workflow_display_label(node_by_id[node_id]),
|
|
1592
|
+
}
|
|
1593
|
+
for node_id in path
|
|
1594
|
+
],
|
|
1595
|
+
"messages": messages,
|
|
1596
|
+
"activations": [
|
|
1597
|
+
{
|
|
1598
|
+
"participant": participant_ids[node_id],
|
|
1599
|
+
"from": message_start_y + index * message_gap - 6,
|
|
1600
|
+
"to": min(
|
|
1601
|
+
view_box_height - 22,
|
|
1602
|
+
message_start_y + (index + 1) * message_gap + 6,
|
|
1603
|
+
),
|
|
1604
|
+
"type": _classify_component(node_by_id[node_id]),
|
|
1605
|
+
}
|
|
1606
|
+
for index, node_id in enumerate(path[1:])
|
|
1607
|
+
],
|
|
1608
|
+
"cards": [
|
|
1609
|
+
{
|
|
1610
|
+
"dot": "cyan",
|
|
1611
|
+
"title": "Evidence boundary",
|
|
1612
|
+
"items": [
|
|
1613
|
+
"Every arrow is a direct extracted Graphify calls/invokes edge.",
|
|
1614
|
+
"No runtime request, response, timing, or trace is implied.",
|
|
1615
|
+
],
|
|
1616
|
+
}
|
|
1617
|
+
],
|
|
1618
|
+
}
|
|
1619
|
+
|
|
1620
|
+
|
|
1621
|
+
def render_sequence_trace(
|
|
1622
|
+
understanding: GroundedUnderstanding,
|
|
1623
|
+
title: str,
|
|
1624
|
+
out_path: Path,
|
|
1625
|
+
*,
|
|
1626
|
+
archify_bin: str | None = None,
|
|
1627
|
+
auto_install: bool = True,
|
|
1628
|
+
) -> Path:
|
|
1629
|
+
"""Render the optional evidence-gated sequence with bundled Archify."""
|
|
1630
|
+
bin_cmd = _resolve_archify_bin(archify_bin, auto_install=auto_install)
|
|
1631
|
+
spec_path = out_path.with_suffix(".sequence.json")
|
|
1632
|
+
spec_path.write_text(json.dumps(to_sequence_spec(understanding, title), indent=2), encoding="utf-8")
|
|
1633
|
+
report = archify_repair.run_deliver_json(
|
|
1634
|
+
bin_cmd, spec_path, out_path, diagram_type="sequence"
|
|
1635
|
+
)
|
|
1636
|
+
if report.get("ok"):
|
|
1637
|
+
return out_path
|
|
1638
|
+
raise subprocess.CalledProcessError(
|
|
1639
|
+
1,
|
|
1640
|
+
[*bin_cmd, "deliver", "sequence", str(spec_path), str(out_path)],
|
|
1641
|
+
output=json.dumps(report),
|
|
1642
|
+
)
|
|
1643
|
+
|
|
1644
|
+
|
|
1645
|
+
def to_architecture_ir(
|
|
1646
|
+
understanding: GroundedUnderstanding,
|
|
1647
|
+
title: str,
|
|
1648
|
+
*,
|
|
1649
|
+
spacing_multiplier: float = 1.0,
|
|
1650
|
+
view: Literal["story", "overview", "full"] = "full",
|
|
1651
|
+
) -> ArchitectureIR:
|
|
1652
|
+
"""`spacing_multiplier` widens the vertical gap between grid rows beyond
|
|
1653
|
+
the normal default - the only knob render() has to make the layout more
|
|
1654
|
+
forgiving without an LLM. Confirmed live (12 Sep 2026): even a genuinely
|
|
1655
|
+
small, non-aggregated 5-node graph can fail archify's validator with
|
|
1656
|
+
"label overlaps component" (archify's own default auto-placement for a
|
|
1657
|
+
connector's label sometimes lands inside the FROM box when the row gap
|
|
1658
|
+
is tight) - a `layout/constraint` failure, not the large-graph
|
|
1659
|
+
`clean-flow/edge-through-node` crossing issue the LLM repair loop
|
|
1660
|
+
targets. Since "no LLM key configured" must not mean "no diagram"
|
|
1661
|
+
(Naman, 12 Sep 2026: "the diagram is a must"), render() retries this
|
|
1662
|
+
purely mechanically with progressively more room before giving up.
|
|
1663
|
+
"""
|
|
1664
|
+
if understanding.diagram_kind != "architecture":
|
|
1665
|
+
raise ValueError(
|
|
1666
|
+
f"to_architecture_ir only handles diagram_kind='architecture', got "
|
|
1667
|
+
f"{understanding.diagram_kind!r}"
|
|
1668
|
+
)
|
|
1669
|
+
|
|
1670
|
+
if view not in {"story", "overview", "full"}:
|
|
1671
|
+
raise ValueError(f"view must be 'story', 'overview', or 'full', got {view!r}")
|
|
1672
|
+
|
|
1673
|
+
nodes, all_edges = understanding.nodes, understanding.edges
|
|
1674
|
+
if len(nodes) > _MAX_RAW_COMPONENTS:
|
|
1675
|
+
nodes, all_edges = _aggregate_by_community(nodes, all_edges, understanding.community_labels)
|
|
1676
|
+
|
|
1677
|
+
# Both tabs use the same component positions. Filtering overview edges
|
|
1678
|
+
# must not rewrite the dependency layers or make the two maps disagree.
|
|
1679
|
+
positions = _assign_grid(nodes, all_edges, cols=_GRID_COLUMNS)
|
|
1680
|
+
# Story shares Overview's compact structural subset. Its additional value
|
|
1681
|
+
# is guided focus, not a second dense map.
|
|
1682
|
+
edges = _select_overview_edges(nodes, all_edges) if view in {"story", "overview"} else all_edges
|
|
1683
|
+
labels = [n.get("label", n["id"]) for n in nodes]
|
|
1684
|
+
box_width, box_height = _estimate_box_size(labels)
|
|
1685
|
+
|
|
1686
|
+
components = [
|
|
1687
|
+
Component(
|
|
1688
|
+
id=n["id"],
|
|
1689
|
+
type=_classify_component(n),
|
|
1690
|
+
label=n.get("label", n["id"]),
|
|
1691
|
+
sublabel=n.get("sublabel"),
|
|
1692
|
+
row=positions[n["id"]][0],
|
|
1693
|
+
col=positions[n["id"]][1],
|
|
1694
|
+
size=(box_width, box_height),
|
|
1695
|
+
)
|
|
1696
|
+
for n in nodes
|
|
1697
|
+
]
|
|
1698
|
+
connections = [
|
|
1699
|
+
Connection(
|
|
1700
|
+
id=e.get("id") or f"{e.get('from') or e.get('source')}-{e.get('to') or e.get('target')}",
|
|
1701
|
+
**{"from": e.get("from") or e.get("source")},
|
|
1702
|
+
to=e.get("to") or e.get("target"),
|
|
1703
|
+
label=e.get("label"),
|
|
1704
|
+
)
|
|
1705
|
+
for e in edges
|
|
1706
|
+
]
|
|
1707
|
+
gap_y = int(max(60, box_height + 20) * spacing_multiplier)
|
|
1708
|
+
connections = _route_around_obstacles(
|
|
1709
|
+
connections,
|
|
1710
|
+
positions,
|
|
1711
|
+
box_width=box_width,
|
|
1712
|
+
box_height=box_height,
|
|
1713
|
+
cell_w=box_width + 30,
|
|
1714
|
+
cell_h=box_height,
|
|
1715
|
+
gap_x=_GRID_GAP_X,
|
|
1716
|
+
gap_y=gap_y,
|
|
1717
|
+
cols=_GRID_COLUMNS,
|
|
1718
|
+
)
|
|
1719
|
+
view_box = _view_box_for_routes(
|
|
1720
|
+
positions,
|
|
1721
|
+
connections,
|
|
1722
|
+
box_width=box_width,
|
|
1723
|
+
box_height=box_height,
|
|
1724
|
+
cell_w=box_width + 30,
|
|
1725
|
+
cell_h=box_height,
|
|
1726
|
+
gap_x=_GRID_GAP_X,
|
|
1727
|
+
gap_y=gap_y,
|
|
1728
|
+
)
|
|
1729
|
+
return ArchitectureIR(
|
|
1730
|
+
meta=Meta(
|
|
1731
|
+
title=title,
|
|
1732
|
+
viewBox=view_box,
|
|
1733
|
+
animation="trace" if view == "story" else "none",
|
|
1734
|
+
visual_preset="signal-flow" if view == "story" else None,
|
|
1735
|
+
views=_story_views(understanding, components, all_edges) if view == "story" else [],
|
|
1736
|
+
),
|
|
1737
|
+
layout=GridLayout(
|
|
1738
|
+
cols=_GRID_COLUMNS,
|
|
1739
|
+
cellW=box_width + 30,
|
|
1740
|
+
cellH=box_height,
|
|
1741
|
+
gapX=_GRID_GAP_X,
|
|
1742
|
+
# Wider than archify's own 40px default - confirmed live that a
|
|
1743
|
+
# labeled connection between two vertically-adjacent grid rows
|
|
1744
|
+
# otherwise has nowhere to sit without overlapping one of the
|
|
1745
|
+
# two component boxes it connects (multiple "label overlaps
|
|
1746
|
+
# component" failures against a real 12-node graph).
|
|
1747
|
+
gapY=gap_y,
|
|
1748
|
+
),
|
|
1749
|
+
components=components,
|
|
1750
|
+
connections=connections,
|
|
1751
|
+
)
|
|
1752
|
+
|
|
1753
|
+
|
|
1754
|
+
def _resolve_archify_bin(archify_bin: str | None, *, auto_install: bool) -> list[str]:
|
|
1755
|
+
"""Return the explicit override or Graphitect's bundled Archify runtime.
|
|
1756
|
+
|
|
1757
|
+
``auto_install`` remains in the signature for API compatibility with the
|
|
1758
|
+
earlier bridge, but is intentionally ignored: a Graphitect run must never
|
|
1759
|
+
download another tool at runtime.
|
|
1760
|
+
"""
|
|
1761
|
+
if archify_bin:
|
|
1762
|
+
# shlex.split handles both a single executable ("archify", once on
|
|
1763
|
+
# PATH) and a multi-token invocation ("node C:\path\to\archify.mjs").
|
|
1764
|
+
# posix=False is required on Windows: shlex's default POSIX mode
|
|
1765
|
+
# treats backslash as an escape character and silently mangles
|
|
1766
|
+
# Windows paths (confirmed live - "node C:\Users\...\archify.mjs"
|
|
1767
|
+
# became "C:UsersnamanOneDrive..." with every backslash-letter pair
|
|
1768
|
+
# eaten as a fake escape sequence).
|
|
1769
|
+
return shlex.split(archify_bin, posix=False)
|
|
1770
|
+
|
|
1771
|
+
if not _BUNDLED_ARCHIFY_PATH.exists():
|
|
1772
|
+
raise FileNotFoundError(
|
|
1773
|
+
"bundled Archify runtime is missing from this Graphitect installation"
|
|
1774
|
+
)
|
|
1775
|
+
|
|
1776
|
+
node = shutil.which("node")
|
|
1777
|
+
if not node:
|
|
1778
|
+
raise FileNotFoundError(
|
|
1779
|
+
"Graphitect includes Archify, but Node.js 18+ is required to run "
|
|
1780
|
+
"the bundled renderer"
|
|
1781
|
+
)
|
|
1782
|
+
return [node, str(_BUNDLED_ARCHIFY_PATH)]
|
|
1783
|
+
|
|
1784
|
+
|
|
1785
|
+
def render(
|
|
1786
|
+
understanding: GroundedUnderstanding,
|
|
1787
|
+
title: str,
|
|
1788
|
+
out_path: Path,
|
|
1789
|
+
*,
|
|
1790
|
+
archify_bin: str | None = None,
|
|
1791
|
+
auto_install: bool = True,
|
|
1792
|
+
repair_backend: LLMBackend | None = None,
|
|
1793
|
+
max_repair_iterations: int = 3,
|
|
1794
|
+
view: Literal["story", "overview", "full"] = "full",
|
|
1795
|
+
) -> Path:
|
|
1796
|
+
"""Write the archify IR, then shell out to `archify deliver` to render it.
|
|
1797
|
+
|
|
1798
|
+
Archify ships inside Graphitect and is never installed or downloaded at
|
|
1799
|
+
runtime. ``auto_install`` is accepted only for backward compatibility.
|
|
1800
|
+
|
|
1801
|
+
When `repair_backend` is given, a rejected layout (archify's own strict
|
|
1802
|
+
validator - dense real-world graphs routinely fail on edges crossing
|
|
1803
|
+
unrelated components) is handed to archify_repair's LLM loop instead of
|
|
1804
|
+
failing outright: it patches the IR's routing/label-position fields
|
|
1805
|
+
against archify's real structured diagnostics and retries, up to
|
|
1806
|
+
`max_repair_iterations` total attempts. The repair is optional: provider
|
|
1807
|
+
failures and unresolved repair diagnostics fall back to the deterministic
|
|
1808
|
+
renderer, so an LLM quota response can never abort `graphitect build`.
|
|
1809
|
+
|
|
1810
|
+
Without a backend, there's no LLM available to patch anything, but "no
|
|
1811
|
+
key configured" must still not mean "no diagram" (Naman, 12 Sep 2026:
|
|
1812
|
+
"the diagram is a must") - confirmed live that even a small, non-
|
|
1813
|
+
aggregated graph can fail archify's validator on tight label spacing
|
|
1814
|
+
alone (see to_architecture_ir's spacing_multiplier docstring), so this
|
|
1815
|
+
retries the deterministic layout a few times with progressively more
|
|
1816
|
+
room before finally raising CalledProcessError for the caller to handle.
|
|
1817
|
+
"""
|
|
1818
|
+
bin_cmd = _resolve_archify_bin(archify_bin, auto_install=auto_install)
|
|
1819
|
+
ir_path = out_path.with_suffix(".architecture.json")
|
|
1820
|
+
|
|
1821
|
+
if repair_backend is not None:
|
|
1822
|
+
ir = to_architecture_ir(understanding, title, view=view)
|
|
1823
|
+
try:
|
|
1824
|
+
report = archify_repair.repair_and_deliver(
|
|
1825
|
+
ir, bin_cmd, ir_path, out_path, repair_backend, max_iterations=max_repair_iterations
|
|
1826
|
+
)
|
|
1827
|
+
except archify_repair.RepairUnavailableError:
|
|
1828
|
+
warnings.warn(
|
|
1829
|
+
"Archify's optional LLM layout repair was unavailable; "
|
|
1830
|
+
"using the deterministic renderer.",
|
|
1831
|
+
RuntimeWarning,
|
|
1832
|
+
stacklevel=2,
|
|
1833
|
+
)
|
|
1834
|
+
else:
|
|
1835
|
+
if report.get("ok"):
|
|
1836
|
+
return out_path
|
|
1837
|
+
warnings.warn(
|
|
1838
|
+
"Archify's optional LLM layout repair did not resolve the layout; "
|
|
1839
|
+
"using the deterministic renderer.",
|
|
1840
|
+
RuntimeWarning,
|
|
1841
|
+
stacklevel=2,
|
|
1842
|
+
)
|
|
1843
|
+
|
|
1844
|
+
last_report: dict | None = None
|
|
1845
|
+
for multiplier in _SPACING_RETRY_MULTIPLIERS:
|
|
1846
|
+
ir = to_architecture_ir(understanding, title, spacing_multiplier=multiplier, view=view)
|
|
1847
|
+
ir_path.write_text(json.dumps(ir.model_dump_archify(), indent=2), encoding="utf-8")
|
|
1848
|
+
report = archify_repair.run_deliver_json(bin_cmd, ir_path, out_path)
|
|
1849
|
+
if report.get("ok"):
|
|
1850
|
+
return out_path
|
|
1851
|
+
last_report = report
|
|
1852
|
+
|
|
1853
|
+
# archify's own validator already computed the exact fix for a
|
|
1854
|
+
# "layout/constraint" label-overlap failure (a concrete suggested
|
|
1855
|
+
# labelAt) - confirmed live (12 Sep 2026) that widening gapY alone
|
|
1856
|
+
# doesn't move a fixed-offset default label at all, so apply the
|
|
1857
|
+
# suggested fix directly and retry, repeating up to
|
|
1858
|
+
# _MAX_LABEL_FIX_ROUNDS times at this same spacing level (fixing one
|
|
1859
|
+
# overlap can reveal or shift another on a real dense graph -
|
|
1860
|
+
# confirmed live that a single attempt wasn't always enough) before
|
|
1861
|
+
# escalating to more room.
|
|
1862
|
+
for _ in range(_MAX_LABEL_FIX_ROUNDS):
|
|
1863
|
+
patched_ir = _apply_suggested_label_fixes(ir, last_report.get("diagnostics") or [])
|
|
1864
|
+
if patched_ir is None:
|
|
1865
|
+
break
|
|
1866
|
+
ir = patched_ir
|
|
1867
|
+
ir_path.write_text(json.dumps(ir.model_dump_archify(), indent=2), encoding="utf-8")
|
|
1868
|
+
report = archify_repair.run_deliver_json(bin_cmd, ir_path, out_path)
|
|
1869
|
+
if report.get("ok"):
|
|
1870
|
+
return out_path
|
|
1871
|
+
last_report = report
|
|
1872
|
+
|
|
1873
|
+
raise subprocess.CalledProcessError(
|
|
1874
|
+
1,
|
|
1875
|
+
[*bin_cmd, "deliver", "architecture", str(ir_path), str(out_path)],
|
|
1876
|
+
output=json.dumps(last_report),
|
|
1877
|
+
)
|