graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
"""Orchestrates the LLM draft step (llm_backend) with the rubric's own
|
|
2
|
+
verification (rubric.py) - the draft is never trusted blindly. A "confirmed"
|
|
3
|
+
claim citing a repository file gets re-checked against the actual file; if
|
|
4
|
+
the cited text isn't really there, the claim is downgraded to inferred
|
|
5
|
+
rather than shipping a fabricated citation. A file-less ``code`` citation is
|
|
6
|
+
reserved for Graphitect's own computed Graphify facts; user citations remain
|
|
7
|
+
the user's own statement.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import subprocess
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from ..models import Claim, Confidence, DesignDocSection, Evidence, GroundedUnderstanding
|
|
16
|
+
from .llm_backend import LLMBackend
|
|
17
|
+
from .rubric import build_question
|
|
18
|
+
|
|
19
|
+
# Keep the draft prompt bounded; truncate, don't fail, on a huge repo. 200K
|
|
20
|
+
# chars (~50K tokens) comfortably covers a well-documented small-to-medium
|
|
21
|
+
# repo's README+CHANGELOG+docs without truncating mid-document - the
|
|
22
|
+
# previous 40K limit was confirmed live to truncate git-resume-agent's own
|
|
23
|
+
# README+CHANGELOG (44KB combined) before the model ever saw all of it,
|
|
24
|
+
# directly undercutting doc richness on exactly the repos this matters most
|
|
25
|
+
# for. Still a single fixed number across every backend regardless of its
|
|
26
|
+
# actual context window (Gemini's is far larger than a small local Ollama
|
|
27
|
+
# model's) - real per-backend tuning is future work, not solved here.
|
|
28
|
+
_MAX_CONTEXT_CHARS = 200_000
|
|
29
|
+
|
|
30
|
+
# How many of the busiest source files (by node count - a proxy for
|
|
31
|
+
# architectural significance, the same "god nodes" intuition
|
|
32
|
+
# archify_adapter's community-overflow cap uses) get a real code excerpt
|
|
33
|
+
# pulled into the prompt, and how much of each. Naman's request (12 Sep
|
|
34
|
+
# 2026) for a more in-depth, explanatory doc needed real mechanism-level
|
|
35
|
+
# material to explain FROM, not just a flat node/edge list - a node/edge
|
|
36
|
+
# summary alone tells the model *what* is connected, never *how* it works.
|
|
37
|
+
_MAX_CODE_EXCERPT_FILES = 6
|
|
38
|
+
_MAX_CODE_EXCERPT_CHARS = 2_000
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def build_context(grounding: dict, repo_path: Path) -> str:
|
|
42
|
+
"""Combine Graphify's structure (if present) with repo text (README,
|
|
43
|
+
CHANGELOG, docs) and a handful of real code excerpts into one
|
|
44
|
+
prompt-sized context. Graphify tells the model *what* is connected; the
|
|
45
|
+
repo text and code excerpts are where *why* and *how* actually come from
|
|
46
|
+
(plan.md §02).
|
|
47
|
+
"""
|
|
48
|
+
parts: list[str] = []
|
|
49
|
+
|
|
50
|
+
if grounding.get("mode") == "graphify":
|
|
51
|
+
nodes = grounding.get("nodes", [])
|
|
52
|
+
edges = grounding.get("edges", [])
|
|
53
|
+
parts.append(f"Structure graph: {len(nodes)} nodes, {len(edges)} edges.")
|
|
54
|
+
node_lines = [f"- {n.get('id')}: {n.get('label', n.get('id'))}" for n in nodes[:200]]
|
|
55
|
+
parts.append("Nodes:\n" + "\n".join(node_lines))
|
|
56
|
+
edge_lines = [
|
|
57
|
+
f"- {e.get('source') or e.get('from')} -> {e.get('target') or e.get('to')}"
|
|
58
|
+
f" ({e.get('relation', e.get('label', 'related')) })"
|
|
59
|
+
for e in edges[:300]
|
|
60
|
+
]
|
|
61
|
+
parts.append("Edges:\n" + "\n".join(edge_lines))
|
|
62
|
+
|
|
63
|
+
# Graphify's own `source_file` field on AST nodes (confirmed live
|
|
64
|
+
# against FundersAI's real graph.json) names exactly which real file
|
|
65
|
+
# backs each node; the files referenced by the most nodes are a
|
|
66
|
+
# reasonable proxy for "central to the architecture" without any
|
|
67
|
+
# extra analysis. Best-effort: a file that no longer exists (stale
|
|
68
|
+
# grounding, a moved file) is silently skipped, same tolerance
|
|
69
|
+
# README/docs reading below already has.
|
|
70
|
+
file_counts: dict[str, int] = {}
|
|
71
|
+
for n in nodes:
|
|
72
|
+
source_file = n.get("source_file")
|
|
73
|
+
if source_file:
|
|
74
|
+
file_counts[source_file] = file_counts.get(source_file, 0) + 1
|
|
75
|
+
top_files = sorted(file_counts, key=file_counts.get, reverse=True)[:_MAX_CODE_EXCERPT_FILES]
|
|
76
|
+
for source_file in top_files:
|
|
77
|
+
f = repo_path / source_file
|
|
78
|
+
if f.is_file():
|
|
79
|
+
excerpt = f.read_text(encoding="utf-8", errors="ignore")[:_MAX_CODE_EXCERPT_CHARS]
|
|
80
|
+
parts.append(f"\n--- code excerpt: {source_file} ---\n{excerpt}")
|
|
81
|
+
elif grounding.get("mode") == "describe":
|
|
82
|
+
parts.append("User's own description:")
|
|
83
|
+
for claim in grounding.get("claims", []):
|
|
84
|
+
parts.append(f"- {claim.get('text')}")
|
|
85
|
+
|
|
86
|
+
for name in ("README.md", "CHANGELOG.md"):
|
|
87
|
+
f = repo_path / name
|
|
88
|
+
if f.exists():
|
|
89
|
+
parts.append(f"\n--- {name} ---\n" + f.read_text(encoding="utf-8", errors="ignore"))
|
|
90
|
+
|
|
91
|
+
docs_dir = repo_path / "docs"
|
|
92
|
+
if docs_dir.is_dir():
|
|
93
|
+
for doc in sorted(docs_dir.rglob("*.md"))[:10]:
|
|
94
|
+
parts.append(
|
|
95
|
+
f"\n--- {doc.relative_to(repo_path)} ---\n"
|
|
96
|
+
+ doc.read_text(encoding="utf-8", errors="ignore")
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
context = "\n\n".join(parts)
|
|
100
|
+
if len(context) > _MAX_CONTEXT_CHARS:
|
|
101
|
+
context = context[:_MAX_CONTEXT_CHARS] + "\n\n[... truncated ...]"
|
|
102
|
+
return context
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _citation_note_is_present(text: str, note: str) -> bool:
|
|
106
|
+
"""Require the full cited phrase, ignoring only case and whitespace."""
|
|
107
|
+
normalized_note = " ".join(note.casefold().split())
|
|
108
|
+
return bool(normalized_note) and normalized_note in " ".join(text.casefold().split())
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _git_log_contains(repo_path: Path, note: str) -> bool:
|
|
112
|
+
"""Verify a commit citation against Git when it has no source file."""
|
|
113
|
+
try:
|
|
114
|
+
result = subprocess.run(
|
|
115
|
+
["git", "log", "--all", "--oneline", "--format=%h %s"],
|
|
116
|
+
cwd=repo_path,
|
|
117
|
+
capture_output=True,
|
|
118
|
+
text=True,
|
|
119
|
+
encoding="utf-8",
|
|
120
|
+
errors="replace",
|
|
121
|
+
check=False,
|
|
122
|
+
)
|
|
123
|
+
except FileNotFoundError:
|
|
124
|
+
return False
|
|
125
|
+
return note.strip() in {line.strip() for line in result.stdout.splitlines()}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _repository_file(repo_path: Path, evidence_file: str | None) -> Path | None:
|
|
129
|
+
"""Resolve a citation only when it remains inside the analyzed repository."""
|
|
130
|
+
if not evidence_file:
|
|
131
|
+
return None
|
|
132
|
+
try:
|
|
133
|
+
target = (repo_path / evidence_file).resolve()
|
|
134
|
+
target.relative_to(repo_path.resolve())
|
|
135
|
+
except ValueError:
|
|
136
|
+
return None
|
|
137
|
+
return target
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _verify_claim(claim: Claim, repo_path: Path) -> Claim:
|
|
141
|
+
"""Re-check a "confirmed" claim's own citations against the actual file.
|
|
142
|
+
Downgrades to inferred (with a note) rather than trusting the model's
|
|
143
|
+
say-so - the honesty guarantee has to survive the LLM being wrong about
|
|
144
|
+
its own evidence, not just the mining rubric being right.
|
|
145
|
+
"""
|
|
146
|
+
if claim.confidence != Confidence.CONFIRMED:
|
|
147
|
+
return claim
|
|
148
|
+
|
|
149
|
+
for evidence in claim.cites:
|
|
150
|
+
if evidence.source == "user":
|
|
151
|
+
continue
|
|
152
|
+
if (
|
|
153
|
+
evidence.source == "code"
|
|
154
|
+
and evidence.file is None
|
|
155
|
+
and evidence.note == "Graphify structural extraction"
|
|
156
|
+
):
|
|
157
|
+
# The diagram-only fallback carries this exact computed Graphify
|
|
158
|
+
# summary. Every host-authored code citation must name a file.
|
|
159
|
+
continue
|
|
160
|
+
if evidence.source == "git_log" and not evidence.file:
|
|
161
|
+
if evidence.note and _git_log_contains(repo_path, evidence.note):
|
|
162
|
+
continue
|
|
163
|
+
else:
|
|
164
|
+
target = _repository_file(repo_path, evidence.file)
|
|
165
|
+
if target and target.exists() and evidence.note and _citation_note_is_present(
|
|
166
|
+
target.read_text(encoding="utf-8", errors="ignore"), evidence.note
|
|
167
|
+
):
|
|
168
|
+
continue
|
|
169
|
+
# Citation didn't check out - downgrade rather than ship a
|
|
170
|
+
# fabricated "confirmed" tag.
|
|
171
|
+
return claim.model_copy(
|
|
172
|
+
update={
|
|
173
|
+
"confidence": Confidence.INFERRED,
|
|
174
|
+
"text": f"{claim.text} (citation could not be independently verified)",
|
|
175
|
+
}
|
|
176
|
+
)
|
|
177
|
+
return claim
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def verify_understanding(understanding: GroundedUnderstanding, repo_path: Path) -> GroundedUnderstanding:
|
|
181
|
+
"""Verify every host- or provider-authored claim before delivery."""
|
|
182
|
+
pending = []
|
|
183
|
+
for section in understanding.doc:
|
|
184
|
+
for i, claim in enumerate(section.claims):
|
|
185
|
+
verified = _verify_claim(claim, repo_path)
|
|
186
|
+
section.claims[i] = verified
|
|
187
|
+
if verified.confidence == Confidence.INFERRED:
|
|
188
|
+
question = build_question(verified, f"{section.heading}::{i}", section.heading, guess=verified.text)
|
|
189
|
+
if question:
|
|
190
|
+
pending.append(question)
|
|
191
|
+
|
|
192
|
+
for tradeoff in section.tradeoffs:
|
|
193
|
+
tradeoff.decision = _verify_claim(tradeoff.decision, repo_path)
|
|
194
|
+
tradeoff.pros = [_verify_claim(c, repo_path) for c in tradeoff.pros]
|
|
195
|
+
tradeoff.cons = [_verify_claim(c, repo_path) for c in tradeoff.cons]
|
|
196
|
+
|
|
197
|
+
understanding.pending_questions = pending
|
|
198
|
+
return understanding
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def synthesize(
|
|
202
|
+
grounding: dict,
|
|
203
|
+
repo_path: Path,
|
|
204
|
+
backend: LLMBackend,
|
|
205
|
+
*,
|
|
206
|
+
diagram_kind: str = "architecture",
|
|
207
|
+
) -> GroundedUnderstanding:
|
|
208
|
+
context = build_context(grounding, repo_path)
|
|
209
|
+
draft = backend.draft(context, diagram_kind=diagram_kind)
|
|
210
|
+
|
|
211
|
+
if grounding.get("mode") == "graphify":
|
|
212
|
+
# Use Graphify's real, deterministic AST-extracted nodes/edges for
|
|
213
|
+
# the diagram rather than trusting whatever the LLM's draft
|
|
214
|
+
# invented - the model's summary is prone to dropping/renaming/
|
|
215
|
+
# inventing structure since nothing obliges it to reproduce the real
|
|
216
|
+
# graph faithfully (Naman's point, 12 Sep 2026: "in-depth coverage
|
|
217
|
+
# of graphify - what's connected to what" was the actual goal, not
|
|
218
|
+
# an LLM's approximation of it). The LLM's own nodes/edges draft is
|
|
219
|
+
# discarded entirely here; only its `doc` (rationale claims) survives.
|
|
220
|
+
draft["nodes"] = grounding.get("nodes", [])
|
|
221
|
+
draft["edges"] = grounding.get("edges", [])
|
|
222
|
+
draft["community_labels"] = grounding.get("community_labels", {})
|
|
223
|
+
|
|
224
|
+
understanding = GroundedUnderstanding.model_validate(draft)
|
|
225
|
+
|
|
226
|
+
if grounding.get("mode") == "graphify":
|
|
227
|
+
# The LLM's related_node_ids referenced ITS OWN invented node ids,
|
|
228
|
+
# which mean nothing now that real Graphify nodes/edges replaced the
|
|
229
|
+
# draft's - drop any that don't exist in the real graph rather than
|
|
230
|
+
# relying on the model having been told (and having obeyed) to cite
|
|
231
|
+
# real ids. A stale reference should disappear, not silently point
|
|
232
|
+
# at a node that no longer exists.
|
|
233
|
+
real_ids = {n["id"] for n in understanding.nodes if "id" in n}
|
|
234
|
+
for section in understanding.doc:
|
|
235
|
+
section.related_node_ids = [rid for rid in section.related_node_ids if rid in real_ids]
|
|
236
|
+
|
|
237
|
+
return verify_understanding(understanding, repo_path)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def synthesize_diagram_only(grounding: dict, *, diagram_kind: str = "architecture") -> GroundedUnderstanding:
|
|
241
|
+
"""No LLM call at all - the fallback when no backend key is configured.
|
|
242
|
+
|
|
243
|
+
The diagram itself needs zero LLM involvement in graphify mode:
|
|
244
|
+
nodes/edges/community_labels are already deterministic Graphify output,
|
|
245
|
+
untouched by the draft step even when an LLM IS available (see
|
|
246
|
+
synthesize() above - the model's own nodes/edges get discarded and
|
|
247
|
+
replaced with these same real ones). Only the narrative `doc` ever
|
|
248
|
+
actually required a model. Rather than failing outright and leaving the
|
|
249
|
+
user with nothing, hand back the real diagram data with one purely
|
|
250
|
+
factual, computed overview claim (node/edge/community counts - no
|
|
251
|
+
interpretation involved, so it's honestly "confirmed" with no
|
|
252
|
+
verification pass needed) - so `deliver` can still produce a full
|
|
253
|
+
interactive diagram (Naman, 12 Sep 2026: "the diagram is a must").
|
|
254
|
+
|
|
255
|
+
Only meaningful for graphify-mode grounding - `describe` mode has no
|
|
256
|
+
structural data to fall back on and genuinely needs the LLM to turn a
|
|
257
|
+
free-text description into anything at all; callers should keep failing
|
|
258
|
+
loudly in that case rather than call this with an empty result.
|
|
259
|
+
"""
|
|
260
|
+
nodes = grounding.get("nodes", [])
|
|
261
|
+
edges = grounding.get("edges", [])
|
|
262
|
+
communities = grounding.get("communities", {})
|
|
263
|
+
|
|
264
|
+
summary = f"{len(nodes)} structural nodes and {len(edges)} relationships were extracted by Graphify"
|
|
265
|
+
if communities:
|
|
266
|
+
summary += f", grouped into {len(communities)} communities"
|
|
267
|
+
summary += ". No LLM backend was available, so this doc is structure-only - no rationale was drafted."
|
|
268
|
+
|
|
269
|
+
overview_claim = Claim(
|
|
270
|
+
text=summary,
|
|
271
|
+
confidence=Confidence.CONFIRMED,
|
|
272
|
+
cites=[Evidence(source="code", note="Graphify structural extraction")],
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
return GroundedUnderstanding(
|
|
276
|
+
diagram_kind=diagram_kind,
|
|
277
|
+
nodes=nodes,
|
|
278
|
+
edges=edges,
|
|
279
|
+
doc=[DesignDocSection(heading="Overview", claims=[overview_claim])],
|
|
280
|
+
community_labels=grounding.get("community_labels", {}),
|
|
281
|
+
)
|
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
"""The one genuinely LLM-dependent piece of Synthesize: turning raw grounding
|
|
2
|
+
data (Graphify's nodes/edges, a plain description, or repo text) into
|
|
3
|
+
Claim/DesignDocSection objects in the first place.
|
|
4
|
+
|
|
5
|
+
Two concrete backends for standalone CLI use (`--anthropic` / `--gemini`,
|
|
6
|
+
plan.md §05's "what differs standalone" note) plus the seam a skill uses
|
|
7
|
+
instead: SkillBackend.draft() is never called in-process at all - the
|
|
8
|
+
SKILL.md instructs the host to do this reasoning itself and hand back JSON
|
|
9
|
+
matching the same schema, so graphitect's own code never needs to know it's
|
|
10
|
+
running inside an agent.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
from typing import Protocol
|
|
17
|
+
|
|
18
|
+
_CANONICAL_SECTIONS = [
|
|
19
|
+
"Overview",
|
|
20
|
+
"Components & responsibilities",
|
|
21
|
+
"Technology choices & why",
|
|
22
|
+
"Tradeoffs & alternatives considered",
|
|
23
|
+
"Key workflows",
|
|
24
|
+
"Limitations & future work",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
_SYSTEM_PROMPT = f"""\
|
|
28
|
+
You are graphitect's Synthesize step. Given grounding data about a
|
|
29
|
+
repository (a structural graph from Graphify, repo text: README, CHANGELOG,
|
|
30
|
+
docs, and excerpts of its most central source files), produce a
|
|
31
|
+
GroundedUnderstanding: a diagram structure plus a sourced, IN-DEPTH design
|
|
32
|
+
doc. A reader should come away understanding HOW the system actually works,
|
|
33
|
+
not just a list of facts about it - write substantive claims, not headlines.
|
|
34
|
+
For every claim, explain the mechanism, grounded in whatever real evidence
|
|
35
|
+
you have. "Uses a queue for background jobs" is a headline; "Background
|
|
36
|
+
jobs are pushed onto a Redis-backed queue (worker.py) and processed by a
|
|
37
|
+
separate worker process, decoupling slow operations like PDF generation
|
|
38
|
+
from the request/response cycle" is a substantive claim. Prefer fewer,
|
|
39
|
+
denser claims over many thin ones.
|
|
40
|
+
|
|
41
|
+
Write this as an interview-ready architecture explanation, not a dependency
|
|
42
|
+
inventory. For each applicable section, aim for 3-6 self-contained,
|
|
43
|
+
mechanism-level claims. In particular:
|
|
44
|
+
- In "Technology choices & why", every load-bearing choice should name the
|
|
45
|
+
technology, its concrete role in this repository, why that role fits the
|
|
46
|
+
system, and one operational consequence. Do not merely repeat a package
|
|
47
|
+
name.
|
|
48
|
+
- In "Tradeoffs & alternatives considered", analyze 3-5 significant choices
|
|
49
|
+
when the evidence supports them. Each analysis should name plausible
|
|
50
|
+
alternatives, at least two concrete benefits, and at least one cost,
|
|
51
|
+
constraint, or risk. Keep uncertainty explicit rather than inventing a
|
|
52
|
+
historical decision record.
|
|
53
|
+
- In "Key workflows", explain the actual sequence of handoffs, persistence,
|
|
54
|
+
background work, and user-visible result where those are evidenced. A
|
|
55
|
+
workflow claim should let a reader narrate the path without opening the
|
|
56
|
+
source tree.
|
|
57
|
+
- Cover security, data boundaries, operations, and failure behavior when
|
|
58
|
+
they are present in the evidence; do not force them into a repository that
|
|
59
|
+
does not implement them.
|
|
60
|
+
|
|
61
|
+
Use ONLY these section headings, spelled exactly as shown, and only the ones
|
|
62
|
+
that actually apply - do not invent your own heading, and do not rename or
|
|
63
|
+
paraphrase these:
|
|
64
|
+
{chr(10).join(f' - "{h}"' for h in _CANONICAL_SECTIONS)}
|
|
65
|
+
This matters beyond formatting: only claims placed in "Technology choices &
|
|
66
|
+
why" or "Tradeoffs & alternatives considered" ever get offered back to the
|
|
67
|
+
user as a follow-up question when they're inferred and unresolved. A
|
|
68
|
+
genuinely uncertain claim placed under an invented heading instead silently
|
|
69
|
+
skips that step - always use the closest matching canonical heading above
|
|
70
|
+
rather than inventing a more specific-sounding one.
|
|
71
|
+
|
|
72
|
+
"Components & responsibilities" and "Key workflows" should read as a real
|
|
73
|
+
walkthrough - what a component actually does and why it exists, or how a
|
|
74
|
+
workflow actually proceeds step by step - not just a names-only inventory.
|
|
75
|
+
Include a claim for every component/workflow that's genuinely worth
|
|
76
|
+
explaining, not only the ones with a citable "why" (that stricter load-
|
|
77
|
+
bearing bar below still applies to "Technology choices & why" and
|
|
78
|
+
"Tradeoffs & alternatives considered" specifically).
|
|
79
|
+
|
|
80
|
+
Follow the rationale-mining rubric for every claim you write:
|
|
81
|
+
1. If the grounding data or repo text directly states a fact or reason,
|
|
82
|
+
the claim is "confirmed" and MUST cite where (source: code/readme/
|
|
83
|
+
git_log/docs, with a file/note).
|
|
84
|
+
2. If you cannot find direct evidence but the claim is still worth making,
|
|
85
|
+
mark it "inferred" and say so plainly in the text - never phrase a guess
|
|
86
|
+
as if it were verified.
|
|
87
|
+
3. Classify every claim's kind:
|
|
88
|
+
- "descriptive": a fact about a past decision (why X was chosen, what
|
|
89
|
+
tradeoff was made). These are the ones a human could later confirm or
|
|
90
|
+
correct.
|
|
91
|
+
- "prescriptive": your own recommendation for what to do next. These are
|
|
92
|
+
never verifiable against the past, so always "inferred" and never
|
|
93
|
+
phrased as a fact.
|
|
94
|
+
4. Only put a claim in "Technology choices & why" or "Tradeoffs &
|
|
95
|
+
alternatives considered" if it's genuinely load-bearing - not every
|
|
96
|
+
detail needs a claim.
|
|
97
|
+
|
|
98
|
+
For each genuinely significant decision in "Technology choices & why" or
|
|
99
|
+
"Tradeoffs & alternatives considered", also add an entry to that section's
|
|
100
|
+
`tradeoffs` list - a real pros/cons comparison, not just the flat claim.
|
|
101
|
+
`decision` is itself a normal cited/confidence-tagged claim (it may restate
|
|
102
|
+
or elaborate a claim already in this section's own `claims`);
|
|
103
|
+
`alternatives_considered` names what else was plausible (say plainly if
|
|
104
|
+
this is your own inference rather than something the repo actually
|
|
105
|
+
discussed); `pros` and `cons` are themselves lists of normal cited/
|
|
106
|
+
confidence-tagged claims, never bare strings - do not invent a benefit or
|
|
107
|
+
drawback with no basis, mark it "inferred" like any other unverified claim.
|
|
108
|
+
Not every claim needs a tradeoffs entry - reserve it for choices substantial
|
|
109
|
+
enough to warrant a real comparison.
|
|
110
|
+
|
|
111
|
+
Never fabricate a citation. Never mark a claim "confirmed" without a
|
|
112
|
+
specific source you can point to. When genuinely unsure whether something
|
|
113
|
+
is confirmed or inferred, choose inferred.
|
|
114
|
+
|
|
115
|
+
If a structure graph (from Graphify) is included in the context below, your
|
|
116
|
+
own "nodes" and "edges" in the response are discarded and replaced with the
|
|
117
|
+
real graph - don't spend effort inventing a diagram structure in that case.
|
|
118
|
+
Instead, set each doc section's related_node_ids to actual node ids quoted
|
|
119
|
+
from that structure graph (not names you invent) - anything that isn't a
|
|
120
|
+
real id from the graph is silently dropped rather than shown as a broken
|
|
121
|
+
reference. Code excerpts in the context (when present) are real file
|
|
122
|
+
contents from the repository's most central files - use them as your
|
|
123
|
+
primary source for HOW claims, citing them as source "code" with the file
|
|
124
|
+
path shown in the excerpt's header.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
_DRAFT_TOOL_NAME = "emit_grounded_understanding"
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class LLMBackend(Protocol):
|
|
131
|
+
def draft(self, context: str, *, diagram_kind: str) -> dict:
|
|
132
|
+
"""Return a dict matching GroundedUnderstanding's schema (not yet
|
|
133
|
+
validated - callers run it through GroundedUnderstanding.model_validate).
|
|
134
|
+
"""
|
|
135
|
+
...
|
|
136
|
+
|
|
137
|
+
def complete_json(self, system_prompt: str, user_prompt: str, schema: dict, *, tool_name: str) -> dict:
|
|
138
|
+
"""General structured-output call: force the model to return a dict
|
|
139
|
+
matching `schema`. `draft()` is just this with a fixed prompt/schema;
|
|
140
|
+
archify_repair's layout-fix loop is the other caller, with a
|
|
141
|
+
different prompt/schema (ArchitectureIR, not GroundedUnderstanding) -
|
|
142
|
+
shared here rather than duplicated per backend.
|
|
143
|
+
"""
|
|
144
|
+
...
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _draft_tool_schema() -> dict:
|
|
148
|
+
from ..models import GroundedUnderstanding
|
|
149
|
+
|
|
150
|
+
schema = GroundedUnderstanding.model_json_schema()
|
|
151
|
+
return {
|
|
152
|
+
"name": _DRAFT_TOOL_NAME,
|
|
153
|
+
"description": "Emit the drafted GroundedUnderstanding for this repository.",
|
|
154
|
+
"input_schema": schema,
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class AnthropicBackend:
|
|
159
|
+
"""Standalone backend backed by the caller's own ANTHROPIC_API_KEY.
|
|
160
|
+
The `anthropic` client ships with Graphitect.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
def __init__(self, api_key: str, model: str = "claude-sonnet-5"):
|
|
164
|
+
try:
|
|
165
|
+
import anthropic
|
|
166
|
+
except ImportError as exc:
|
|
167
|
+
raise ImportError(
|
|
168
|
+
"AnthropicBackend needs the bundled `anthropic` package; "
|
|
169
|
+
"reinstall Graphitect with `pip install --force-reinstall graphitect`."
|
|
170
|
+
) from exc
|
|
171
|
+
self._client = anthropic.Anthropic(api_key=api_key)
|
|
172
|
+
self._model = model
|
|
173
|
+
|
|
174
|
+
def draft(self, context: str, *, diagram_kind: str) -> dict:
|
|
175
|
+
schema = _draft_tool_schema()["input_schema"]
|
|
176
|
+
user = f"diagram_kind: {diagram_kind}\n\nGrounding data and repo text:\n\n{context}"
|
|
177
|
+
return self.complete_json(_SYSTEM_PROMPT, user, schema, tool_name=_DRAFT_TOOL_NAME)
|
|
178
|
+
|
|
179
|
+
def complete_json(self, system_prompt: str, user_prompt: str, schema: dict, *, tool_name: str) -> dict:
|
|
180
|
+
tool = {"name": tool_name, "description": f"Emit {tool_name}.", "input_schema": schema}
|
|
181
|
+
response = self._client.messages.create(
|
|
182
|
+
model=self._model,
|
|
183
|
+
max_tokens=8192,
|
|
184
|
+
system=system_prompt,
|
|
185
|
+
tools=[tool],
|
|
186
|
+
tool_choice={"type": "tool", "name": tool_name},
|
|
187
|
+
messages=[{"role": "user", "content": user_prompt}],
|
|
188
|
+
)
|
|
189
|
+
for block in response.content:
|
|
190
|
+
if block.type == "tool_use" and block.name == tool_name:
|
|
191
|
+
return block.input
|
|
192
|
+
raise RuntimeError(f"Model did not call the expected tool {tool_name!r} - no output produced.")
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
class GeminiBackend:
|
|
196
|
+
"""Standalone backend backed by the caller's own GEMINI_API_KEY /
|
|
197
|
+
GOOGLE_API_KEY, via Gemini's OpenAI-compatible endpoint (matching how
|
|
198
|
+
Graphify's own gemini extra is structured: openai + tiktoken).
|
|
199
|
+
"""
|
|
200
|
+
|
|
201
|
+
def __init__(self, api_key: str, model: str = "gemini-3-flash-preview"):
|
|
202
|
+
try:
|
|
203
|
+
from openai import OpenAI
|
|
204
|
+
except ImportError as exc:
|
|
205
|
+
raise ImportError(
|
|
206
|
+
"GeminiBackend needs the bundled `openai` package; "
|
|
207
|
+
"reinstall Graphitect with `pip install --force-reinstall graphitect`."
|
|
208
|
+
) from exc
|
|
209
|
+
self._client = OpenAI(
|
|
210
|
+
api_key=api_key, base_url="https://generativelanguage.googleapis.com/v1beta/openai/"
|
|
211
|
+
)
|
|
212
|
+
self._model = model
|
|
213
|
+
|
|
214
|
+
def draft(self, context: str, *, diagram_kind: str) -> dict:
|
|
215
|
+
schema = _draft_tool_schema()["input_schema"]
|
|
216
|
+
user = f"diagram_kind: {diagram_kind}\n\nGrounding data and repo text:\n\n{context}"
|
|
217
|
+
return self.complete_json(_SYSTEM_PROMPT, user, schema, tool_name=_DRAFT_TOOL_NAME)
|
|
218
|
+
|
|
219
|
+
def complete_json(self, system_prompt: str, user_prompt: str, schema: dict, *, tool_name: str) -> dict:
|
|
220
|
+
response = self._client.chat.completions.create(
|
|
221
|
+
model=self._model,
|
|
222
|
+
response_format={"type": "json_schema", "json_schema": {"name": tool_name, "schema": schema}},
|
|
223
|
+
messages=[
|
|
224
|
+
{"role": "system", "content": system_prompt},
|
|
225
|
+
{"role": "user", "content": user_prompt},
|
|
226
|
+
],
|
|
227
|
+
)
|
|
228
|
+
return json.loads(response.choices[0].message.content)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
class OllamaBackend:
|
|
232
|
+
"""Local Ollama (no key needed, `http://localhost:11434`) or Ollama
|
|
233
|
+
Cloud (an API key, generous free tier) - same OpenAI-compatible wire
|
|
234
|
+
protocol either way, so one class covers both; only the default
|
|
235
|
+
base_url changes depending on whether a key was given. Mirrors
|
|
236
|
+
git-resume-agent's own confirmed Ollama-first LLM strategy (this
|
|
237
|
+
project's own Phase 0 doc, §02 "Ollama-first" row).
|
|
238
|
+
|
|
239
|
+
The bundled `openai` package is used for Ollama's compatibility layer;
|
|
240
|
+
it does not require an OpenAI account.
|
|
241
|
+
"""
|
|
242
|
+
|
|
243
|
+
# Confirmed live against https://ollama.com/v1/models (12 Sep 2026) - Cloud's
|
|
244
|
+
# catalog is a completely different namespace from local model names (no
|
|
245
|
+
# "llama3.1" on Cloud at all). A model name valid on one is very likely
|
|
246
|
+
# invalid on the other, and Cloud returns a bare 401 for an inaccessible/
|
|
247
|
+
# unrecognized model rather than 404 - which looks exactly like a bad key
|
|
248
|
+
# until you check /v1/models directly with the same key and see it's fine.
|
|
249
|
+
_CLOUD_DEFAULT_MODEL = "gpt-oss:20b"
|
|
250
|
+
_LOCAL_DEFAULT_MODEL = "llama3.1"
|
|
251
|
+
|
|
252
|
+
def __init__(
|
|
253
|
+
self,
|
|
254
|
+
model: str | None = None,
|
|
255
|
+
*,
|
|
256
|
+
api_key: str | None = None,
|
|
257
|
+
base_url: str | None = None,
|
|
258
|
+
):
|
|
259
|
+
try:
|
|
260
|
+
from openai import OpenAI
|
|
261
|
+
except ImportError as exc:
|
|
262
|
+
raise ImportError(
|
|
263
|
+
"OllamaBackend needs the bundled `openai` package; reinstall "
|
|
264
|
+
"Graphitect with `pip install --force-reinstall graphitect`."
|
|
265
|
+
) from exc
|
|
266
|
+
is_cloud = bool(api_key)
|
|
267
|
+
resolved_base = base_url or ("https://ollama.com/v1" if is_cloud else "http://localhost:11434/v1")
|
|
268
|
+
# Local Ollama ignores the key entirely but the OpenAI client requires
|
|
269
|
+
# a non-empty string; Cloud actually checks it.
|
|
270
|
+
self._client = OpenAI(api_key=api_key or "ollama-local", base_url=resolved_base)
|
|
271
|
+
self._model = model or (self._CLOUD_DEFAULT_MODEL if is_cloud else self._LOCAL_DEFAULT_MODEL)
|
|
272
|
+
|
|
273
|
+
def draft(self, context: str, *, diagram_kind: str) -> dict:
|
|
274
|
+
schema = _draft_tool_schema()["input_schema"]
|
|
275
|
+
user = f"diagram_kind: {diagram_kind}\n\nGrounding data and repo text:\n\n{context}"
|
|
276
|
+
return self.complete_json(_SYSTEM_PROMPT, user, schema, tool_name=_DRAFT_TOOL_NAME)
|
|
277
|
+
|
|
278
|
+
def complete_json(self, system_prompt: str, user_prompt: str, schema: dict, *, tool_name: str) -> dict:
|
|
279
|
+
# Unlike Gemini's stricter json_schema mode, Ollama's OpenAI-compat
|
|
280
|
+
# layer support for response_format varies by model, so the schema
|
|
281
|
+
# is embedded directly in the prompt as the more portable path -
|
|
282
|
+
# every model can at least follow instructions in plain text even
|
|
283
|
+
# if it can't honor a strict schema constraint.
|
|
284
|
+
system_with_schema = (
|
|
285
|
+
f"{system_prompt}\n\nRespond with ONLY a JSON object matching this schema "
|
|
286
|
+
f"exactly - no markdown fences, no explanation before or after:\n{json.dumps(schema)}"
|
|
287
|
+
)
|
|
288
|
+
response = self._client.chat.completions.create(
|
|
289
|
+
model=self._model,
|
|
290
|
+
response_format={"type": "json_object"},
|
|
291
|
+
messages=[
|
|
292
|
+
{"role": "system", "content": system_with_schema},
|
|
293
|
+
{"role": "user", "content": user_prompt},
|
|
294
|
+
],
|
|
295
|
+
)
|
|
296
|
+
return json.loads(response.choices[0].message.content)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def resolve_backend(
|
|
300
|
+
*,
|
|
301
|
+
anthropic_key: str | None,
|
|
302
|
+
gemini_key: str | None,
|
|
303
|
+
ollama_key: str | None = None,
|
|
304
|
+
ollama_base_url: str | None = None,
|
|
305
|
+
ollama_model: str | None = None,
|
|
306
|
+
ollama_local: bool = False,
|
|
307
|
+
) -> LLMBackend:
|
|
308
|
+
"""Standalone CLI resolution order: Ollama (key, or `--ollama-local` for
|
|
309
|
+
a locally-running server with no key at all) first - free tier, no
|
|
310
|
+
account needed for local use - then Anthropic, then Gemini, then fail
|
|
311
|
+
loudly. Matches Graphify's own "no other keys are read" honesty rule:
|
|
312
|
+
never silently fall back to a key or endpoint the user didn't ask for.
|
|
313
|
+
"""
|
|
314
|
+
if ollama_key or ollama_local:
|
|
315
|
+
return OllamaBackend(
|
|
316
|
+
# Preserve None so OllamaBackend can select the correct default
|
|
317
|
+
# for Cloud (gpt-oss:20b) versus a local server (llama3.1).
|
|
318
|
+
model=ollama_model,
|
|
319
|
+
api_key=ollama_key,
|
|
320
|
+
base_url=ollama_base_url,
|
|
321
|
+
)
|
|
322
|
+
if anthropic_key:
|
|
323
|
+
return AnthropicBackend(anthropic_key)
|
|
324
|
+
if gemini_key:
|
|
325
|
+
return GeminiBackend(gemini_key)
|
|
326
|
+
raise RuntimeError(
|
|
327
|
+
"No LLM backend available. Set OLLAMA_API_KEY (Ollama Cloud), pass "
|
|
328
|
+
"--ollama-local for a locally-running Ollama server, or set ANTHROPIC_API_KEY / "
|
|
329
|
+
"GEMINI_API_KEY - or run graphitect as a Claude Code/Cursor/Codex skill instead, "
|
|
330
|
+
"where the host's own reasoning does this step for free (plan.md §05)."
|
|
331
|
+
)
|