graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/cache.py
ADDED
|
@@ -0,0 +1,1746 @@
|
|
|
1
|
+
# per-file extraction cache - skip unchanged files on re-run
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import atexit
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
import tempfile
|
|
10
|
+
import time
|
|
11
|
+
import warnings
|
|
12
|
+
from collections.abc import Callable, Iterable
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
# Output directory name — override with GRAPHIFY_OUT env var for worktrees or
|
|
16
|
+
# shared-output setups. Accepts a relative name ("graphify-out-feature") or an
|
|
17
|
+
# absolute path ("/shared/graphify-out"). Single source of truth in graphify.paths
|
|
18
|
+
# (#1423); re-exported here as _GRAPHIFY_OUT for the existing call sites.
|
|
19
|
+
from graphify.paths import GRAPHIFY_OUT as _GRAPHIFY_OUT
|
|
20
|
+
|
|
21
|
+
# AST cache entries are the output of graphify's own extractor code, so they
|
|
22
|
+
# are only valid for the version that wrote them: keying purely on file
|
|
23
|
+
# content means extractor fixes shipped in a new release keep serving stale
|
|
24
|
+
# pre-fix results. The AST cache is therefore namespaced by package version
|
|
25
|
+
# and cache-key schema (cache/ast/v{version}-s{schema}/), with entries from
|
|
26
|
+
# other versions or schemas removed on first
|
|
27
|
+
# use. The semantic cache is deliberately NOT versioned — its entries are
|
|
28
|
+
# produced by the LLM from file contents, and invalidating them on every
|
|
29
|
+
# release would re-bill extraction for unchanged files.
|
|
30
|
+
try:
|
|
31
|
+
from importlib.metadata import version as _pkg_version
|
|
32
|
+
|
|
33
|
+
_EXTRACTOR_VERSION = _pkg_version("graphifyy")
|
|
34
|
+
except Exception:
|
|
35
|
+
_EXTRACTOR_VERSION = "0.9.58-bundled"
|
|
36
|
+
|
|
37
|
+
# Bump when AST cache-key semantics change independently of the package version.
|
|
38
|
+
_AST_CACHE_SCHEMA = 2
|
|
39
|
+
|
|
40
|
+
# Version dirs already swept this process — cleanup runs once per (base, version).
|
|
41
|
+
_cleaned_ast_dirs: set[str] = set()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _cleanup_stale_ast_entries(ast_base: Path, current_dir: Path) -> None:
|
|
45
|
+
"""Remove AST cache entries left behind by other graphify versions.
|
|
46
|
+
|
|
47
|
+
Sweeps sibling ``v*/`` directories and unversioned ``*.json`` entries
|
|
48
|
+
(the pre-versioning layout) under ``cache/ast/``. Best-effort: failures
|
|
49
|
+
are ignored, stragglers are retried on the next run.
|
|
50
|
+
"""
|
|
51
|
+
key = str(current_dir)
|
|
52
|
+
if key in _cleaned_ast_dirs:
|
|
53
|
+
return
|
|
54
|
+
_cleaned_ast_dirs.add(key)
|
|
55
|
+
if not ast_base.is_dir():
|
|
56
|
+
return
|
|
57
|
+
import shutil
|
|
58
|
+
|
|
59
|
+
for child in ast_base.iterdir():
|
|
60
|
+
if child == current_dir:
|
|
61
|
+
continue
|
|
62
|
+
try:
|
|
63
|
+
if child.is_dir() and child.name.startswith("v"):
|
|
64
|
+
shutil.rmtree(child, ignore_errors=True)
|
|
65
|
+
elif child.suffix == ".json":
|
|
66
|
+
child.unlink()
|
|
67
|
+
except OSError:
|
|
68
|
+
pass
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# Semantic cache entries are LLM output, so they depend on the extraction prompt
|
|
72
|
+
# that produced them, not just on file contents. Keying purely on content means a
|
|
73
|
+
# release that changes the prompt keeps replaying entries from the older prompt on
|
|
74
|
+
# every unchanged file, silently mixing extraction vintages in one graph (#1939).
|
|
75
|
+
# Versioning them by package version (as the AST cache does) would re-bill LLM
|
|
76
|
+
# extraction on every patch release — the reason #1252 deliberately left them
|
|
77
|
+
# unversioned. Fingerprinting the prompt itself keeps both properties: entries
|
|
78
|
+
# survive releases that don't touch the prompt, and invalidate only when it
|
|
79
|
+
# actually changed. Entries live under cache/semantic/p{fingerprint}/ when the
|
|
80
|
+
# caller supplies its prompt; callers that don't keep the historical flat layout.
|
|
81
|
+
_PROMPT_FP_LEN = 12
|
|
82
|
+
|
|
83
|
+
# Count of pre-fingerprint (flat-layout) entries served this process, so
|
|
84
|
+
# check_semantic_cache can report N to the user (#1939).
|
|
85
|
+
_legacy_semantic_hits = 0
|
|
86
|
+
|
|
87
|
+
# Count of cache entries that failed to parse as JSON this process. A corrupt
|
|
88
|
+
# entry is not a miss: left in place it fails on every future run, silently
|
|
89
|
+
# re-extracting (and, for semantic kinds, re-billing) the file forever. The
|
|
90
|
+
# counter lets check_semantic_cache surface one aggregate warning (#2405).
|
|
91
|
+
_corrupt_cache_entries = 0
|
|
92
|
+
|
|
93
|
+
# Prompt-file fingerprints already computed, keyed by (path, size, mtime_ns) —
|
|
94
|
+
# the same stat signature the hash index uses. check_semantic_cache resolves the
|
|
95
|
+
# prompt once per FILE in the corpus, so without this a 500-doc run re-reads and
|
|
96
|
+
# re-hashes the same spec 500 times (and warns 500 times when it is unreadable).
|
|
97
|
+
_prompt_fp_cache: dict[tuple, str] = {}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def prompt_fingerprint(prompt: "str | Path") -> str:
|
|
101
|
+
"""Return a short stable fingerprint of an extraction prompt.
|
|
102
|
+
|
|
103
|
+
``prompt`` is either the prompt text itself (the Python extraction path owns
|
|
104
|
+
its system prompt, :func:`graphify.llm._extraction_system`) or a Path to the
|
|
105
|
+
prompt file an agent loaded (the skill path's
|
|
106
|
+
``references/extraction-spec.md``).
|
|
107
|
+
|
|
108
|
+
Line endings and trailing whitespace are normalized before hashing: the same
|
|
109
|
+
spec file checked out with CRLF on Windows must not fingerprint differently
|
|
110
|
+
from the LF checkout that wrote the cache, or every Windows run would look
|
|
111
|
+
like a prompt change and re-bill extraction.
|
|
112
|
+
"""
|
|
113
|
+
if isinstance(prompt, Path):
|
|
114
|
+
text = prompt.read_text(encoding="utf-8", errors="replace")
|
|
115
|
+
else:
|
|
116
|
+
text = prompt
|
|
117
|
+
normalized = "\n".join(
|
|
118
|
+
line.rstrip() for line in text.replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
|
119
|
+
).strip()
|
|
120
|
+
return hashlib.sha256(normalized.encode()).hexdigest()[:_PROMPT_FP_LEN]
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _resolve_prompt_fp(prompt: "str | Path | None" = None,
|
|
124
|
+
prompt_file: "str | Path | None" = None) -> str | None:
|
|
125
|
+
"""Fingerprint the caller's extraction prompt, or None when it supplied none.
|
|
126
|
+
|
|
127
|
+
``prompt`` is prompt TEXT; ``prompt_file`` is a path to a file CONTAINING the
|
|
128
|
+
prompt. They are separate parameters rather than one overloaded argument
|
|
129
|
+
because the skill-driven callers are markdown snippets an agent copies with a
|
|
130
|
+
path substituted in — passing that path as ``prompt`` would hash the path
|
|
131
|
+
string itself, yielding a fingerprint that is stable, plausible, and tracks
|
|
132
|
+
nothing about the prompt. A silent wrong fingerprint is the exact failure
|
|
133
|
+
class #1939 is about, so the two are not inferred from each other.
|
|
134
|
+
|
|
135
|
+
Best-effort: an unreadable ``prompt_file`` falls back to the flat, unattributed
|
|
136
|
+
layout rather than failing the run — a cache is never worth aborting an
|
|
137
|
+
extraction over. It warns rather than falling back quietly, because that
|
|
138
|
+
fallback silently restores the very behavior this fixes, and the skill-side
|
|
139
|
+
caller substitutes this path by hand.
|
|
140
|
+
"""
|
|
141
|
+
memo_key = None
|
|
142
|
+
if prompt_file is not None:
|
|
143
|
+
prompt = Path(prompt_file)
|
|
144
|
+
try:
|
|
145
|
+
st = prompt.stat()
|
|
146
|
+
memo_key = (str(prompt), st.st_size, st.st_mtime_ns)
|
|
147
|
+
if memo_key in _prompt_fp_cache:
|
|
148
|
+
return _prompt_fp_cache[memo_key]
|
|
149
|
+
except OSError:
|
|
150
|
+
pass # unreadable — fall through to the warning below
|
|
151
|
+
if prompt is None:
|
|
152
|
+
return None
|
|
153
|
+
try:
|
|
154
|
+
fp = prompt_fingerprint(prompt)
|
|
155
|
+
if memo_key is not None:
|
|
156
|
+
_prompt_fp_cache[memo_key] = fp
|
|
157
|
+
return fp
|
|
158
|
+
except (OSError, UnicodeError) as exc:
|
|
159
|
+
warnings.warn(
|
|
160
|
+
f"could not read extraction prompt {str(prompt)!r} ({exc}); semantic cache "
|
|
161
|
+
"entries cannot be attributed to a prompt version and fall back to the "
|
|
162
|
+
"unversioned layout, so this run may replay entries from an older "
|
|
163
|
+
"extraction prompt (#1939).",
|
|
164
|
+
RuntimeWarning,
|
|
165
|
+
stacklevel=3,
|
|
166
|
+
)
|
|
167
|
+
return None
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
# A frontmatter delimiter is a whole line of exactly three dashes (optional
|
|
171
|
+
# trailing whitespace). Substring checks like startswith("---") /
|
|
172
|
+
# find("\n---") also match `----` thematic breaks and `--- text` prose,
|
|
173
|
+
# silently dropping everything above them from the hash (#1259).
|
|
174
|
+
_FRONTMATTER_DELIM = re.compile(r"^---[ \t]*\r?$", re.MULTILINE)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _body_content(content: bytes) -> bytes:
|
|
178
|
+
"""Strip YAML frontmatter from Markdown content, returning only the body."""
|
|
179
|
+
text = content.decode(errors="replace")
|
|
180
|
+
opener = _FRONTMATTER_DELIM.match(text)
|
|
181
|
+
if opener is None:
|
|
182
|
+
return content
|
|
183
|
+
closer = _FRONTMATTER_DELIM.search(text, opener.end())
|
|
184
|
+
if closer is None:
|
|
185
|
+
return content
|
|
186
|
+
# Slice right after the closing `---` (not after its line) so the output
|
|
187
|
+
# stays byte-identical with the historical implementation for well-formed
|
|
188
|
+
# frontmatter -- existing semantic-cache hashes must not churn.
|
|
189
|
+
return text[closer.start() + 3:].encode()
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# Stat-based index: maps absolute path → {size, mtime_ns, indexed_at_ns, ...}.
|
|
193
|
+
# Loaded once per process, flushed via atexit. Skips full file reads when
|
|
194
|
+
# size+mtime_ns are unchanged — same trade-off as make(1).
|
|
195
|
+
# Correctness risks: `touch` causes a harmless extra re-hash. Same-size edits
|
|
196
|
+
# inside one mtime tick used to return the PREVIOUS content's digest; the
|
|
197
|
+
# racily-clean guard below closes that hole (see _stat_sig_fresh).
|
|
198
|
+
# `graphify extract --force` / `graphify update --force` (or GRAPHIFY_FORCE=1)
|
|
199
|
+
# skip the cache reads and re-dispatch everything when needed (#1894).
|
|
200
|
+
_stat_index: dict[str, dict] = {}
|
|
201
|
+
_stat_index_root: Path | None = None
|
|
202
|
+
# Key anchor for the ON-DISK index (#2199): the first caller's key-root, i.e.
|
|
203
|
+
# the corpus. Distinct from _stat_index_root, which is the cache-FILE location
|
|
204
|
+
# (cache_root, #1774) — the two differ under --out and must not be conflated.
|
|
205
|
+
_stat_index_anchor: Path | None = None
|
|
206
|
+
_stat_index_dirty: bool = False
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
# Filesystem mtime granularity, in nanoseconds. A stat signature only proves a
|
|
210
|
+
# file is unchanged when the clock that stamped its mtime is finer-grained than
|
|
211
|
+
# the interval between two writes — which is false almost everywhere: NTFS
|
|
212
|
+
# advances mtime on the ~15.6 ms system tick, FAT/exFAT on 2 s, and Linux
|
|
213
|
+
# stamps from the coarse (jiffies) clock even though ext4 stores nanoseconds.
|
|
214
|
+
# 2 s is the conservative default that covers all of them. It costs nothing in
|
|
215
|
+
# practice: only files modified within the last 2 s lose the fastpath, and in a
|
|
216
|
+
# real corpus those are exactly the handful of files that changed and have to be
|
|
217
|
+
# read anyway. Override with GRAPHIFY_MTIME_GRANULARITY_MS (0 disables the
|
|
218
|
+
# guard and restores the pre-fix behaviour).
|
|
219
|
+
_MTIME_GRANULARITY_NS = 2_000_000_000
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _mtime_granularity_ns() -> int:
|
|
223
|
+
"""Return the assumed filesystem mtime granularity in nanoseconds.
|
|
224
|
+
|
|
225
|
+
Read fresh on every call so the env var can be set after import (and so
|
|
226
|
+
tests can flip it without reloading the module).
|
|
227
|
+
"""
|
|
228
|
+
raw = os.environ.get("GRAPHIFY_MTIME_GRANULARITY_MS", "").strip()
|
|
229
|
+
if raw:
|
|
230
|
+
try:
|
|
231
|
+
ms = float(raw)
|
|
232
|
+
except ValueError:
|
|
233
|
+
return _MTIME_GRANULARITY_NS
|
|
234
|
+
if ms >= 0:
|
|
235
|
+
return int(ms * 1_000_000)
|
|
236
|
+
return _MTIME_GRANULARITY_NS
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _stat_sig_fresh(entry: object, st: "os.stat_result") -> bool:
|
|
240
|
+
"""True if ``entry`` provably describes the file's CURRENT content.
|
|
241
|
+
|
|
242
|
+
Beyond matching (size, mtime_ns), the entry must be *racily clean* in git's
|
|
243
|
+
sense: we must have read the content strictly after the file's mtime tick
|
|
244
|
+
had already closed. Otherwise a write that landed between our read and the
|
|
245
|
+
end of that tick would have left mtime (and, for a same-length edit, size)
|
|
246
|
+
untouched, and the stored digest would describe content that is no longer
|
|
247
|
+
on disk.
|
|
248
|
+
|
|
249
|
+
``indexed_at_ns`` is the wall clock captured immediately BEFORE the content
|
|
250
|
+
was read. Requiring ``mtime + granularity <= indexed_at`` means any later
|
|
251
|
+
write necessarily lands in a new tick and so changes mtime, making it
|
|
252
|
+
visible to the next signature comparison.
|
|
253
|
+
|
|
254
|
+
Entries written by an older graphify carry no ``indexed_at_ns``; they are
|
|
255
|
+
treated as untrusted (one re-read each), and gain the field when rewritten.
|
|
256
|
+
"""
|
|
257
|
+
if not isinstance(entry, dict):
|
|
258
|
+
return False
|
|
259
|
+
if entry.get("size") != st.st_size or entry.get("mtime_ns") != st.st_mtime_ns:
|
|
260
|
+
return False
|
|
261
|
+
indexed_at = entry.get("indexed_at_ns")
|
|
262
|
+
if not isinstance(indexed_at, int):
|
|
263
|
+
return False
|
|
264
|
+
return st.st_mtime_ns + _mtime_granularity_ns() <= indexed_at
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _stat_entry_for(abs_key: str, st: "os.stat_result", observed_at_ns: int) -> dict:
|
|
268
|
+
"""Get-or-reset the index entry for ``abs_key`` and stamp when it was read.
|
|
269
|
+
|
|
270
|
+
Reuses the existing dict when the stat signature still matches, so
|
|
271
|
+
co-located values (other salts' digests, ``word_count``) survive; resets it
|
|
272
|
+
otherwise, so a stale ``word_count`` cannot outlive the content it counted.
|
|
273
|
+
|
|
274
|
+
``observed_at_ns`` must be the clock reading taken *before* the content was
|
|
275
|
+
read — see :func:`_stat_sig_fresh` for why the ordering matters.
|
|
276
|
+
"""
|
|
277
|
+
entry = _stat_index.get(abs_key)
|
|
278
|
+
if (not isinstance(entry, dict)
|
|
279
|
+
or entry.get("size") != st.st_size
|
|
280
|
+
or entry.get("mtime_ns") != st.st_mtime_ns):
|
|
281
|
+
entry = {"size": st.st_size, "mtime_ns": st.st_mtime_ns}
|
|
282
|
+
_stat_index[abs_key] = entry
|
|
283
|
+
entry["indexed_at_ns"] = observed_at_ns
|
|
284
|
+
return entry
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _stat_key_to_relative(key: str, anchor: Path) -> str:
|
|
288
|
+
"""Return ``key`` as a forward-slash relative path from ``anchor``.
|
|
289
|
+
|
|
290
|
+
Local duplicate of :func:`graphify.detect._to_relative_for_storage` —
|
|
291
|
+
detect imports cache, so cache cannot import detect without a cycle
|
|
292
|
+
(and pulling detect in during the atexit flush would be fragile).
|
|
293
|
+
Out-of-anchor and already-relative keys pass through unchanged, and
|
|
294
|
+
``..``-escaping relpaths are rejected (kept absolute), mirroring the
|
|
295
|
+
manifest's portability rules.
|
|
296
|
+
"""
|
|
297
|
+
p = Path(key)
|
|
298
|
+
if not p.is_absolute():
|
|
299
|
+
return key
|
|
300
|
+
try:
|
|
301
|
+
rel = os.path.relpath(p, anchor)
|
|
302
|
+
except (ValueError, OSError):
|
|
303
|
+
return key # outside anchor (e.g. Windows cross-drive)
|
|
304
|
+
if rel == ".." or rel.startswith(".." + os.sep) or rel.startswith("../"):
|
|
305
|
+
return key # escaped anchor — keep absolute
|
|
306
|
+
return rel.replace(os.sep, "/")
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _stat_key_to_absolute(key: str, anchor: Path) -> str:
|
|
310
|
+
"""Inverse of :func:`_stat_key_to_relative`.
|
|
311
|
+
|
|
312
|
+
Re-anchor a stored relative key against ``anchor``. Already-absolute keys
|
|
313
|
+
(legacy indexes, out-of-anchor entries) pass through unchanged so an index
|
|
314
|
+
written by an older graphify remains readable.
|
|
315
|
+
"""
|
|
316
|
+
p = Path(key)
|
|
317
|
+
if p.is_absolute():
|
|
318
|
+
return str(p)
|
|
319
|
+
return str(anchor / p)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _stat_index_file(root: Path) -> Path:
|
|
323
|
+
_out = Path(_GRAPHIFY_OUT)
|
|
324
|
+
base = _out if _out.is_absolute() else Path(root).resolve() / _out
|
|
325
|
+
return base / "cache" / "stat-index.json"
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def _ensure_stat_index(root: Path, cache_root: "Path | None" = None) -> None:
|
|
329
|
+
global _stat_index, _stat_index_root, _stat_index_anchor, _stat_index_dirty
|
|
330
|
+
if _stat_index_root is not None:
|
|
331
|
+
return
|
|
332
|
+
# _stat_index_root determines the cache FILE location, so honoring an
|
|
333
|
+
# explicit cache_root keeps detect()'s word-count cache under the requested
|
|
334
|
+
# --out dir instead of polluting the scanned corpus with a stray
|
|
335
|
+
# graphify-out/ (#1747). _stat_index_anchor is the separate KEY anchor:
|
|
336
|
+
# in-memory keys stay absolute, but the on-disk index stores in-anchor keys
|
|
337
|
+
# relative so a moved/cloned corpus still hits (#2199) — same load/save
|
|
338
|
+
# re-anchoring the detect manifest uses.
|
|
339
|
+
_stat_index_root = Path(cache_root if cache_root is not None else root).resolve()
|
|
340
|
+
_stat_index_anchor = Path(root).resolve()
|
|
341
|
+
p = _stat_index_file(_stat_index_root)
|
|
342
|
+
_stat_index = {}
|
|
343
|
+
if p.exists():
|
|
344
|
+
try:
|
|
345
|
+
raw = json.loads(p.read_text(encoding="utf-8"))
|
|
346
|
+
if isinstance(raw, dict):
|
|
347
|
+
for k, v in raw.items():
|
|
348
|
+
if not isinstance(k, str):
|
|
349
|
+
continue
|
|
350
|
+
if Path(k).is_absolute():
|
|
351
|
+
# Legacy/out-of-anchor key: pass through, but never
|
|
352
|
+
# clobber a re-anchored relative (new-format) entry
|
|
353
|
+
# that resolved to the same absolute path.
|
|
354
|
+
_stat_index.setdefault(k, v)
|
|
355
|
+
else:
|
|
356
|
+
_stat_index[_stat_key_to_absolute(k, _stat_index_anchor)] = v
|
|
357
|
+
except (json.JSONDecodeError, OSError):
|
|
358
|
+
_stat_index = {}
|
|
359
|
+
atexit.register(_flush_stat_index)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _flush_stat_index() -> None:
|
|
363
|
+
global _stat_index_dirty, _stat_index_root
|
|
364
|
+
if not _stat_index_dirty or _stat_index_root is None:
|
|
365
|
+
return
|
|
366
|
+
p = _stat_index_file(_stat_index_root)
|
|
367
|
+
# Build the on-disk form (#2199): prune entries whose file is gone (the
|
|
368
|
+
# index otherwise grows without bound), then store in-anchor keys as
|
|
369
|
+
# forward-slash relative paths so the index survives a corpus move/clone.
|
|
370
|
+
# Out-of-anchor keys stay absolute (same rule as the detect manifest); a
|
|
371
|
+
# reader tells the formats apart by absoluteness, so no version marker is
|
|
372
|
+
# needed. In-memory keys are untouched — only the serialization changes.
|
|
373
|
+
on_disk: dict[str, dict] = {}
|
|
374
|
+
for k, v in _stat_index.items():
|
|
375
|
+
try:
|
|
376
|
+
if not os.path.exists(k):
|
|
377
|
+
continue
|
|
378
|
+
except OSError:
|
|
379
|
+
continue
|
|
380
|
+
dk = _stat_key_to_relative(k, _stat_index_anchor) if _stat_index_anchor is not None else k
|
|
381
|
+
on_disk[dk] = v
|
|
382
|
+
# Never resurrect a corpus that was deleted while graphify was running
|
|
383
|
+
# (#2974): a hook-launched `graphify update . &` in a short-lived worktree
|
|
384
|
+
# outlives `git worktree remove`, and an unconditional `mkdir -p` here
|
|
385
|
+
# rebuilt the dead path as a husk holding nothing but this index. The
|
|
386
|
+
# index is a pure optimisation, so when its root is gone it is simply not
|
|
387
|
+
# written. Creating graphify-out/cache/ under a root that still exists is
|
|
388
|
+
# unchanged (a first run writes the index before anything else does).
|
|
389
|
+
try:
|
|
390
|
+
if not _stat_index_root.is_dir():
|
|
391
|
+
_stat_index_dirty = False
|
|
392
|
+
return
|
|
393
|
+
except OSError:
|
|
394
|
+
_stat_index_dirty = False
|
|
395
|
+
return
|
|
396
|
+
try:
|
|
397
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
398
|
+
fd, tmp = tempfile.mkstemp(dir=p.parent, prefix="stat-index.", suffix=".tmp")
|
|
399
|
+
try:
|
|
400
|
+
os.write(fd, json.dumps(on_disk, separators=(",", ":")).encode())
|
|
401
|
+
os.close(fd)
|
|
402
|
+
os.replace(tmp, p)
|
|
403
|
+
except Exception:
|
|
404
|
+
try:
|
|
405
|
+
os.close(fd)
|
|
406
|
+
except OSError:
|
|
407
|
+
pass
|
|
408
|
+
try:
|
|
409
|
+
os.unlink(tmp)
|
|
410
|
+
except OSError:
|
|
411
|
+
pass
|
|
412
|
+
except OSError:
|
|
413
|
+
pass
|
|
414
|
+
_stat_index_dirty = False
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _normalize_path(path: Path) -> Path:
|
|
418
|
+
"""Normalize path for consistent cache keys across Windows path spellings."""
|
|
419
|
+
import sys
|
|
420
|
+
if sys.platform != "win32":
|
|
421
|
+
return path
|
|
422
|
+
s = str(path)
|
|
423
|
+
if s.startswith("\\\\?\\"):
|
|
424
|
+
s = s[4:] # strip extended-length prefix \\?\
|
|
425
|
+
return Path(os.path.normcase(s))
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def file_hash(path: Path, root: Path = Path("."), cache_root: "Path | None" = None) -> str:
|
|
429
|
+
"""SHA256 of file contents + path relative to root.
|
|
430
|
+
|
|
431
|
+
Uses a stat-based fastpath (size + mtime_ns) to skip full reads when the
|
|
432
|
+
file hasn't changed. Falls through to full SHA256 on first encounter, when
|
|
433
|
+
stat changes, and when the recorded signature is not yet provably stable
|
|
434
|
+
(see :func:`_stat_sig_fresh`) — so two different contents can never share a
|
|
435
|
+
digest. Index is flushed atomically at process exit.
|
|
436
|
+
|
|
437
|
+
Using the walked path relative to root keeps distinct symlink aliases from
|
|
438
|
+
sharing an extraction entry while preserving portability across machines
|
|
439
|
+
and checkout directories. Falls back to the resolved path when the walked
|
|
440
|
+
path cannot be expressed relative to root.
|
|
441
|
+
|
|
442
|
+
For Markdown files (.md), only the body below the YAML frontmatter is hashed,
|
|
443
|
+
so metadata-only changes (e.g. reviewed, status, tags) do not invalidate the cache.
|
|
444
|
+
"""
|
|
445
|
+
global _stat_index_dirty
|
|
446
|
+
p = _normalize_path(Path(path))
|
|
447
|
+
root = _normalize_path(Path(root))
|
|
448
|
+
if not p.is_file():
|
|
449
|
+
raise IsADirectoryError(f"file_hash requires a file, got: {p}")
|
|
450
|
+
|
|
451
|
+
# The stat index is a cache artifact, so it must follow the cache location
|
|
452
|
+
# (cache_root), not the key-anchor root — otherwise it leaves a stray
|
|
453
|
+
# graphify-out/cache/stat-index.json inside the analyzed source tree even when
|
|
454
|
+
# the AST cache itself is redirected to CWD (#1774 completion).
|
|
455
|
+
_ensure_stat_index(root, cache_root=cache_root)
|
|
456
|
+
resolved = p.resolve()
|
|
457
|
+
abs_key = str(resolved)
|
|
458
|
+
# The salt is the path component that enters the digest (relative to root, or
|
|
459
|
+
# the absolute-path fallback). The stat-index memo MUST be keyed by it too:
|
|
460
|
+
# the same file hashed under two different roots yields two different digests
|
|
461
|
+
# (this happens within one `--out` run), and a memo keyed only by absolute
|
|
462
|
+
# path served whichever was computed first — making file_hash order-dependent
|
|
463
|
+
# and poisoning the persisted stat-index across runs (#1989). Store one digest
|
|
464
|
+
# per salt so alternating roots don't force re-reads.
|
|
465
|
+
resolved_root = root.resolve()
|
|
466
|
+
try:
|
|
467
|
+
resolved_rel = resolved.relative_to(resolved_root)
|
|
468
|
+
except ValueError:
|
|
469
|
+
# Preserve the existing fallback for a target outside the corpus. An
|
|
470
|
+
# in-root symlink to such a target is excluded by collect_files(), but
|
|
471
|
+
# direct cache callers still rely on the resolved external identity.
|
|
472
|
+
salt = resolved.as_posix().lower()
|
|
473
|
+
else:
|
|
474
|
+
walked = Path(os.path.abspath(p))
|
|
475
|
+
walked_root = Path(os.path.abspath(root))
|
|
476
|
+
try:
|
|
477
|
+
walked_rel = walked.relative_to(walked_root)
|
|
478
|
+
except ValueError:
|
|
479
|
+
# extract() resolves its operational root, while paths collected
|
|
480
|
+
# through a symlinked scan root retain that walked spelling. Find
|
|
481
|
+
# the lexical ancestor representing the resolved corpus root so a
|
|
482
|
+
# leaf symlink still contributes its own relative path to the key.
|
|
483
|
+
walked_rel = None
|
|
484
|
+
for parent in walked.parents:
|
|
485
|
+
try:
|
|
486
|
+
if parent.resolve() == resolved_root:
|
|
487
|
+
walked_rel = walked.relative_to(parent)
|
|
488
|
+
break
|
|
489
|
+
except OSError:
|
|
490
|
+
continue
|
|
491
|
+
if walked_rel is None:
|
|
492
|
+
walked_rel = resolved_rel
|
|
493
|
+
salt = walked_rel.as_posix().lower()
|
|
494
|
+
|
|
495
|
+
st: "os.stat_result | None" = None
|
|
496
|
+
try:
|
|
497
|
+
st = p.stat()
|
|
498
|
+
if _stat_sig_fresh(_stat_index.get(abs_key), st):
|
|
499
|
+
hashes = _stat_index[abs_key].get("hashes")
|
|
500
|
+
if isinstance(hashes, dict):
|
|
501
|
+
cached = hashes.get(salt)
|
|
502
|
+
if isinstance(cached, str):
|
|
503
|
+
return cached
|
|
504
|
+
# Legacy single-digest entries ("hash") don't record which salt
|
|
505
|
+
# produced them, so they are never trusted (#1989) — recompute once.
|
|
506
|
+
except OSError:
|
|
507
|
+
pass
|
|
508
|
+
|
|
509
|
+
# Captured BEFORE the read so the stamp can never post-date content that
|
|
510
|
+
# changed while we were reading it (see _stat_sig_fresh).
|
|
511
|
+
observed_at_ns = time.time_ns()
|
|
512
|
+
raw = p.read_bytes()
|
|
513
|
+
content = _body_content(raw) if p.suffix.lower() == ".md" else raw
|
|
514
|
+
h = hashlib.sha256()
|
|
515
|
+
h.update(content)
|
|
516
|
+
h.update(b"\x00")
|
|
517
|
+
h.update(salt.encode())
|
|
518
|
+
digest = h.hexdigest()
|
|
519
|
+
|
|
520
|
+
if st is not None:
|
|
521
|
+
entry = _stat_entry_for(abs_key, st, observed_at_ns)
|
|
522
|
+
hashes = entry.get("hashes")
|
|
523
|
+
if not isinstance(hashes, dict):
|
|
524
|
+
hashes = {}
|
|
525
|
+
entry["hashes"] = hashes
|
|
526
|
+
hashes[salt] = digest # preserve a co-located word_count / other salts
|
|
527
|
+
entry.pop("hash", None) # retire the un-salted legacy digest
|
|
528
|
+
_stat_index_dirty = True
|
|
529
|
+
|
|
530
|
+
return digest
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def cached_word_count(path: Path, root: Path, compute, cache_root: "Path | None" = None) -> int:
|
|
534
|
+
"""Word count with the same (size, mtime_ns) stat-fastpath cache as
|
|
535
|
+
:func:`file_hash`, persisted in the shared stat index.
|
|
536
|
+
|
|
537
|
+
``detect()`` counts words in every PDF/docx/text file to size the corpus,
|
|
538
|
+
which re-opens and re-parses every binary on each run — minutes on a large
|
|
539
|
+
docs corpus even when only a handful of files changed (#1656). This caches
|
|
540
|
+
the count against the file's stat signature so an unchanged file is counted
|
|
541
|
+
once and read from the index thereafter. ``compute(path)`` produces the
|
|
542
|
+
count on a miss. A file that can't be stat'd (e.g. a Windows long path the
|
|
543
|
+
index normalization can't reach) simply recomputes and isn't cached —
|
|
544
|
+
correct, just not accelerated.
|
|
545
|
+
"""
|
|
546
|
+
global _stat_index_dirty
|
|
547
|
+
p = _normalize_path(Path(path))
|
|
548
|
+
root = _normalize_path(Path(root))
|
|
549
|
+
_ensure_stat_index(root, cache_root=cache_root)
|
|
550
|
+
abs_key = str(p.resolve())
|
|
551
|
+
st: "os.stat_result | None" = None
|
|
552
|
+
try:
|
|
553
|
+
st = p.stat()
|
|
554
|
+
entry = _stat_index.get(abs_key)
|
|
555
|
+
if _stat_sig_fresh(entry, st) and "word_count" in entry:
|
|
556
|
+
return entry["word_count"]
|
|
557
|
+
except OSError:
|
|
558
|
+
pass
|
|
559
|
+
|
|
560
|
+
# Captured BEFORE compute() reads the file, for the same reason file_hash
|
|
561
|
+
# stamps before its read (see _stat_sig_fresh).
|
|
562
|
+
observed_at_ns = time.time_ns()
|
|
563
|
+
wc = compute(Path(path))
|
|
564
|
+
|
|
565
|
+
if st is not None:
|
|
566
|
+
_stat_entry_for(abs_key, st, observed_at_ns)["word_count"] = wc
|
|
567
|
+
_stat_index_dirty = True
|
|
568
|
+
|
|
569
|
+
return wc
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _relativize_source_files_in(payload: dict, root: Path) -> None:
|
|
573
|
+
"""Mutate ``payload`` to rewrite absolute ``source_file`` fields as
|
|
574
|
+
forward-slash relative paths from ``root``.
|
|
575
|
+
|
|
576
|
+
Mirror of :func:`graphify.watch._relativize_source_files` so cached
|
|
577
|
+
extraction fragments persist in portable form (#777). Out-of-root paths
|
|
578
|
+
pass through unchanged.
|
|
579
|
+
|
|
580
|
+
A CWD-relative field is re-anchored too. Extractors stamp ``source_file``
|
|
581
|
+
with the path string ``extract()`` was handed, so relative inputs yield a
|
|
582
|
+
CWD-relative stamp — but the stored format is root-relative, and
|
|
583
|
+
:func:`_absolutize_source_files_in` reads it back as such. When CWD is not
|
|
584
|
+
the inferred root the two disagree and a warm hit resurrects a path that
|
|
585
|
+
names no file (``<root>/src/pages/index.astro`` for an input of
|
|
586
|
+
``src/pages/index.astro`` under root ``<root>``). Every source_file-GATED
|
|
587
|
+
remap in ``extract()`` then misses — the file-stem prefix pass looks up
|
|
588
|
+
``Path(source_file).resolve()`` in ``prefix_remap`` — so a warm hit keeps
|
|
589
|
+
the raw-path symbol ids a cold run canonicalizes: symbols stop sharing
|
|
590
|
+
their file node's stem, and for absolute inputs the on-disk path survives
|
|
591
|
+
into the persisted id (#2630). Only rewritten when the CWD-relative
|
|
592
|
+
reading is a real file and the root-relative reading is a different path,
|
|
593
|
+
so a fragment that already stores root-relative (a semantic subagent's,
|
|
594
|
+
see :func:`_normalize_source_file_value`) is left alone.
|
|
595
|
+
|
|
596
|
+
Only ``root`` is resolved — ``source_file`` itself is relativized
|
|
597
|
+
symbolically so in-root symlinks keep their original name rather than
|
|
598
|
+
pointing at the resolved target. Same reasoning as
|
|
599
|
+
:func:`graphify.detect._to_relative_for_storage`.
|
|
600
|
+
"""
|
|
601
|
+
try:
|
|
602
|
+
root_resolved = Path(root).resolve()
|
|
603
|
+
except OSError:
|
|
604
|
+
return
|
|
605
|
+
# raw_calls (#: Pascal/Delphi cross-file inherited-call resolution) carries
|
|
606
|
+
# source_file the same way nodes/edges/hyperedges do, so it needs the same
|
|
607
|
+
# portable-path treatment for cache entries to round-trip correctly across
|
|
608
|
+
# machines/checkout directories.
|
|
609
|
+
# definition_file (#2990) is a path into the scanned tree exactly like
|
|
610
|
+
# source_file; a cache entry keeping it absolute replayed the build host's
|
|
611
|
+
# layout on every warm hit (#3223).
|
|
612
|
+
for bucket in ("nodes", "edges", "hyperedges", "raw_calls"):
|
|
613
|
+
for item in payload.get(bucket, []):
|
|
614
|
+
if not isinstance(item, dict):
|
|
615
|
+
continue
|
|
616
|
+
for key in ("source_file", "definition_file"):
|
|
617
|
+
source = item.get(key)
|
|
618
|
+
if not source:
|
|
619
|
+
continue
|
|
620
|
+
sp = Path(source)
|
|
621
|
+
if not sp.is_absolute():
|
|
622
|
+
# os.path.abspath is lexical (no symlink resolution),
|
|
623
|
+
# matching the symbolic relativization below.
|
|
624
|
+
cwd_form = Path(os.path.abspath(sp))
|
|
625
|
+
try:
|
|
626
|
+
if cwd_form == root_resolved / sp or not cwd_form.exists():
|
|
627
|
+
continue # already root-relative, or a ghost path
|
|
628
|
+
except OSError:
|
|
629
|
+
continue
|
|
630
|
+
sp = cwd_form
|
|
631
|
+
try:
|
|
632
|
+
rel = os.path.relpath(sp, root_resolved)
|
|
633
|
+
except (ValueError, OSError):
|
|
634
|
+
continue # out-of-root (e.g. Windows cross-drive)
|
|
635
|
+
if rel == ".." or rel.startswith(".." + os.sep) or rel.startswith("../"):
|
|
636
|
+
continue # escaped root — keep absolute
|
|
637
|
+
item[key] = rel.replace(os.sep, "/")
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
def _normalize_source_file_value(src: "str | Path", root_resolved: Path) -> str:
|
|
641
|
+
"""Return ``src`` in portable form: backslashes flipped to forward slashes,
|
|
642
|
+
then relativized against ``root_resolved`` when the path is in-root.
|
|
643
|
+
|
|
644
|
+
Windows ``detect()`` emits absolute backslash paths, and a semantic
|
|
645
|
+
fragment carrying one verbatim used to be persisted as-is — poisoning later
|
|
646
|
+
``graphify update`` runs with a machine-specific ``source_file`` (#2197).
|
|
647
|
+
Out-of-root absolute paths pass through (slash-normalized only), same
|
|
648
|
+
in/out rule and ``..``-rejection as :func:`_relativize_source_files_in`.
|
|
649
|
+
"""
|
|
650
|
+
s = str(src).replace("\\", "/")
|
|
651
|
+
p = Path(s)
|
|
652
|
+
if not p.is_absolute():
|
|
653
|
+
return s
|
|
654
|
+
try:
|
|
655
|
+
rel = os.path.relpath(p, root_resolved)
|
|
656
|
+
except (ValueError, OSError):
|
|
657
|
+
return s # out-of-root (e.g. Windows cross-drive)
|
|
658
|
+
if rel == ".." or rel.startswith(".." + os.sep) or rel.startswith("../"):
|
|
659
|
+
return s # escaped root — keep absolute
|
|
660
|
+
return rel.replace(os.sep, "/")
|
|
661
|
+
|
|
662
|
+
|
|
663
|
+
def _semantic_entry_matches_path(result: dict, path: Path, root: Path) -> bool:
|
|
664
|
+
"""Whether cached semantic groups belong to the requested walked path.
|
|
665
|
+
|
|
666
|
+
Before walked paths entered the cache salt, a symlink could overwrite its
|
|
667
|
+
target's unversioned semantic entry. Rejecting that mismatched legacy
|
|
668
|
+
payload makes the next extraction self-heal instead of replaying it forever.
|
|
669
|
+
"""
|
|
670
|
+
expected = _normalize_path(Path(os.path.abspath(path)))
|
|
671
|
+
for bucket in ("nodes", "edges", "hyperedges"):
|
|
672
|
+
for item in result.get(bucket, []):
|
|
673
|
+
if not isinstance(item, dict):
|
|
674
|
+
continue
|
|
675
|
+
source = item.get("source_file")
|
|
676
|
+
if not source:
|
|
677
|
+
continue
|
|
678
|
+
source_path = Path(source)
|
|
679
|
+
if not source_path.is_absolute():
|
|
680
|
+
source_path = Path(root) / source_path
|
|
681
|
+
if _normalize_path(Path(os.path.abspath(source_path))) != expected:
|
|
682
|
+
return False
|
|
683
|
+
return True
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
# Storage marker standing in for the absolute root a cached id/path was minted
|
|
687
|
+
# under (#2257). Extractors mint node ids from the path STRING they are handed
|
|
688
|
+
# (``_make_id(str(path))``, ``_file_node_id(path)``), so a cache entry written
|
|
689
|
+
# under root A embeds A's slug in every id and edge endpoint. Those are only
|
|
690
|
+
# rewritten to the canonical root-relative form by extract()'s whole-graph
|
|
691
|
+
# id-remap, which keys its rewrites off the CURRENT run's paths — so on a warm
|
|
692
|
+
# hit under root B (a clone, a moved checkout, a second mount) the stored ids
|
|
693
|
+
# match no key and A's machine slug survives into graph.json. Entries are
|
|
694
|
+
# therefore stored root-anchored: the root's contribution is replaced by this
|
|
695
|
+
# marker on write and re-anchored to the current root on read, so a replay is
|
|
696
|
+
# portable by construction and reproduces exactly what a cold run under the
|
|
697
|
+
# current root would have minted (the pre-remap form every downstream pass in
|
|
698
|
+
# extract() expects). Same store-portable/re-anchor-on-load contract as
|
|
699
|
+
# ``source_file`` (#777) and the stat index (#2199). Neither ``$`` nor ``-`` can
|
|
700
|
+
# occur in a normalized id (``normalize_id`` drops every non-word character), and
|
|
701
|
+
# no plausible source literal — a shell ``$root``, a template ``${root}`` — opens
|
|
702
|
+
# with this exact token, so the marker cannot collide with extractor output.
|
|
703
|
+
_ROOT_MARKER = "$graphify-root$"
|
|
704
|
+
|
|
705
|
+
|
|
706
|
+
def _id_anchor(path_str: str, rel_str: str) -> str:
|
|
707
|
+
"""Return the id-slug prefix ``path_str`` contributes above ``rel_str``.
|
|
708
|
+
|
|
709
|
+
``_make_id`` normalizes a whole path string, and normalization distributes
|
|
710
|
+
over path joins (every separator run collapses to one ``_``), so an id
|
|
711
|
+
minted from an absolute path decomposes exactly as
|
|
712
|
+
``normalize_id(root) + "_" + normalize_id(rel)``. Recovering the root half
|
|
713
|
+
from the file's own two spellings — rather than assuming it equals the
|
|
714
|
+
scan root — keeps the decomposition exact when the extractor was handed an
|
|
715
|
+
unresolved or relative path (a symlinked root such as macOS ``/tmp`` ->
|
|
716
|
+
``/private/tmp``, or inputs given relative to CWD).
|
|
717
|
+
|
|
718
|
+
Returns "" when ``path_str`` contributes no prefix (it already IS the
|
|
719
|
+
relative form) or when the two spellings disagree about the tail — both
|
|
720
|
+
mean "nothing to re-anchor", which leaves the entry untouched.
|
|
721
|
+
"""
|
|
722
|
+
from graphify.ids import normalize_id # ids imports only re/unicodedata: no cycle
|
|
723
|
+
|
|
724
|
+
full = normalize_id(path_str)
|
|
725
|
+
tail = normalize_id(rel_str)
|
|
726
|
+
if not tail or full == tail:
|
|
727
|
+
return ""
|
|
728
|
+
suffix = "_" + tail
|
|
729
|
+
return full[: -len(suffix)] if full.endswith(suffix) else ""
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def _portability_anchors(path: "str | Path", root: "str | Path") -> tuple[list[str], str, list[str], str]:
|
|
733
|
+
"""Root forms to strip from / restore into one cache entry (#2257).
|
|
734
|
+
|
|
735
|
+
Returns ``(id_anchors, id_restore, path_anchors, path_restore)``. The two
|
|
736
|
+
``*_anchors`` lists are the forms a stored value may have been minted from,
|
|
737
|
+
longest first so a shorter form can't shadow a longer one; the two
|
|
738
|
+
``*_restore`` values are what the CURRENT run's extractor would mint.
|
|
739
|
+
|
|
740
|
+
Several spellings are collected because one entry can hold ids minted from
|
|
741
|
+
different forms: this file's own path as passed and as resolved, plus
|
|
742
|
+
cross-file edge targets minted from ANOTHER in-root file's absolute path
|
|
743
|
+
(which share the scan root's spelling). They collapse to a single marker on
|
|
744
|
+
write; that is safe because extract()'s remap registers both the input-form
|
|
745
|
+
and absolute-resolved-form ids for every path (#1529) and maps them to the
|
|
746
|
+
same canonical id, so restoring either one canonicalizes identically.
|
|
747
|
+
"""
|
|
748
|
+
from graphify.ids import normalize_id
|
|
749
|
+
|
|
750
|
+
try:
|
|
751
|
+
root_resolved = Path(root).resolve()
|
|
752
|
+
except OSError:
|
|
753
|
+
return [], "", [], ""
|
|
754
|
+
try:
|
|
755
|
+
path_resolved = Path(path).resolve()
|
|
756
|
+
except (OSError, RuntimeError):
|
|
757
|
+
path_resolved = Path(path)
|
|
758
|
+
try:
|
|
759
|
+
rel = os.path.relpath(path_resolved, root_resolved)
|
|
760
|
+
except (ValueError, OSError):
|
|
761
|
+
rel = ""
|
|
762
|
+
|
|
763
|
+
# Ordered by preference for the restore form: the spelling the extractor was
|
|
764
|
+
# actually handed first, then its resolved form, then the scan root.
|
|
765
|
+
from_given = _id_anchor(str(path), rel)
|
|
766
|
+
from_resolved = _id_anchor(str(path_resolved), rel)
|
|
767
|
+
id_restore = next(
|
|
768
|
+
(a for a in (from_given, from_resolved, normalize_id(str(root_resolved))) if a), ""
|
|
769
|
+
)
|
|
770
|
+
# Every strippable form must be one this same call would RESTORE, or an id
|
|
771
|
+
# is re-anchored under a prefix it was never minted with. That rules out a
|
|
772
|
+
# relative ``root`` spelling ("src"): with an absolute ``path`` the restore
|
|
773
|
+
# form is the resolved slug, so admitting ``src`` as an anchor would rewrite
|
|
774
|
+
# an already-canonical ``src_utils_foo`` into ``<abs-root-slug>_utils_foo``.
|
|
775
|
+
# The two path-derived forms are always safe — they ARE the restore
|
|
776
|
+
# candidates — and cover a relative root on their own whenever the extractor
|
|
777
|
+
# was handed a matching relative path.
|
|
778
|
+
root_id_forms = (normalize_id(str(root_resolved)),)
|
|
779
|
+
if Path(root).is_absolute():
|
|
780
|
+
root_id_forms += (normalize_id(str(root)),)
|
|
781
|
+
id_anchors = sorted(
|
|
782
|
+
{a for a in (from_given, from_resolved, *root_id_forms) if a},
|
|
783
|
+
key=len, reverse=True,
|
|
784
|
+
)
|
|
785
|
+
# Only absolute roots may anchor a PATH value: a relative one ("corpus")
|
|
786
|
+
# would also match a genuinely relative value that merely starts with the
|
|
787
|
+
# same segment, and there is no way to tell the two apart on read.
|
|
788
|
+
path_anchors = sorted(
|
|
789
|
+
{s for s in (str(root_resolved), str(root)) if Path(s).is_absolute()},
|
|
790
|
+
key=len, reverse=True,
|
|
791
|
+
)
|
|
792
|
+
return id_anchors, id_restore, path_anchors, str(root_resolved)
|
|
793
|
+
|
|
794
|
+
|
|
795
|
+
def _rewrite_id_keyed_table_keys(payload: object, fn) -> None:
|
|
796
|
+
"""Apply ``fn`` to objc_field_types["tables"] KEYS (#3150).
|
|
797
|
+
|
|
798
|
+
That table is the one extractor bucket keyed BY node id, which
|
|
799
|
+
:func:`_rewrite_strings` deliberately never touches - so a cached ObjC
|
|
800
|
+
shard replayed under another root kept absolute-derived class ids as keys
|
|
801
|
+
while the node ids themselves were re-anchored, and the receiver-typing
|
|
802
|
+
pass missed every class.
|
|
803
|
+
"""
|
|
804
|
+
ft = payload.get("objc_field_types") if isinstance(payload, dict) else None
|
|
805
|
+
tables = ft.get("tables") if isinstance(ft, dict) else None
|
|
806
|
+
if isinstance(tables, dict):
|
|
807
|
+
ft["tables"] = {
|
|
808
|
+
(fn(k) if isinstance(k, str) else k): v for k, v in tables.items()
|
|
809
|
+
}
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
def _rewrite_strings(obj: object, fn) -> None:
|
|
813
|
+
"""Apply ``fn`` to every string VALUE reachable in ``obj``, in place.
|
|
814
|
+
|
|
815
|
+
Values only, never dict keys: rewriting keys blindly could silently
|
|
816
|
+
collide two entries into one. The single id-keyed bucket -
|
|
817
|
+
``objc_field_types["tables"]`` - is handled by
|
|
818
|
+
:func:`_rewrite_id_keyed_table_keys` beside each call to this (#3150).
|
|
819
|
+
"""
|
|
820
|
+
if isinstance(obj, dict):
|
|
821
|
+
items: "Iterable" = obj.items()
|
|
822
|
+
elif isinstance(obj, list):
|
|
823
|
+
items = enumerate(obj)
|
|
824
|
+
else:
|
|
825
|
+
return
|
|
826
|
+
for key, value in list(items):
|
|
827
|
+
if isinstance(value, str):
|
|
828
|
+
new = fn(value)
|
|
829
|
+
if new != value:
|
|
830
|
+
obj[key] = new # type: ignore[index]
|
|
831
|
+
else:
|
|
832
|
+
_rewrite_strings(value, fn)
|
|
833
|
+
|
|
834
|
+
|
|
835
|
+
def _relativize_ids_in(payload: dict, path: "str | Path", root: Path) -> None:
|
|
836
|
+
"""Replace the absolute root inside every stored id / path with the marker.
|
|
837
|
+
|
|
838
|
+
Walks the WHOLE payload rather than a list of known buckets. The id form is
|
|
839
|
+
self-identifying — a long casefolded path slug that nothing but an
|
|
840
|
+
absolute-path-derived id can start with — so this reaches carriers a
|
|
841
|
+
hand-maintained bucket list misses or has yet to grow: ``nodes[].id``,
|
|
842
|
+
``edges[].source``/``target``, hyperedge member lists under any of their
|
|
843
|
+
aliases, ``raw_calls[].caller_nid``, ``swift_extensions[].nid``, plus the
|
|
844
|
+
path-valued ``edges[].target_file``, ``bash_sources[].source_file`` and
|
|
845
|
+
``*_type_table.path``, which resolution passes need pointing at the current
|
|
846
|
+
root to reproduce a cold run's edges.
|
|
847
|
+
|
|
848
|
+
Call AFTER :func:`_relativize_source_files_in`: that stores ``source_file``
|
|
849
|
+
as a bare relative path (the #777 format), and running this first would
|
|
850
|
+
leave it marker-prefixed instead, churning the on-disk format for nothing.
|
|
851
|
+
"""
|
|
852
|
+
id_anchors, _, path_anchors, _ = _portability_anchors(path, root)
|
|
853
|
+
if not id_anchors and not path_anchors:
|
|
854
|
+
return
|
|
855
|
+
|
|
856
|
+
def anchor(value: str) -> str:
|
|
857
|
+
# Path form first: it requires a separator, which an id never contains.
|
|
858
|
+
for a in path_anchors:
|
|
859
|
+
if value == a:
|
|
860
|
+
return _ROOT_MARKER
|
|
861
|
+
for sep in ("/", "\\"):
|
|
862
|
+
if value.startswith(a + sep):
|
|
863
|
+
return _ROOT_MARKER + "/" + value[len(a) + 1:].replace("\\", "/")
|
|
864
|
+
for a in id_anchors:
|
|
865
|
+
if value.startswith(a + "_"):
|
|
866
|
+
return _ROOT_MARKER + "_" + value[len(a) + 1:]
|
|
867
|
+
return value
|
|
868
|
+
|
|
869
|
+
_rewrite_strings(payload, anchor)
|
|
870
|
+
_rewrite_id_keyed_table_keys(payload, anchor)
|
|
871
|
+
|
|
872
|
+
|
|
873
|
+
def _absolutize_ids_in(payload: dict, path: "str | Path", root: Path) -> None:
|
|
874
|
+
"""Inverse of :func:`_relativize_ids_in` — re-anchor to the current root.
|
|
875
|
+
|
|
876
|
+
The restored string is assembled by slicing, never by re-normalizing: the
|
|
877
|
+
stored form (``$graphify-root$_pkg_mod_base``) is a storage encoding, not a valid id,
|
|
878
|
+
and running it back through ``normalize_id`` would drop the marker's ``$``
|
|
879
|
+
and fuse it into the slug. Entries written before #2257 carry no marker and
|
|
880
|
+
pass through untouched — they are swept anyway, since AST entries live under
|
|
881
|
+
a per-version directory (:func:`cache_dir`).
|
|
882
|
+
"""
|
|
883
|
+
_, id_restore, _, path_restore = _portability_anchors(path, root)
|
|
884
|
+
|
|
885
|
+
def restore(value: str) -> str:
|
|
886
|
+
if not value.startswith(_ROOT_MARKER):
|
|
887
|
+
return value
|
|
888
|
+
rest = value[len(_ROOT_MARKER):]
|
|
889
|
+
if not rest:
|
|
890
|
+
return path_restore
|
|
891
|
+
if rest[0] == "/":
|
|
892
|
+
tail = rest[1:]
|
|
893
|
+
return str(Path(path_restore) / tail) if tail else path_restore
|
|
894
|
+
if rest[0] == "_":
|
|
895
|
+
return (id_restore + rest) if id_restore else rest[1:]
|
|
896
|
+
return value
|
|
897
|
+
|
|
898
|
+
_rewrite_strings(payload, restore)
|
|
899
|
+
_rewrite_id_keyed_table_keys(payload, restore)
|
|
900
|
+
|
|
901
|
+
|
|
902
|
+
def _absolutize_source_files_in(payload: dict, root: Path) -> None:
|
|
903
|
+
"""Inverse of :func:`_relativize_source_files_in`.
|
|
904
|
+
|
|
905
|
+
Re-anchor relative ``source_file`` fields against ``root`` so callers
|
|
906
|
+
that load a cached fragment see the same absolute-path shape that a
|
|
907
|
+
fresh in-process extraction would produce. Legacy cache entries with
|
|
908
|
+
absolute ``source_file`` values pass through unchanged.
|
|
909
|
+
"""
|
|
910
|
+
try:
|
|
911
|
+
root_resolved = Path(root).resolve()
|
|
912
|
+
except OSError:
|
|
913
|
+
return
|
|
914
|
+
for bucket in ("nodes", "edges", "hyperedges", "raw_calls"):
|
|
915
|
+
for item in payload.get(bucket, []):
|
|
916
|
+
if not isinstance(item, dict):
|
|
917
|
+
continue
|
|
918
|
+
# Mirror of the relativize side: definition_file re-anchors too (#3223).
|
|
919
|
+
for key in ("source_file", "definition_file"):
|
|
920
|
+
source = item.get(key)
|
|
921
|
+
if not source:
|
|
922
|
+
continue
|
|
923
|
+
sp = Path(source)
|
|
924
|
+
if sp.is_absolute():
|
|
925
|
+
continue
|
|
926
|
+
try:
|
|
927
|
+
item[key] = str(root_resolved / sp)
|
|
928
|
+
except (TypeError, OSError):
|
|
929
|
+
continue
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
def cache_dir(root: Path = Path("."), kind: str = "ast",
|
|
933
|
+
prompt_fp: str | None = None) -> Path:
|
|
934
|
+
"""Returns the cache directory for ``kind`` - creates it if needed.
|
|
935
|
+
|
|
936
|
+
kind is "ast", "semantic", or a mode-namespaced semantic kind such as
|
|
937
|
+
"semantic-deep" (#1894). Separate subdirectories prevent semantic cache
|
|
938
|
+
entries from overwriting AST cache entries for the same source_file (#582).
|
|
939
|
+
|
|
940
|
+
AST entries live in graphify-out/cache/ast/v{version}-s{schema}/, namespaced
|
|
941
|
+
by graphify version and cache-key schema because they depend on extractor
|
|
942
|
+
code and key semantics, not just file contents. Semantic entries are still
|
|
943
|
+
NOT version-namespaced (re-extraction
|
|
944
|
+
costs LLM calls, #1252): they live in graphify-out/cache/semantic/, with
|
|
945
|
+
deep-mode entries beside them in graphify-out/cache/semantic-deep/.
|
|
946
|
+
|
|
947
|
+
``prompt_fp`` (semantic kinds only) adds a p{fingerprint}/ subdirectory so
|
|
948
|
+
entries are attributed to the extraction prompt that produced them (#1939).
|
|
949
|
+
Omitting it yields the historical flat layout, where entries of unknown
|
|
950
|
+
vintage live.
|
|
951
|
+
"""
|
|
952
|
+
_out = Path(_GRAPHIFY_OUT)
|
|
953
|
+
base = _out if _out.is_absolute() else Path(root).resolve() / _out
|
|
954
|
+
d = base / "cache" / kind
|
|
955
|
+
if kind == "ast":
|
|
956
|
+
d = d / f"v{_EXTRACTOR_VERSION}-s{_AST_CACHE_SCHEMA}"
|
|
957
|
+
_cleanup_stale_ast_entries(d.parent, d)
|
|
958
|
+
elif prompt_fp:
|
|
959
|
+
d = d / f"p{prompt_fp}"
|
|
960
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
961
|
+
return d
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
def load_cached(path: Path, root: Path = Path("."), kind: str = "ast",
|
|
965
|
+
cache_root: Path | None = None, prompt: "str | Path | None" = None,
|
|
966
|
+
prompt_file: "str | Path | None" = None,
|
|
967
|
+
allow_legacy: bool = True,
|
|
968
|
+
allow_partial: bool = False) -> dict | None:
|
|
969
|
+
"""Return cached extraction for this file if hash matches, else None.
|
|
970
|
+
|
|
971
|
+
Cache key: SHA256 of file contents.
|
|
972
|
+
Cache value: stored as graphify-out/cache/{kind}/{hash}.json (AST entries
|
|
973
|
+
under the per-version subdirectory, see :func:`cache_dir`).
|
|
974
|
+
|
|
975
|
+
``root`` anchors the content-hash key and source_file relativization (it
|
|
976
|
+
must stay the inferred common parent so keys remain portable). ``cache_root``
|
|
977
|
+
decouples *where* the cache directory lives from that anchor — the cache is
|
|
978
|
+
an output and must not land inside a read-only/analyzed source tree (#1774).
|
|
979
|
+
When ``cache_root`` is None the location falls back to ``root`` (unchanged
|
|
980
|
+
behavior for existing callers).
|
|
981
|
+
|
|
982
|
+
AST entries written by other graphify versions — including the legacy
|
|
983
|
+
flat cache/ layout (pre-0.5.3) and the unversioned cache/ast/ layout —
|
|
984
|
+
are deliberately not consulted: they were produced by a different
|
|
985
|
+
extractor and may be stale.
|
|
986
|
+
|
|
987
|
+
``prompt`` (semantic kinds) is the extraction prompt — text, or a Path to
|
|
988
|
+
the prompt file — that the caller is about to extract with. It selects the
|
|
989
|
+
p{fingerprint}/ namespace, so an entry produced by a different prompt is a
|
|
990
|
+
miss rather than a silent stale hit (#1939). When it is given and the
|
|
991
|
+
fingerprinted namespace misses, ``allow_legacy`` (default True) falls back
|
|
992
|
+
to a flat-layout entry: those predate fingerprinting, so their vintage is
|
|
993
|
+
unknowable — they are served rather than re-billed, and the hit is counted
|
|
994
|
+
so :func:`check_semantic_cache` can report N to the user. Callers that must
|
|
995
|
+
not mix vintages within one entry (see :func:`save_semantic_cache`'s
|
|
996
|
+
``merge_existing``) pass allow_legacy=False.
|
|
997
|
+
Returns None if no cache entry or file has changed.
|
|
998
|
+
"""
|
|
999
|
+
global _legacy_semantic_hits, _corrupt_cache_entries
|
|
1000
|
+
location = cache_root if cache_root is not None else root
|
|
1001
|
+
try:
|
|
1002
|
+
h = file_hash(path, root, cache_root=cache_root)
|
|
1003
|
+
except OSError:
|
|
1004
|
+
return None
|
|
1005
|
+
prompt_fp = _resolve_prompt_fp(prompt, prompt_file)
|
|
1006
|
+
entry = cache_dir(location, kind, prompt_fp) / f"{h}.json"
|
|
1007
|
+
legacy_hit = False
|
|
1008
|
+
if prompt_fp and not entry.exists() and allow_legacy:
|
|
1009
|
+
legacy = cache_dir(location, kind) / f"{h}.json"
|
|
1010
|
+
if legacy.exists():
|
|
1011
|
+
entry, legacy_hit = legacy, True
|
|
1012
|
+
if entry.exists():
|
|
1013
|
+
try:
|
|
1014
|
+
result = json.loads(entry.read_text(encoding="utf-8"))
|
|
1015
|
+
except json.JSONDecodeError:
|
|
1016
|
+
# Corrupt entry, not a miss: a truncated write or a bad producer
|
|
1017
|
+
# (e.g. unescaped Windows backslashes in source_file) leaves JSON
|
|
1018
|
+
# that fails to parse on every future run, so the file is silently
|
|
1019
|
+
# re-extracted forever. Count it so the run can report it (#2405).
|
|
1020
|
+
_corrupt_cache_entries += 1
|
|
1021
|
+
return None
|
|
1022
|
+
except OSError:
|
|
1023
|
+
return None
|
|
1024
|
+
# A ``partial`` entry was produced from a truncated LLM response and
|
|
1025
|
+
# covers only part of the file's symbols. Serving it as authoritative
|
|
1026
|
+
# would return the incomplete node set forever until the file is
|
|
1027
|
+
# re-extracted. Treat it as a cache MISS (the normal read path) so the
|
|
1028
|
+
# file is re-dispatched and retried. Self-heals: a later complete
|
|
1029
|
+
# extraction overwrites the same content-hash key with a non-partial
|
|
1030
|
+
# entry. ``allow_partial`` is the one exception — the merge_existing
|
|
1031
|
+
# checkpoint peeks at a partial prev so it can accumulate a file's slices
|
|
1032
|
+
# across chunks without losing the truncated one (it stays partial).
|
|
1033
|
+
if not allow_partial and isinstance(result, dict) and result.get("partial"):
|
|
1034
|
+
return None
|
|
1035
|
+
# A semantic entry with zero nodes and zero hyperedges is invalid (#2927):
|
|
1036
|
+
# an edge-only or empty result (e.g. LLM omitted entities for the file)
|
|
1037
|
+
# is not a valid standalone extraction. Treating it as a cache MISS
|
|
1038
|
+
# ensures the file is re-dispatched and retried (#933/#1666).
|
|
1039
|
+
if (
|
|
1040
|
+
not allow_partial
|
|
1041
|
+
and kind.startswith("semantic")
|
|
1042
|
+
and isinstance(result, dict)
|
|
1043
|
+
and not result.get("nodes")
|
|
1044
|
+
and not result.get("hyperedges")
|
|
1045
|
+
):
|
|
1046
|
+
return None
|
|
1047
|
+
if (
|
|
1048
|
+
kind.startswith("semantic")
|
|
1049
|
+
and isinstance(result, dict)
|
|
1050
|
+
and not _semantic_entry_matches_path(result, Path(path), Path(root))
|
|
1051
|
+
):
|
|
1052
|
+
return None
|
|
1053
|
+
if legacy_hit:
|
|
1054
|
+
_legacy_semantic_hits += 1
|
|
1055
|
+
# Re-anchor relative source_file fields so callers see the same
|
|
1056
|
+
# absolute-path shape that a fresh in-process extraction produces
|
|
1057
|
+
# (#777). Legacy entries with absolute source_file pass through.
|
|
1058
|
+
if isinstance(result, dict):
|
|
1059
|
+
_absolutize_source_files_in(result, root)
|
|
1060
|
+
# Same contract for the ids and remaining paths the entry embeds
|
|
1061
|
+
# (#2257): without this a warm hit under a different absolute root
|
|
1062
|
+
# replays ids minted from the ORIGINAL root, which extract()'s
|
|
1063
|
+
# id-remap cannot fix because they match none of its current-path
|
|
1064
|
+
# keys. Order is free — source_file never carries the marker.
|
|
1065
|
+
_absolutize_ids_in(result, path, root)
|
|
1066
|
+
return result
|
|
1067
|
+
return None
|
|
1068
|
+
|
|
1069
|
+
|
|
1070
|
+
def save_cached(path: Path, result: dict, root: Path = Path("."), kind: str = "ast",
|
|
1071
|
+
cache_root: Path | None = None, prompt: "str | Path | None" = None,
|
|
1072
|
+
prompt_file: "str | Path | None" = None) -> None:
|
|
1073
|
+
"""Save extraction result for this file.
|
|
1074
|
+
|
|
1075
|
+
Stores as graphify-out/cache/{kind}/{hash}.json where hash = SHA256 of current file contents.
|
|
1076
|
+
result should be a dict with 'nodes' and 'edges' lists.
|
|
1077
|
+
|
|
1078
|
+
``root`` anchors the content-hash key and source_file relativization;
|
|
1079
|
+
``cache_root`` (when given) is where the cache directory is written, decoupled
|
|
1080
|
+
from ``root`` so the cache never lands inside the analyzed source tree (#1774).
|
|
1081
|
+
|
|
1082
|
+
``prompt`` (semantic kinds) is the extraction prompt that produced ``result``
|
|
1083
|
+
— text, or a Path to the prompt file. It stamps the entry into the
|
|
1084
|
+
p{fingerprint}/ namespace so a later run under a different prompt does not
|
|
1085
|
+
replay it (#1939). Writes always land in the fingerprinted namespace when a
|
|
1086
|
+
prompt is given: an entry of known vintage is never written back into the
|
|
1087
|
+
flat unknown-vintage layout.
|
|
1088
|
+
|
|
1089
|
+
No-ops if `path` is not a regular file. Subagent-produced semantic fragments
|
|
1090
|
+
occasionally carry a directory path in `source_file`; skipping them prevents
|
|
1091
|
+
IsADirectoryError from aborting the whole batch.
|
|
1092
|
+
"""
|
|
1093
|
+
p = Path(path)
|
|
1094
|
+
if not p.is_file():
|
|
1095
|
+
return
|
|
1096
|
+
# Relativize source_file fields against ``root`` before write so the
|
|
1097
|
+
# cache file on disk is portable across machines and checkout
|
|
1098
|
+
# directories (#777). The cache key is content-hashed so lookup is
|
|
1099
|
+
# already path-independent; this fixes the embedded path leak.
|
|
1100
|
+
#
|
|
1101
|
+
# Serialize a relativized copy rather than mutating the caller's dict —
|
|
1102
|
+
# downstream pipeline steps (notably extract.py's AST prefix remap, which
|
|
1103
|
+
# looks up Path(source_file).resolve() in a prefix table) depend on the
|
|
1104
|
+
# source_file field's original absolute form. Mutating the input here would
|
|
1105
|
+
# silently break those remaps on the first extraction pass.
|
|
1106
|
+
#
|
|
1107
|
+
# The copy is unconditional (it used to be gated on a non-empty
|
|
1108
|
+
# nodes/edges/hyperedges/raw_calls bucket): a truthiness gate skips the copy
|
|
1109
|
+
# for a result whose only payload lives in another bucket — an empty
|
|
1110
|
+
# ``nodes`` beside a populated ``bash_sources`` — and the id/path anchoring
|
|
1111
|
+
# below would then mutate the caller's dict for real.
|
|
1112
|
+
on_disk = result
|
|
1113
|
+
if isinstance(result, dict):
|
|
1114
|
+
import copy as _copy
|
|
1115
|
+
on_disk = _copy.deepcopy(result)
|
|
1116
|
+
_relativize_source_files_in(on_disk, root)
|
|
1117
|
+
# Then replace the absolute root inside the ids and remaining paths, so
|
|
1118
|
+
# the entry replays portably under any root (#2257). Strictly after the
|
|
1119
|
+
# source_file pass, which owns that field's bare-relative format.
|
|
1120
|
+
_relativize_ids_in(on_disk, p, root)
|
|
1121
|
+
h = file_hash(p, root, cache_root=cache_root)
|
|
1122
|
+
location = cache_root if cache_root is not None else root
|
|
1123
|
+
target_dir = cache_dir(location, kind, _resolve_prompt_fp(prompt, prompt_file))
|
|
1124
|
+
entry = target_dir / f"{h}.json"
|
|
1125
|
+
fd, tmp_path = tempfile.mkstemp(dir=target_dir, prefix=f"{h}.", suffix=".tmp")
|
|
1126
|
+
try:
|
|
1127
|
+
os.write(fd, json.dumps(on_disk).encode())
|
|
1128
|
+
os.close(fd)
|
|
1129
|
+
try:
|
|
1130
|
+
os.replace(tmp_path, entry)
|
|
1131
|
+
except PermissionError:
|
|
1132
|
+
# Windows: os.replace can fail with WinError 5 if the target is
|
|
1133
|
+
# briefly locked. Fall back to copy-then-delete.
|
|
1134
|
+
import shutil
|
|
1135
|
+
shutil.copy2(tmp_path, entry)
|
|
1136
|
+
os.unlink(tmp_path)
|
|
1137
|
+
except Exception:
|
|
1138
|
+
try:
|
|
1139
|
+
os.close(fd)
|
|
1140
|
+
except OSError:
|
|
1141
|
+
pass
|
|
1142
|
+
try:
|
|
1143
|
+
os.unlink(tmp_path)
|
|
1144
|
+
except OSError:
|
|
1145
|
+
pass
|
|
1146
|
+
raise
|
|
1147
|
+
|
|
1148
|
+
|
|
1149
|
+
def cached_files(root: Path = Path(".")) -> set[str]:
|
|
1150
|
+
"""Return set of file hashes that have a valid cache entry (any kind)."""
|
|
1151
|
+
base = Path(root).resolve() / _GRAPHIFY_OUT / "cache"
|
|
1152
|
+
hashes: set[str] = set()
|
|
1153
|
+
# Legacy flat entries
|
|
1154
|
+
if base.is_dir():
|
|
1155
|
+
hashes.update(p.stem for p in base.glob("*.json"))
|
|
1156
|
+
# Namespaced entries, all globbed recursively: ast/ has per-version subdirs,
|
|
1157
|
+
# semantic-deep/ holds --mode deep entries (#1894), and both semantic kinds
|
|
1158
|
+
# have per-prompt-fingerprint subdirs alongside pre-fingerprint flat entries
|
|
1159
|
+
# (#1939).
|
|
1160
|
+
for kind in ("ast", "semantic", "semantic-deep"):
|
|
1161
|
+
d = base / kind
|
|
1162
|
+
if d.is_dir():
|
|
1163
|
+
hashes.update(p.stem for p in d.glob("**/*.json"))
|
|
1164
|
+
return hashes
|
|
1165
|
+
|
|
1166
|
+
|
|
1167
|
+
def clear_cache(root: Path = Path(".")) -> None:
|
|
1168
|
+
"""Delete all cache entries (ast/, semantic/, semantic-deep/, and legacy
|
|
1169
|
+
flat entries)."""
|
|
1170
|
+
base = Path(root).resolve() / _GRAPHIFY_OUT / "cache"
|
|
1171
|
+
# Legacy flat entries
|
|
1172
|
+
if base.is_dir():
|
|
1173
|
+
for f in base.glob("*.json"):
|
|
1174
|
+
f.unlink()
|
|
1175
|
+
# Namespaced entries, all globbed recursively: ast/ has per-version subdirs,
|
|
1176
|
+
# semantic-deep/ holds --mode deep entries (#1894), and both semantic kinds
|
|
1177
|
+
# have per-prompt-fingerprint subdirs (#1939).
|
|
1178
|
+
for kind in ("ast", "semantic", "semantic-deep"):
|
|
1179
|
+
d = base / kind
|
|
1180
|
+
if d.is_dir():
|
|
1181
|
+
for f in d.glob("**/*.json"):
|
|
1182
|
+
f.unlink()
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def prune_semantic_cache(root: Path, live_hashes: set[str]) -> int:
|
|
1186
|
+
"""Remove orphaned semantic cache entries, returning the count pruned.
|
|
1187
|
+
|
|
1188
|
+
The semantic cache is content-hash-keyed (``{file_hash}.json`` under
|
|
1189
|
+
``cache/semantic/``) and deliberately UNVERSIONED — entries are produced by
|
|
1190
|
+
the LLM from file contents, so invalidating them on every release would
|
|
1191
|
+
re-bill extraction. Because it is unversioned it is also never swept by the
|
|
1192
|
+
AST version-cleanup, so every content change or file deletion leaves a
|
|
1193
|
+
permanent orphan entry that accumulates unbounded.
|
|
1194
|
+
|
|
1195
|
+
This sweeps ``cache/semantic/*.json`` AND ``cache/semantic-deep/*.json``
|
|
1196
|
+
(the ``--mode deep`` namespace, #1894) and deletes any entry whose stem
|
|
1197
|
+
(the content hash) is not in ``live_hashes`` — the hashes of the current
|
|
1198
|
+
live document set. Both namespaces are pruned against the SAME live set:
|
|
1199
|
+
liveness is content-based and mode-independent, so a hash that is live for
|
|
1200
|
+
one namespace is live for both. Skipping the deep namespace would re-grow
|
|
1201
|
+
the unbounded-orphan problem this function fixed (#1527). ``*.tmp``
|
|
1202
|
+
atomic-write temporaries are skipped, and only these directories are
|
|
1203
|
+
touched (never ``cache/ast/**`` or anything else). The unversioned design
|
|
1204
|
+
is preserved: we prune by liveness, not by version.
|
|
1205
|
+
|
|
1206
|
+
The sweep recurses into the per-prompt-fingerprint subdirs (#1939) for the
|
|
1207
|
+
same reason it covers the deep namespace: a glob that stopped at the top
|
|
1208
|
+
level would leave every fingerprinted entry permanently unprunable. Entries
|
|
1209
|
+
under a fingerprint other than the current one are pruned by liveness only,
|
|
1210
|
+
never swept wholesale the way :func:`_cleanup_stale_ast_entries` sweeps old
|
|
1211
|
+
AST versions — two hosts with different prompts (verbose vs compact
|
|
1212
|
+
extraction-spec) can share one graphify-out/, and a wholesale sweep would
|
|
1213
|
+
have each run delete the other's entries and re-bill extraction on every
|
|
1214
|
+
alternation. Liveness keeps the total bounded by live docs × prompts seen.
|
|
1215
|
+
|
|
1216
|
+
Best-effort, mirroring :func:`_cleanup_stale_ast_entries`: each unlink is
|
|
1217
|
+
wrapped in ``try/except OSError`` and a failure is ignored. The worst-case
|
|
1218
|
+
failure mode is benign — a surviving orphan costs only one re-extraction of
|
|
1219
|
+
one doc on a future run, never incorrect output.
|
|
1220
|
+
"""
|
|
1221
|
+
_out = Path(_GRAPHIFY_OUT)
|
|
1222
|
+
base = _out if _out.is_absolute() else Path(root).resolve() / _out
|
|
1223
|
+
pruned = 0
|
|
1224
|
+
for kind in ("semantic", "semantic-deep"):
|
|
1225
|
+
semantic_dir = base / "cache" / kind
|
|
1226
|
+
if not semantic_dir.is_dir():
|
|
1227
|
+
continue
|
|
1228
|
+
for entry in semantic_dir.glob("**/*.json"):
|
|
1229
|
+
if entry.stem in live_hashes:
|
|
1230
|
+
continue
|
|
1231
|
+
try:
|
|
1232
|
+
entry.unlink()
|
|
1233
|
+
pruned += 1
|
|
1234
|
+
except OSError:
|
|
1235
|
+
pass
|
|
1236
|
+
return pruned
|
|
1237
|
+
|
|
1238
|
+
|
|
1239
|
+
def check_semantic_cache(
|
|
1240
|
+
files: list[str],
|
|
1241
|
+
root: Path = Path("."),
|
|
1242
|
+
mode: str | None = None,
|
|
1243
|
+
prompt: "str | Path | None" = None,
|
|
1244
|
+
prompt_file: "str | Path | None" = None,
|
|
1245
|
+
cache_root: "Path | None" = None,
|
|
1246
|
+
) -> tuple[list[dict], list[dict], list[dict], list[str]]:
|
|
1247
|
+
"""Check semantic extraction cache for a list of absolute file paths.
|
|
1248
|
+
|
|
1249
|
+
Returns (cached_nodes, cached_edges, cached_hyperedges, uncached_files).
|
|
1250
|
+
Uncached files need Claude extraction; cached files are merged directly.
|
|
1251
|
+
|
|
1252
|
+
``mode`` selects the cache namespace: ``None`` (the default) reads
|
|
1253
|
+
``cache/semantic/`` — byte-identical to the historical behavior, so
|
|
1254
|
+
existing callers that omit it (including older installed skill flows)
|
|
1255
|
+
are unaffected. A non-None mode (e.g. ``"deep"``) reads
|
|
1256
|
+
``cache/semantic-{mode}/`` instead, so deep-mode results never shadow
|
|
1257
|
+
(or get shadowed by) standard-mode entries for the same content (#1894).
|
|
1258
|
+
|
|
1259
|
+
``prompt`` is the extraction prompt this run will use for the uncached
|
|
1260
|
+
files — the prompt text (Python path) or a Path to the prompt file the
|
|
1261
|
+
agent loaded (skill path, ``references/extraction-spec.md``). Supplying it
|
|
1262
|
+
restricts hits to entries produced by that same prompt, so an upgrade that
|
|
1263
|
+
changed the prompt re-extracts instead of replaying the older vintage
|
|
1264
|
+
(#1939). Entries written before fingerprinting existed still hit — their
|
|
1265
|
+
vintage is unknowable and dropping them would re-bill a whole corpus — but
|
|
1266
|
+
a warning reports how many were served. Omitting ``prompt`` keeps the
|
|
1267
|
+
historical behavior for existing callers.
|
|
1268
|
+
|
|
1269
|
+
``cache_root`` decouples *where* the cache is read from the key-anchor
|
|
1270
|
+
``root``, mirroring :func:`load_cached` and :func:`save_semantic_cache`
|
|
1271
|
+
(#1774 / #1990). With ``--out``, pass the corpus as ``root`` (so content-hash
|
|
1272
|
+
keys and relative-path resolution stay anchored to the source tree) and the
|
|
1273
|
+
output directory as ``cache_root``. Omitting it keeps ``root`` for both.
|
|
1274
|
+
"""
|
|
1275
|
+
global _legacy_semantic_hits
|
|
1276
|
+
kind = "semantic" if mode is None else f"semantic-{mode}"
|
|
1277
|
+
cached_nodes: list[dict] = []
|
|
1278
|
+
cached_edges: list[dict] = []
|
|
1279
|
+
cached_hyperedges: list[dict] = []
|
|
1280
|
+
uncached: list[str] = []
|
|
1281
|
+
legacy_before = _legacy_semantic_hits
|
|
1282
|
+
corrupt_before = _corrupt_cache_entries
|
|
1283
|
+
|
|
1284
|
+
for fpath in files:
|
|
1285
|
+
p = Path(fpath)
|
|
1286
|
+
if not p.is_absolute():
|
|
1287
|
+
p = Path(root) / p
|
|
1288
|
+
result = load_cached(p, root, kind=kind, cache_root=cache_root,
|
|
1289
|
+
prompt=prompt, prompt_file=prompt_file)
|
|
1290
|
+
if result is not None:
|
|
1291
|
+
cached_nodes.extend(result.get("nodes", []))
|
|
1292
|
+
cached_edges.extend(result.get("edges", []))
|
|
1293
|
+
cached_hyperedges.extend(result.get("hyperedges", []))
|
|
1294
|
+
else:
|
|
1295
|
+
uncached.append(fpath)
|
|
1296
|
+
|
|
1297
|
+
legacy = _legacy_semantic_hits - legacy_before
|
|
1298
|
+
if legacy:
|
|
1299
|
+
warnings.warn(
|
|
1300
|
+
f"{legacy} semantic cache entr{'y' if legacy == 1 else 'ies'} predate "
|
|
1301
|
+
"extraction-prompt fingerprinting and were written by an unknown prompt "
|
|
1302
|
+
"version; they were replayed as-is, so this graph may mix extraction "
|
|
1303
|
+
"vintages. Re-run with --force (or GRAPHIFY_FORCE=1) to re-extract them "
|
|
1304
|
+
"with the current prompt (#1939).",
|
|
1305
|
+
RuntimeWarning,
|
|
1306
|
+
stacklevel=2,
|
|
1307
|
+
)
|
|
1308
|
+
|
|
1309
|
+
corrupt = _corrupt_cache_entries - corrupt_before
|
|
1310
|
+
if corrupt:
|
|
1311
|
+
warnings.warn(
|
|
1312
|
+
f"{corrupt} semantic cache entr{'y' if corrupt == 1 else 'ies'} could "
|
|
1313
|
+
"not be parsed as JSON and were treated as misses, so those files were "
|
|
1314
|
+
"re-extracted. A corrupt entry stays on disk and fails again every run; "
|
|
1315
|
+
"run with --force (or GRAPHIFY_FORCE=1) to rewrite them, or clear the "
|
|
1316
|
+
"cache to stop paying for the re-extraction (#2405).",
|
|
1317
|
+
RuntimeWarning,
|
|
1318
|
+
stacklevel=2,
|
|
1319
|
+
)
|
|
1320
|
+
|
|
1321
|
+
return cached_nodes, cached_edges, cached_hyperedges, uncached
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
def _group_has_partial_marker(group: dict) -> bool:
|
|
1325
|
+
"""True if any node/edge/hyperedge in a per-file group carries the internal
|
|
1326
|
+
``_partial`` truncation marker set by the adaptive-retry give-up sites.
|
|
1327
|
+
|
|
1328
|
+
The marker rides the item dicts up through every chunk merge, so it reaches
|
|
1329
|
+
``save_semantic_cache`` on BOTH the incremental checkpoint path (llm.py) and
|
|
1330
|
+
the final authoritative save (cli.py) without either caller having to thread
|
|
1331
|
+
an extra argument — the final save would otherwise overwrite a checkpoint's
|
|
1332
|
+
``partial`` flag with a clean-looking entry.
|
|
1333
|
+
"""
|
|
1334
|
+
for bucket in ("nodes", "edges", "hyperedges"):
|
|
1335
|
+
for item in group.get(bucket, []):
|
|
1336
|
+
if isinstance(item, dict) and item.get("_partial"):
|
|
1337
|
+
return True
|
|
1338
|
+
return False
|
|
1339
|
+
|
|
1340
|
+
|
|
1341
|
+
def _semantic_source_matcher(
|
|
1342
|
+
root: Path,
|
|
1343
|
+
) -> tuple[Callable[[str | Path], Path], Callable[[str | Path], str]]:
|
|
1344
|
+
"""Shared path-identity machinery for the semantic-scope guards (#1757/#2926).
|
|
1345
|
+
|
|
1346
|
+
``save_semantic_cache``'s write allowlist and ``scope_semantic_result``'s
|
|
1347
|
+
graph-feed filter must agree on exactly which ``source_file`` values are in
|
|
1348
|
+
scope, so both derive their matching from this one implementation rather
|
|
1349
|
+
than parallel copies that could drift apart.
|
|
1350
|
+
|
|
1351
|
+
Returns ``(source_identity, normalize_value)`` closed over ``root``:
|
|
1352
|
+
|
|
1353
|
+
- ``normalize_value(src)`` maps a raw ``source_file`` to its portable
|
|
1354
|
+
relative forward-slash form (#2197),
|
|
1355
|
+
- ``source_identity(value)`` maps any form (relative or absolute, against
|
|
1356
|
+
the walked or the resolved root) to the single canonical walked path
|
|
1357
|
+
that identities are compared against.
|
|
1358
|
+
"""
|
|
1359
|
+
root_walked = _normalize_path(Path(os.path.abspath(root)))
|
|
1360
|
+
root_resolved = _normalize_path(Path(root).resolve())
|
|
1361
|
+
|
|
1362
|
+
def normalize_value(src: str | Path) -> str:
|
|
1363
|
+
norm = _normalize_source_file_value(src, root_walked)
|
|
1364
|
+
if Path(norm).is_absolute() and root_walked != root_resolved:
|
|
1365
|
+
norm = _normalize_source_file_value(src, root_resolved)
|
|
1366
|
+
return norm
|
|
1367
|
+
|
|
1368
|
+
def source_identity(value: str | Path) -> Path:
|
|
1369
|
+
path = Path(value)
|
|
1370
|
+
if not path.is_absolute():
|
|
1371
|
+
path = root_walked / path
|
|
1372
|
+
elif root_walked != root_resolved:
|
|
1373
|
+
normalized = _normalize_path(Path(os.path.abspath(path)))
|
|
1374
|
+
try:
|
|
1375
|
+
relative = normalized.relative_to(root_resolved)
|
|
1376
|
+
except ValueError:
|
|
1377
|
+
pass
|
|
1378
|
+
else:
|
|
1379
|
+
path = root_walked / relative
|
|
1380
|
+
return _normalize_path(Path(os.path.abspath(path)))
|
|
1381
|
+
|
|
1382
|
+
return source_identity, normalize_value
|
|
1383
|
+
|
|
1384
|
+
|
|
1385
|
+
def save_semantic_cache(
|
|
1386
|
+
nodes: list[dict],
|
|
1387
|
+
edges: list[dict],
|
|
1388
|
+
hyperedges: list[dict] | None = None,
|
|
1389
|
+
root: Path = Path("."),
|
|
1390
|
+
merge_existing: bool = False,
|
|
1391
|
+
allowed_source_files: Iterable[str | Path] | None = None,
|
|
1392
|
+
mode: str | None = None,
|
|
1393
|
+
prompt: "str | Path | None" = None,
|
|
1394
|
+
prompt_file: "str | Path | None" = None,
|
|
1395
|
+
partial_source_files: Iterable[str | Path] | None = None,
|
|
1396
|
+
cache_root: "Path | None" = None,
|
|
1397
|
+
) -> int:
|
|
1398
|
+
"""Save semantic extraction results to cache, keyed by source_file.
|
|
1399
|
+
|
|
1400
|
+
Groups nodes and edges by source_file, then saves one cache entry per file
|
|
1401
|
+
under cache/semantic/ (separate from AST entries in cache/ast/) to prevent
|
|
1402
|
+
hash-key collisions (#582).
|
|
1403
|
+
|
|
1404
|
+
``mode`` selects the cache namespace, mirroring
|
|
1405
|
+
:func:`check_semantic_cache`: ``None`` (the default) writes
|
|
1406
|
+
``cache/semantic/`` — byte-identical to the historical behavior for
|
|
1407
|
+
existing callers that omit it — while a non-None mode (e.g. ``"deep"``)
|
|
1408
|
+
writes ``cache/semantic-{mode}/`` so richer deep-mode results never
|
|
1409
|
+
overwrite standard-mode entries and vice versa (#1894).
|
|
1410
|
+
|
|
1411
|
+
When ``merge_existing`` is True, any already-cached entry for a file is
|
|
1412
|
+
unioned with the new results before saving instead of being overwritten.
|
|
1413
|
+
This lets callers checkpoint incrementally (e.g. once per chunk) without
|
|
1414
|
+
dropping a prior slice of a large file that was split across chunks.
|
|
1415
|
+
|
|
1416
|
+
When ``allowed_source_files`` is provided, only those files may be used as
|
|
1417
|
+
cache-write keys. Semantic nodes can legitimately mention another corpus
|
|
1418
|
+
file, but a model must not be able to replace that file's complete cache
|
|
1419
|
+
entry unless the file was part of the current extraction batch (#1757).
|
|
1420
|
+
|
|
1421
|
+
When ``partial_source_files`` is provided, entries for those files are
|
|
1422
|
+
stamped ``partial: True`` — the extraction was truncated, so the entry is
|
|
1423
|
+
incomplete and :func:`load_cached` must treat it as a miss. Partial-ness is
|
|
1424
|
+
ALSO detected intrinsically from a ``_partial`` marker on any grouped item,
|
|
1425
|
+
so the flag survives even when a caller (e.g. cli.py's final save) does not
|
|
1426
|
+
pass ``partial_source_files``.
|
|
1427
|
+
|
|
1428
|
+
``prompt`` is the extraction prompt that produced these results — text, or
|
|
1429
|
+
a Path to the prompt file. It stamps entries into the p{fingerprint}/
|
|
1430
|
+
namespace so a later run under a different prompt re-extracts rather than
|
|
1431
|
+
replaying them (#1939). Pass the same prompt here as to
|
|
1432
|
+
:func:`check_semantic_cache`, or the write lands in a namespace the next
|
|
1433
|
+
read won't consult.
|
|
1434
|
+
|
|
1435
|
+
``cache_root`` decouples *where* the cache directory is written from the
|
|
1436
|
+
source-key anchor ``root`` — mirroring the same split that :func:`load_cached`
|
|
1437
|
+
and :func:`save_cached` already expose (#1774). When given, cache files land
|
|
1438
|
+
under ``cache_root`` while ``source_file`` paths are still resolved and
|
|
1439
|
+
relativized against ``root``. When omitted, ``root`` is used for both
|
|
1440
|
+
purposes (unchanged behaviour for existing callers). This fixes checkpoints
|
|
1441
|
+
and the final save going to the corpus tree instead of ``--out`` (#1990,
|
|
1442
|
+
#1991).
|
|
1443
|
+
|
|
1444
|
+
Returns the number of files cached.
|
|
1445
|
+
"""
|
|
1446
|
+
from collections import defaultdict
|
|
1447
|
+
|
|
1448
|
+
kind = "semantic" if mode is None else f"semantic-{mode}"
|
|
1449
|
+
source_path, _normalize_value = _semantic_source_matcher(root)
|
|
1450
|
+
|
|
1451
|
+
def _normalized(item: dict) -> dict:
|
|
1452
|
+
"""Copy of ``item`` with a portable ``source_file`` (#2197).
|
|
1453
|
+
|
|
1454
|
+
Normalizing BEFORE grouping means both the group key and the persisted
|
|
1455
|
+
item carry the relative forward-slash form, so a fragment whose
|
|
1456
|
+
source_file arrived absolute (Windows detect() output) can never be
|
|
1457
|
+
cached verbatim. A shallow copy keeps the caller's dicts untouched —
|
|
1458
|
+
downstream steps may still rely on the original absolute shape (same
|
|
1459
|
+
reasoning as :func:`save_cached`'s on-disk deepcopy).
|
|
1460
|
+
"""
|
|
1461
|
+
src = item.get("source_file")
|
|
1462
|
+
if not src:
|
|
1463
|
+
return item
|
|
1464
|
+
norm = _normalize_value(src)
|
|
1465
|
+
if norm != src:
|
|
1466
|
+
item = {**item, "source_file": norm}
|
|
1467
|
+
return item
|
|
1468
|
+
|
|
1469
|
+
by_file: dict[str, dict] = defaultdict(lambda: {"nodes": [], "edges": [], "hyperedges": []})
|
|
1470
|
+
for n in nodes:
|
|
1471
|
+
n = _normalized(n)
|
|
1472
|
+
src = n.get("source_file", "")
|
|
1473
|
+
if src:
|
|
1474
|
+
by_file[src]["nodes"].append(n)
|
|
1475
|
+
for e in edges:
|
|
1476
|
+
e = _normalized(e)
|
|
1477
|
+
src = e.get("source_file", "")
|
|
1478
|
+
if src:
|
|
1479
|
+
by_file[src]["edges"].append(e)
|
|
1480
|
+
for h in (hyperedges or []):
|
|
1481
|
+
h = _normalized(h)
|
|
1482
|
+
src = h.get("source_file", "")
|
|
1483
|
+
if src:
|
|
1484
|
+
by_file[src]["hyperedges"].append(h)
|
|
1485
|
+
|
|
1486
|
+
def resolved_source_path(value: str | Path) -> Path:
|
|
1487
|
+
path = source_path(value)
|
|
1488
|
+
try:
|
|
1489
|
+
return path.resolve()
|
|
1490
|
+
except (OSError, RuntimeError):
|
|
1491
|
+
# Keep the cache write best-effort for inaccessible paths or a
|
|
1492
|
+
# symlink loop emitted by an untrusted semantic result.
|
|
1493
|
+
return Path(os.path.abspath(path))
|
|
1494
|
+
|
|
1495
|
+
allowed_paths = None
|
|
1496
|
+
if allowed_source_files is not None:
|
|
1497
|
+
allowed_paths = {source_path(path) for path in allowed_source_files}
|
|
1498
|
+
|
|
1499
|
+
partial_paths = None
|
|
1500
|
+
if partial_source_files is not None:
|
|
1501
|
+
partial_paths = {source_path(path) for path in partial_source_files}
|
|
1502
|
+
# A chunk that truncated to an EMPTY parse contributes no grouped items,
|
|
1503
|
+
# so its file is absent from by_file and the write loop below would never
|
|
1504
|
+
# stamp it partial — leaving a prior clean slice looking complete (#1950
|
|
1505
|
+
# empty-parse gap). Seed an empty group for each named partial file that
|
|
1506
|
+
# isn't already present, so the loop merges its existing entry and stamps
|
|
1507
|
+
# it partial. Keyed by walked path (deduped against present groups).
|
|
1508
|
+
_present = {source_path(k) for k in by_file}
|
|
1509
|
+
for _pp in partial_paths:
|
|
1510
|
+
if _pp not in _present:
|
|
1511
|
+
by_file[str(_pp)] # defaultdict: create an empty {nodes,edges,hyperedges}
|
|
1512
|
+
|
|
1513
|
+
def group_skipped(fpath: str) -> bool:
|
|
1514
|
+
"""Mirror the write-loop skip condition for one source_file group."""
|
|
1515
|
+
p = resolved_source_path(fpath)
|
|
1516
|
+
return not p.is_file() or (
|
|
1517
|
+
allowed_paths is not None and source_path(fpath) not in allowed_paths
|
|
1518
|
+
)
|
|
1519
|
+
|
|
1520
|
+
# Dangling-reference pruning (#1916). A node group is skipped by the write
|
|
1521
|
+
# loop below when its source_file is not a real file (ghost path) or is
|
|
1522
|
+
# out-of-scope per the #1757 guard — but an edge/hyperedge in an ALLOWED
|
|
1523
|
+
# group that references a node id from a skipped group used to be written
|
|
1524
|
+
# verbatim, so on replay (check_semantic_cache) it dangled forever (the
|
|
1525
|
+
# #1895 merged-result filter runs AFTER this checkpoint write and is
|
|
1526
|
+
# bypassed entirely on replay). Compute the node ids that will be skipped
|
|
1527
|
+
# and drop any to-be-written edge whose endpoint — or hyperedge whose
|
|
1528
|
+
# member (whole-hyperedge drop, mirroring #1895) — references one. Gated
|
|
1529
|
+
# on allowed_source_files so unscoped callers stay byte-identical.
|
|
1530
|
+
if allowed_paths is not None:
|
|
1531
|
+
skipped_ids: set = set()
|
|
1532
|
+
written_ids: set = set()
|
|
1533
|
+
for fpath, result in by_file.items():
|
|
1534
|
+
target = skipped_ids if group_skipped(fpath) else written_ids
|
|
1535
|
+
for n in result["nodes"]:
|
|
1536
|
+
nid = n.get("id")
|
|
1537
|
+
if nid is None:
|
|
1538
|
+
continue
|
|
1539
|
+
try:
|
|
1540
|
+
hash(nid)
|
|
1541
|
+
except TypeError:
|
|
1542
|
+
continue
|
|
1543
|
+
target.add(nid)
|
|
1544
|
+
# A duplicate-attribution node (defined in a skipped AND a written
|
|
1545
|
+
# group) still reaches the cache — don't over-prune references to it.
|
|
1546
|
+
skipped_ids -= written_ids
|
|
1547
|
+
if skipped_ids:
|
|
1548
|
+
|
|
1549
|
+
def edge_dangles(e: dict) -> bool:
|
|
1550
|
+
try:
|
|
1551
|
+
return e.get("source") in skipped_ids or e.get("target") in skipped_ids
|
|
1552
|
+
except TypeError:
|
|
1553
|
+
# Non-hashable endpoint from an untrusted result; leave it
|
|
1554
|
+
# to build-time validation rather than fail the save.
|
|
1555
|
+
return False
|
|
1556
|
+
|
|
1557
|
+
def hyperedge_dangles(h: dict) -> bool:
|
|
1558
|
+
try:
|
|
1559
|
+
return bool(skipped_ids & set(h.get("nodes") or []))
|
|
1560
|
+
except TypeError:
|
|
1561
|
+
return False
|
|
1562
|
+
|
|
1563
|
+
for fpath, result in by_file.items():
|
|
1564
|
+
if group_skipped(fpath):
|
|
1565
|
+
continue
|
|
1566
|
+
result["edges"] = [e for e in result["edges"] if not edge_dangles(e)]
|
|
1567
|
+
result["hyperedges"] = [
|
|
1568
|
+
h for h in result["hyperedges"] if not hyperedge_dangles(h)
|
|
1569
|
+
]
|
|
1570
|
+
|
|
1571
|
+
saved = 0
|
|
1572
|
+
skipped_not_file = 0
|
|
1573
|
+
for fpath, result in by_file.items():
|
|
1574
|
+
cache_path = source_path(fpath)
|
|
1575
|
+
p = resolved_source_path(fpath)
|
|
1576
|
+
if p.is_file():
|
|
1577
|
+
if allowed_paths is not None and cache_path not in allowed_paths:
|
|
1578
|
+
warnings.warn(
|
|
1579
|
+
"semantic cache skipped out-of-scope source_file "
|
|
1580
|
+
f"{fpath!r}; the file was not dispatched for extraction",
|
|
1581
|
+
RuntimeWarning,
|
|
1582
|
+
stacklevel=2,
|
|
1583
|
+
)
|
|
1584
|
+
continue
|
|
1585
|
+
if merge_existing:
|
|
1586
|
+
# allow_legacy=False: merging a pre-fingerprint entry into this
|
|
1587
|
+
# write would fuse two prompt vintages inside a single entry and
|
|
1588
|
+
# then stamp the result as current-vintage — the exact mixing
|
|
1589
|
+
# #1939 is about, made unfixable because the entry now claims a
|
|
1590
|
+
# prompt that only produced half of it.
|
|
1591
|
+
# allow_partial=True: a file split into slices across chunks
|
|
1592
|
+
# accumulates here; if an earlier slice truncated, keep its nodes
|
|
1593
|
+
# in the union AND let the entry stay partial (the _partial
|
|
1594
|
+
# markers ride through, so is_partial below re-detects it) rather
|
|
1595
|
+
# than a later clean slice silently replacing it and promoting the
|
|
1596
|
+
# half-file to complete.
|
|
1597
|
+
prev = load_cached(cache_path, root, kind=kind, cache_root=cache_root,
|
|
1598
|
+
prompt=prompt, prompt_file=prompt_file,
|
|
1599
|
+
allow_legacy=False, allow_partial=True)
|
|
1600
|
+
_prev_partial = bool(prev.get("partial")) if prev else False
|
|
1601
|
+
if prev:
|
|
1602
|
+
result = {
|
|
1603
|
+
"nodes": (prev.get("nodes", []) or []) + result["nodes"],
|
|
1604
|
+
"edges": (prev.get("edges", []) or []) + result["edges"],
|
|
1605
|
+
"hyperedges": (prev.get("hyperedges", []) or []) + result["hyperedges"],
|
|
1606
|
+
}
|
|
1607
|
+
else:
|
|
1608
|
+
_prev_partial = False
|
|
1609
|
+
# A file is partial if the caller named it, any of its grouped items
|
|
1610
|
+
# carries the intrinsic ``_partial`` marker, OR the entry it merged
|
|
1611
|
+
# onto was already partial (an empty-parse truncation leaves a
|
|
1612
|
+
# ``partial: True`` entry with no item markers, so a later clean slice
|
|
1613
|
+
# merging over it must NOT silently promote the half-file to complete
|
|
1614
|
+
# — #1950). Copy so the caller's dict is never mutated. A genuine
|
|
1615
|
+
# complete re-extraction (merge_existing=False) overwrites the
|
|
1616
|
+
# content-hash key with a non-partial entry that then serves normally.
|
|
1617
|
+
is_partial = (
|
|
1618
|
+
(partial_paths is not None and cache_path in partial_paths)
|
|
1619
|
+
or _group_has_partial_marker(result)
|
|
1620
|
+
or _prev_partial
|
|
1621
|
+
)
|
|
1622
|
+
if is_partial:
|
|
1623
|
+
result = {**result, "partial": True}
|
|
1624
|
+
# A semantic extraction with zero nodes and zero hyperedges is not a valid
|
|
1625
|
+
# standalone extraction (#2927): edge-only or empty results must not be
|
|
1626
|
+
# cached, so that subsequent runs can re-dispatch and retry the file (#933/#1666).
|
|
1627
|
+
if not is_partial and not (result.get("nodes") or result.get("hyperedges")):
|
|
1628
|
+
continue
|
|
1629
|
+
save_cached(cache_path, result, root, kind=kind, cache_root=cache_root,
|
|
1630
|
+
prompt=prompt, prompt_file=prompt_file)
|
|
1631
|
+
saved += 1
|
|
1632
|
+
else:
|
|
1633
|
+
skipped_not_file += 1
|
|
1634
|
+
if skipped_not_file and skipped_not_file == len(by_file):
|
|
1635
|
+
warnings.warn(
|
|
1636
|
+
f"save_semantic_cache: all {skipped_not_file} source_file group(s) were "
|
|
1637
|
+
"skipped because their paths do not resolve to real files. This usually "
|
|
1638
|
+
"means ``root`` is anchored to the wrong directory (e.g. the --out "
|
|
1639
|
+
"directory instead of the corpus root). Pass the corpus directory as "
|
|
1640
|
+
"``root`` and the output directory as ``cache_root`` (#1991).",
|
|
1641
|
+
RuntimeWarning,
|
|
1642
|
+
stacklevel=2,
|
|
1643
|
+
)
|
|
1644
|
+
return saved
|
|
1645
|
+
|
|
1646
|
+
|
|
1647
|
+
def scope_semantic_result(
|
|
1648
|
+
result: dict,
|
|
1649
|
+
root: Path = Path("."),
|
|
1650
|
+
allowed_source_files: "Iterable[str | Path] | None" = None,
|
|
1651
|
+
) -> tuple[set[str], int]:
|
|
1652
|
+
"""Scope an extraction result in place to the files actually dispatched (#2926).
|
|
1653
|
+
|
|
1654
|
+
Graph-side mirror of the ``allowed_source_files`` write-guard in
|
|
1655
|
+
:func:`save_semantic_cache` (#1757). A model can attribute stray
|
|
1656
|
+
nodes/edges to a corpus file that was not part of the current extraction
|
|
1657
|
+
batch; :func:`build_merge` derives its replace-set from the source_files
|
|
1658
|
+
present in the new chunks, so such a stray fragment would REPLACE that
|
|
1659
|
+
file's entire prior contribution in graph.json — while its manifest entry
|
|
1660
|
+
still says unchanged, so no later incremental run re-dispatches it and the
|
|
1661
|
+
loss is permanent until a full rebuild.
|
|
1662
|
+
|
|
1663
|
+
Items whose ``source_file`` resolves outside ``allowed_source_files`` are
|
|
1664
|
+
dropped from ``result``'s ``nodes`` / ``edges`` / ``hyperedges`` lists
|
|
1665
|
+
(mutated in place); items without a ``source_file`` pass through. An edge
|
|
1666
|
+
or hyperedge that survives the scope filter but references a dropped node
|
|
1667
|
+
id is dropped too (#1916 mirror), unless that id is also defined by a kept
|
|
1668
|
+
node (duplicate attribution must not be over-pruned).
|
|
1669
|
+
|
|
1670
|
+
Path matching shares :func:`_semantic_source_matcher` with
|
|
1671
|
+
:func:`save_semantic_cache` (relative against ``root``, walked-path
|
|
1672
|
+
identity), so an item this function keeps can never still hit the save's
|
|
1673
|
+
out-of-scope skip, and vice versa.
|
|
1674
|
+
|
|
1675
|
+
Returns ``(dropped_source_files, dropped_item_count)`` for logging;
|
|
1676
|
+
``dropped_source_files`` holds the normalized ``source_file`` strings of
|
|
1677
|
+
every group that had at least one item removed.
|
|
1678
|
+
"""
|
|
1679
|
+
if allowed_source_files is None:
|
|
1680
|
+
return set(), 0
|
|
1681
|
+
|
|
1682
|
+
source_identity, normalize_value = _semantic_source_matcher(root)
|
|
1683
|
+
|
|
1684
|
+
def _item_identity(item: dict) -> tuple[str | None, Path | None]:
|
|
1685
|
+
"""(display form, walked identity) of an item's source_file."""
|
|
1686
|
+
src = item.get("source_file")
|
|
1687
|
+
if not src:
|
|
1688
|
+
return None, None
|
|
1689
|
+
norm = normalize_value(src)
|
|
1690
|
+
return norm, source_identity(norm)
|
|
1691
|
+
|
|
1692
|
+
allowed_paths = {source_identity(str(path)) for path in allowed_source_files}
|
|
1693
|
+
|
|
1694
|
+
def _hashable(value) -> bool:
|
|
1695
|
+
try:
|
|
1696
|
+
hash(value)
|
|
1697
|
+
except TypeError:
|
|
1698
|
+
return False
|
|
1699
|
+
return True
|
|
1700
|
+
|
|
1701
|
+
dropped_files: set[str] = set()
|
|
1702
|
+
dropped_items = 0
|
|
1703
|
+
dropped_ids: set = set()
|
|
1704
|
+
kept_ids: set = set()
|
|
1705
|
+
for bucket in ("nodes", "edges", "hyperedges"):
|
|
1706
|
+
kept: list[dict] = []
|
|
1707
|
+
for item in result.get(bucket) or []:
|
|
1708
|
+
display, ident = _item_identity(item)
|
|
1709
|
+
if ident is not None and ident not in allowed_paths:
|
|
1710
|
+
dropped_files.add(display)
|
|
1711
|
+
dropped_items += 1
|
|
1712
|
+
if bucket == "nodes" and item.get("id") is not None:
|
|
1713
|
+
nid = item["id"]
|
|
1714
|
+
if _hashable(nid):
|
|
1715
|
+
dropped_ids.add(nid)
|
|
1716
|
+
continue
|
|
1717
|
+
if bucket == "nodes" and item.get("id") is not None and _hashable(item["id"]):
|
|
1718
|
+
kept_ids.add(item["id"])
|
|
1719
|
+
kept.append(item)
|
|
1720
|
+
result[bucket] = kept
|
|
1721
|
+
|
|
1722
|
+
# A duplicate-attribution node (defined in a dropped AND a kept group)
|
|
1723
|
+
# survives the filter — don't prune references to it.
|
|
1724
|
+
dropped_ids -= kept_ids
|
|
1725
|
+
if dropped_ids:
|
|
1726
|
+
|
|
1727
|
+
def edge_dangles(e: dict) -> bool:
|
|
1728
|
+
try:
|
|
1729
|
+
return e.get("source") in dropped_ids or e.get("target") in dropped_ids
|
|
1730
|
+
except TypeError:
|
|
1731
|
+
# Non-hashable endpoint from an untrusted result; leave it
|
|
1732
|
+
# to build-time validation rather than fail here.
|
|
1733
|
+
return False
|
|
1734
|
+
|
|
1735
|
+
def hyperedge_dangles(h: dict) -> bool:
|
|
1736
|
+
try:
|
|
1737
|
+
return bool(dropped_ids & set(h.get("nodes") or []))
|
|
1738
|
+
except TypeError:
|
|
1739
|
+
return False
|
|
1740
|
+
|
|
1741
|
+
result["edges"] = [e for e in result.get("edges") or [] if not edge_dangles(e)]
|
|
1742
|
+
result["hyperedges"] = [
|
|
1743
|
+
h for h in result.get("hyperedges") or [] if not hyperedge_dangles(h)
|
|
1744
|
+
]
|
|
1745
|
+
|
|
1746
|
+
return dropped_files, dropped_items
|