homegraph 1.5.0 → 1.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -21
- package/README.md +305 -305
- package/dist/bin/command-supervision.d.ts.map +1 -1
- package/dist/bin/command-supervision.js +7 -4
- package/dist/bin/command-supervision.js.map +1 -1
- package/dist/bin/homegraph.js +9 -9
- package/dist/db/index.js +36 -36
- package/dist/db/migrations.js +37 -37
- package/dist/db/queries.js +156 -156
- package/dist/db/schema.sql +203 -203
- package/dist/directory.js +5 -5
- package/dist/extraction/languages/arkts-viewtree.d.ts +2 -4
- package/dist/extraction/languages/arkts-viewtree.d.ts.map +1 -1
- package/dist/extraction/languages/arkts-viewtree.js +6 -21
- package/dist/extraction/languages/arkts-viewtree.js.map +1 -1
- package/dist/extraction/languages/arkts.d.ts +16 -6
- package/dist/extraction/languages/arkts.d.ts.map +1 -1
- package/dist/extraction/languages/arkts.js +174 -21
- package/dist/extraction/languages/arkts.js.map +1 -1
- package/dist/extraction/wasm/tree-sitter-c_sharp.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-cfml.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-cfquery.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-cfscript.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-cobol.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-erlang.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-nix.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-pascal.wasm +0 -0
- package/dist/extraction/wasm/tree-sitter-vbnet.wasm +0 -0
- package/dist/installer/instructions-template.js +9 -9
- package/dist/mcp/liveness-watchdog.d.ts +18 -0
- package/dist/mcp/liveness-watchdog.d.ts.map +1 -1
- package/dist/mcp/liveness-watchdog.js +185 -73
- package/dist/mcp/liveness-watchdog.js.map +1 -1
- package/dist/mcp/server-instructions.js +47 -47
- package/dist/reasoning/reasoner.js +32 -32
- package/dist/spec/db/commit-node.js +4 -4
- package/dist/spec/db/fragment-node.js +10 -10
- package/dist/spec/db/fts.js +8 -8
- package/dist/spec/db/relations.js +88 -88
- package/dist/spec/db/schema.js +6 -6
- package/dist/spec/db/schema.sql +121 -121
- package/dist/spec/db/spec-node.js +4 -4
- package/dist/spec/llm/prompts.js +53 -53
- package/dist/spec/utils.d.ts +16 -4
- package/dist/spec/utils.d.ts.map +1 -1
- package/dist/spec/utils.js +58 -6
- package/dist/spec/utils.js.map +1 -1
- package/package.json +62 -62
- package/scripts/_tmp-cfwk-resolve.log +0 -0
- package/scripts/_tmp-cfwk-sig.log +0 -0
- package/scripts/_tmp-cfwk-vt.log +0 -0
- package/scripts/add-lang/bench.sh +60 -60
- package/scripts/add-lang/check-grammar.mjs +75 -75
- package/scripts/add-lang/dump-ast.mjs +103 -103
- package/scripts/add-lang/verify-extraction.mjs +70 -70
- package/scripts/agent-eval/ab-adoption.sh +91 -91
- package/scripts/agent-eval/ab-hook.sh +86 -86
- package/scripts/agent-eval/ab-impl.sh +78 -78
- package/scripts/agent-eval/ab-new-vs-baseline.sh +102 -102
- package/scripts/agent-eval/ab-sufficiency.sh +78 -78
- package/scripts/agent-eval/arms-F.sh +21 -21
- package/scripts/agent-eval/arms-matrix.sh +37 -37
- package/scripts/agent-eval/audit.sh +68 -68
- package/scripts/agent-eval/bench-readme.sh +28 -28
- package/scripts/agent-eval/bench-why-repo.sh +22 -22
- package/scripts/agent-eval/block-read-hook.sh +19 -19
- package/scripts/agent-eval/hook-settings.json +15 -15
- package/scripts/agent-eval/itrun.sh +120 -120
- package/scripts/agent-eval/offload-eval-3arm.sh +72 -72
- package/scripts/agent-eval/offload-eval-cost.mjs +133 -133
- package/scripts/agent-eval/offload-eval-effort.mjs +108 -108
- package/scripts/agent-eval/offload-eval-frontload-matrix.sh +25 -25
- package/scripts/agent-eval/offload-eval-frontload.sh +47 -47
- package/scripts/agent-eval/offload-eval-ground-truth.json +18 -18
- package/scripts/agent-eval/offload-eval-hook.mjs +84 -84
- package/scripts/agent-eval/offload-eval-judge.mjs +103 -103
- package/scripts/agent-eval/offload-eval-matrix.sh +20 -20
- package/scripts/agent-eval/offload-eval-metrics.mjs +94 -94
- package/scripts/agent-eval/offload-eval-refs1.sh +50 -50
- package/scripts/agent-eval/offload-eval-setup.sh +24 -24
- package/scripts/agent-eval/offload-eval-styles.sh +71 -71
- package/scripts/agent-eval/offload-eval-summarize.mjs +68 -68
- package/scripts/agent-eval/offload-eval.md +76 -76
- package/scripts/agent-eval/parse-arms.mjs +116 -116
- package/scripts/agent-eval/parse-bench-readme.mjs +84 -84
- package/scripts/agent-eval/parse-run.mjs +45 -45
- package/scripts/agent-eval/parse-session.mjs +93 -93
- package/scripts/agent-eval/probe-context.mjs +21 -21
- package/scripts/agent-eval/probe-explore.mjs +40 -40
- package/scripts/agent-eval/probe-node.mjs +20 -20
- package/scripts/agent-eval/probe-sweep.mjs +119 -119
- package/scripts/agent-eval/probe-trace.mjs +20 -20
- package/scripts/agent-eval/redirect-read-hook.sh +38 -38
- package/scripts/agent-eval/repro-concurrent-explore.mjs +119 -119
- package/scripts/agent-eval/repro-daemon-clients.mjs +125 -125
- package/scripts/agent-eval/run-agent.sh +34 -34
- package/scripts/agent-eval/run-all.sh +75 -75
- package/scripts/agent-eval/run-arms.sh +56 -56
- package/scripts/agent-eval/seq-matrix.mjs +137 -137
- package/scripts/bench-arkts-init-rss.log +0 -0
- package/scripts/build-bundle.sh +123 -123
- package/scripts/exp_boundary_eval/README.md +247 -247
- package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-310.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-310.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/deveco_arm.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-310.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-310.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-310.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/setup.cpython-310.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/setup.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/__pycache__/win_mcp_launcher.cpython-38.pyc +0 -0
- package/scripts/exp_boundary_eval/_test_mcp_chain.py +78 -78
- package/scripts/exp_boundary_eval/_test_stdin.py +8 -8
- package/scripts/exp_boundary_eval/_utils.py +1116 -1116
- package/scripts/exp_boundary_eval/analyze.py +1313 -1313
- package/scripts/exp_boundary_eval/deveco_arm.py +519 -519
- package/scripts/exp_boundary_eval/run_all.py +378 -378
- package/scripts/exp_boundary_eval/run_one.py +165 -165
- package/scripts/exp_boundary_eval/run_session.py +158 -158
- package/scripts/exp_boundary_eval/setup.py +120 -120
- package/scripts/exp_boundary_eval/win_mcp_launcher.py +73 -73
- package/scripts/exp_boundary_eval/win_mcp_stdio_wrap.js +36 -36
- package/scripts/exp_boundary_eval/win_node_launcher.py +24 -24
- package/scripts/extract-release-notes.mjs +130 -130
- package/scripts/local-install.sh +41 -41
- package/scripts/npm-sdk.js +75 -75
- package/scripts/npm-shim.js +275 -275
- package/scripts/ohos-sdk-publish.mjs +133 -133
- package/scripts/pack-npm.sh +119 -119
- package/scripts/prepare-release.mjs +270 -270
- package/scripts/probe-arkts-mem-why-run.log +0 -0
- package/scripts/probe-banner-livecard-ir.log +0 -0
- package/scripts/probe-cfwk-ast.stderr.log +0 -0
- package/scripts/probe-cfwk-ast.stdout.log +0 -0
- package/scripts/probe-cfwk-attr-shape.log +0 -0
- package/scripts/probe-cfwk-cfgdump.stderr.log +0 -0
- package/scripts/probe-cfwk-cfgdump.stdout.log +0 -0
- package/scripts/probe-cfwk-cvc.stderr.log +0 -0
- package/scripts/probe-cfwk-cvc.stdout.log +0 -0
- package/scripts/probe-cfwk-diag2.log +0 -0
- package/scripts/probe-cfwk-diag3.log +0 -0
- package/scripts/probe-cfwk-fedbg.stderr.log +0 -0
- package/scripts/probe-cfwk-fedbg.stdout.log +0 -0
- package/scripts/probe-cfwk-fileresult.log +0 -0
- package/scripts/probe-cfwk-fix.stderr.log +0 -0
- package/scripts/probe-cfwk-fix.stdout.log +0 -0
- package/scripts/probe-cfwk-fix2.stderr.log +0 -0
- package/scripts/probe-cfwk-fix2.stdout.log +0 -0
- package/scripts/probe-cfwk-fix3.stderr.log +0 -0
- package/scripts/probe-cfwk-fix3.stdout.log +0 -0
- package/scripts/probe-cfwk-foreach.log +0 -0
- package/scripts/probe-cfwk-getmethod-throw.log +0 -0
- package/scripts/probe-cfwk-hg-extract.log +0 -0
- package/scripts/probe-cfwk-pr1003-noprior.stderr.log +0 -0
- package/scripts/probe-cfwk-pr1003-noprior.stdout.log +0 -0
- package/scripts/probe-cfwk-pr1003.stderr.log +0 -0
- package/scripts/probe-cfwk-pr1003.stdout.log +0 -0
- package/scripts/probe-cfwk-preroot-noprior.stderr.log +0 -0
- package/scripts/probe-cfwk-preroot-noprior.stdout.log +0 -0
- package/scripts/probe-cfwk-preroot-prior.stderr.log +0 -0
- package/scripts/probe-cfwk-preroot-prior.stdout.log +0 -0
- package/scripts/probe-cfwk-preroot-skipstate.stderr.log +0 -0
- package/scripts/probe-cfwk-preroot-skipstate.stdout.log +0 -0
- package/scripts/probe-cfwk-resolve-sim.log +0 -0
- package/scripts/probe-cfwk-tree-shape.log +0 -0
- package/scripts/probe-cfwk-walk-abort.log +0 -0
- package/scripts/probe-cfwk.log +0 -0
- package/scripts/probe-force-index.log +0 -0
- package/scripts/probe-no-force-index.log +0 -0
- package/scripts/probe-samefile-ir.stderr.log +0 -0
- package/scripts/probe-samefile-ir.stdout.log +224 -0
- package/scripts/probe-sdk-vs-project.log +0 -0
- package/scripts/probe-viewtree-downgrade.log +0 -0
- package/dist/arkts/ohos-api-index.d.ts +0 -15
- package/dist/arkts/ohos-api-index.d.ts.map +0 -1
- package/dist/arkts/ohos-api-index.js +0 -190
- package/dist/arkts/ohos-api-index.js.map +0 -1
- package/dist/arkts/ohos-sdk-input.d.ts +0 -36
- package/dist/arkts/ohos-sdk-input.d.ts.map +0 -1
- package/dist/arkts/ohos-sdk-input.js +0 -214
- package/dist/arkts/ohos-sdk-input.js.map +0 -1
- package/dist/extraction/languages/arkts-state-decorators.d.ts +0 -13
- package/dist/extraction/languages/arkts-state-decorators.d.ts.map +0 -1
- package/dist/extraction/languages/arkts-state-decorators.js +0 -26
- package/dist/extraction/languages/arkts-state-decorators.js.map +0 -1
- package/dist/extraction/languages/ohos-api-consumer.d.ts +0 -34
- package/dist/extraction/languages/ohos-api-consumer.d.ts.map +0 -1
- package/dist/extraction/languages/ohos-api-consumer.js +0 -283
- package/dist/extraction/languages/ohos-api-consumer.js.map +0 -1
- package/dist/spec/build/git-scanner.d.ts +0 -93
- package/dist/spec/build/git-scanner.d.ts.map +0 -1
- package/dist/spec/build/git-scanner.js +0 -254
- package/dist/spec/build/git-scanner.js.map +0 -1
- package/dist/spec/git-utils.d.ts +0 -8
- package/dist/spec/git-utils.d.ts.map +0 -1
- package/dist/spec/git-utils.js +0 -14
- package/dist/spec/git-utils.js.map +0 -1
- package/dist/spec/mine/clusterer.d.ts +0 -63
- package/dist/spec/mine/clusterer.d.ts.map +0 -1
- package/dist/spec/mine/clusterer.js +0 -904
- package/dist/spec/mine/clusterer.js.map +0 -1
- package/dist/spec/mine/progress-handler.d.ts +0 -22
- package/dist/spec/mine/progress-handler.d.ts.map +0 -1
- package/dist/spec/mine/progress-handler.js +0 -108
- package/dist/spec/mine/progress-handler.js.map +0 -1
- package/dist/spec/mine/progress.d.ts +0 -23
- package/dist/spec/mine/progress.d.ts.map +0 -1
- package/dist/spec/mine/progress.js +0 -12
- package/dist/spec/mine/progress.js.map +0 -1
|
@@ -1,28 +1,28 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# Re-run the README "Benchmark Results" A/B (with vs without homegraph) on the
|
|
3
|
-
# current build: the 7 README repos, same queries, RUNS per arm (default 4).
|
|
4
|
-
# Output → /tmp/ab-readme/<repo>/run<n>/run-headless-{with,without}.jsonl
|
|
5
|
-
# Aggregate with parse-bench-readme.mjs. Repos must be cloned + indexed under
|
|
6
|
-
# $CORPUS (default /tmp/homegraph-corpus) by the build under test.
|
|
7
|
-
set -uo pipefail
|
|
8
|
-
H="$(cd "$(dirname "$0")" && pwd)"
|
|
9
|
-
C="${CORPUS:-/tmp/homegraph-corpus}"
|
|
10
|
-
RUNS="${RUNS:-4}"
|
|
11
|
-
ROWS=(
|
|
12
|
-
"vscode|How does the extension host communicate with the main process?"
|
|
13
|
-
"excalidraw|How does Excalidraw render and update canvas elements?"
|
|
14
|
-
"django|How does Django's ORM build and execute a query from a QuerySet?"
|
|
15
|
-
"tokio|How does tokio schedule and run async tasks on its runtime?"
|
|
16
|
-
"okhttp|How does OkHttp process a request through its interceptor chain?"
|
|
17
|
-
"gin|How does gin route requests through its middleware chain?"
|
|
18
|
-
"alamofire|How does Alamofire build, send, and validate a request?"
|
|
19
|
-
)
|
|
20
|
-
echo "### README A/B START $(date) RUNS=$RUNS"
|
|
21
|
-
for row in "${ROWS[@]}"; do
|
|
22
|
-
repo="${row%%|*}"; q="${row#*|}"
|
|
23
|
-
echo "===== $repo ====="
|
|
24
|
-
for run in $(seq 1 "$RUNS"); do
|
|
25
|
-
AGENT_EVAL_OUT="/tmp/ab-readme/$repo/run$run" bash "$H/run-all.sh" "$C/$repo" "$q" headless 2>&1 | grep -E "exit [0-9]" || echo " run$run: (no exit line)"
|
|
26
|
-
done
|
|
27
|
-
done
|
|
28
|
-
echo "### README A/B DONE $(date)"
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Re-run the README "Benchmark Results" A/B (with vs without homegraph) on the
|
|
3
|
+
# current build: the 7 README repos, same queries, RUNS per arm (default 4).
|
|
4
|
+
# Output → /tmp/ab-readme/<repo>/run<n>/run-headless-{with,without}.jsonl
|
|
5
|
+
# Aggregate with parse-bench-readme.mjs. Repos must be cloned + indexed under
|
|
6
|
+
# $CORPUS (default /tmp/homegraph-corpus) by the build under test.
|
|
7
|
+
set -uo pipefail
|
|
8
|
+
H="$(cd "$(dirname "$0")" && pwd)"
|
|
9
|
+
C="${CORPUS:-/tmp/homegraph-corpus}"
|
|
10
|
+
RUNS="${RUNS:-4}"
|
|
11
|
+
ROWS=(
|
|
12
|
+
"vscode|How does the extension host communicate with the main process?"
|
|
13
|
+
"excalidraw|How does Excalidraw render and update canvas elements?"
|
|
14
|
+
"django|How does Django's ORM build and execute a query from a QuerySet?"
|
|
15
|
+
"tokio|How does tokio schedule and run async tasks on its runtime?"
|
|
16
|
+
"okhttp|How does OkHttp process a request through its interceptor chain?"
|
|
17
|
+
"gin|How does gin route requests through its middleware chain?"
|
|
18
|
+
"alamofire|How does Alamofire build, send, and validate a request?"
|
|
19
|
+
)
|
|
20
|
+
echo "### README A/B START $(date) RUNS=$RUNS"
|
|
21
|
+
for row in "${ROWS[@]}"; do
|
|
22
|
+
repo="${row%%|*}"; q="${row#*|}"
|
|
23
|
+
echo "===== $repo ====="
|
|
24
|
+
for run in $(seq 1 "$RUNS"); do
|
|
25
|
+
AGENT_EVAL_OUT="/tmp/ab-readme/$repo/run$run" bash "$H/run-all.sh" "$C/$repo" "$q" headless 2>&1 | grep -E "exit [0-9]" || echo " run$run: (no exit line)"
|
|
26
|
+
done
|
|
27
|
+
done
|
|
28
|
+
echo "### README A/B DONE $(date)"
|
|
@@ -1,22 +1,22 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# One README repo, WITH-homegraph only, N runs. Each run appends a why-Read
|
|
3
|
-
# diagnostic so the agent explains any Read/Grep. (The WITHOUT baseline is
|
|
4
|
-
# homegraph-independent and already in the README — no point re-running it.)
|
|
5
|
-
# Output -> /tmp/ab-why/<repo>/with<n>.jsonl
|
|
6
|
-
# Usage: bench-why-repo.sh <repo-path> "<query>" [N]
|
|
7
|
-
set -uo pipefail
|
|
8
|
-
REPO="$1"; Q="$2"; N="${3:-4}"
|
|
9
|
-
NAME="$(basename "$REPO")"
|
|
10
|
-
CG="/Users/colby/Development/Personal/homegraph/dist/bin/homegraph.js"
|
|
11
|
-
OUT="/tmp/ab-why/$NAME"; mkdir -p "$OUT"
|
|
12
|
-
WHY=$'\n\nIMPORTANT — diagnostic: if you use the Read or Grep tool at ANY point, for EACH such call explain why homegraph_explore / homegraph_node did not already give you what you needed. End your entire answer with a section titled exactly "## Why I read" listing every Read and Grep you made and the precise reason homegraph fell short for it. If you used neither, write "## Why I read" then "none — homegraph was sufficient."'
|
|
13
|
-
printf '{"mcpServers":{"homegraph":{"command":"%s","args":["serve","--mcp","--path","%s"]}}}' "$CG" "$REPO" > "$OUT/cg.json"
|
|
14
|
-
|
|
15
|
-
for i in $(seq 1 "$N"); do
|
|
16
|
-
pkill -f "serve --mcp" 2>/dev/null; sleep 1; rm -f "$REPO/.homegraph/daemon.sock"
|
|
17
|
-
( cd "$REPO" && claude -p "$Q$WHY" --output-format stream-json --verbose \
|
|
18
|
-
--permission-mode bypassPermissions --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
|
|
19
|
-
--strict-mcp-config --mcp-config "$OUT/cg.json" > "$OUT/with$i.jsonl" 2>"$OUT/with$i.err" )
|
|
20
|
-
echo "WITH run $i: exit $? ($(wc -l < "$OUT/with$i.jsonl" | tr -d ' ') lines)"
|
|
21
|
-
done
|
|
22
|
-
echo "DONE $NAME"
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# One README repo, WITH-homegraph only, N runs. Each run appends a why-Read
|
|
3
|
+
# diagnostic so the agent explains any Read/Grep. (The WITHOUT baseline is
|
|
4
|
+
# homegraph-independent and already in the README — no point re-running it.)
|
|
5
|
+
# Output -> /tmp/ab-why/<repo>/with<n>.jsonl
|
|
6
|
+
# Usage: bench-why-repo.sh <repo-path> "<query>" [N]
|
|
7
|
+
set -uo pipefail
|
|
8
|
+
REPO="$1"; Q="$2"; N="${3:-4}"
|
|
9
|
+
NAME="$(basename "$REPO")"
|
|
10
|
+
CG="/Users/colby/Development/Personal/homegraph/dist/bin/homegraph.js"
|
|
11
|
+
OUT="/tmp/ab-why/$NAME"; mkdir -p "$OUT"
|
|
12
|
+
WHY=$'\n\nIMPORTANT — diagnostic: if you use the Read or Grep tool at ANY point, for EACH such call explain why homegraph_explore / homegraph_node did not already give you what you needed. End your entire answer with a section titled exactly "## Why I read" listing every Read and Grep you made and the precise reason homegraph fell short for it. If you used neither, write "## Why I read" then "none — homegraph was sufficient."'
|
|
13
|
+
printf '{"mcpServers":{"homegraph":{"command":"%s","args":["serve","--mcp","--path","%s"]}}}' "$CG" "$REPO" > "$OUT/cg.json"
|
|
14
|
+
|
|
15
|
+
for i in $(seq 1 "$N"); do
|
|
16
|
+
pkill -f "serve --mcp" 2>/dev/null; sleep 1; rm -f "$REPO/.homegraph/daemon.sock"
|
|
17
|
+
( cd "$REPO" && claude -p "$Q$WHY" --output-format stream-json --verbose \
|
|
18
|
+
--permission-mode bypassPermissions --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
|
|
19
|
+
--strict-mcp-config --mcp-config "$OUT/cg.json" > "$OUT/with$i.jsonl" 2>"$OUT/with$i.err" )
|
|
20
|
+
echo "WITH run $i: exit $? ($(wc -l < "$OUT/with$i.jsonl" | tr -d ' ') lines)"
|
|
21
|
+
done
|
|
22
|
+
echo "DONE $NAME"
|
|
@@ -1,19 +1,19 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# PreToolUse hook (experiment): deny Read of homegraph-indexed source files and
|
|
3
|
-
# steer the agent to homegraph_explore/homegraph_node instead. Tests whether
|
|
4
|
-
# homegraph can FULLY replace Read for code-understanding once the escape hatch
|
|
5
|
-
# is removed. Non-source reads (config, .env, markdown, new files) pass through.
|
|
6
|
-
#
|
|
7
|
-
# Wire via: claude ... --settings scripts/agent-eval/hook-settings.json
|
|
8
|
-
set -uo pipefail
|
|
9
|
-
input="$(cat)"
|
|
10
|
-
fp="$(printf '%s' "$input" | jq -r '.tool_input.file_path // empty' 2>/dev/null)"
|
|
11
|
-
|
|
12
|
-
case "$fp" in
|
|
13
|
-
*.ts|*.tsx|*.js|*.jsx|*.mjs|*.cjs|*.py|*.go|*.rs|*.java|*.rb|*.php|*.swift|*.kt|*.kts|*.c|*.cc|*.cpp|*.h|*.hpp|*.cs|*.lua|*.vue|*.svelte)
|
|
14
|
-
msg="Read is disabled for source files in this session — homegraph already has this file indexed (with line numbers, kept in sync on every change). Use homegraph_explore (several related symbols at once) or homegraph_node (one symbol's full source). If a symbol you need wasn't in a prior explore, run ANOTHER homegraph_explore with its exact name instead of reading the file."
|
|
15
|
-
jq -n --arg m "$msg" '{reason:$m, hookSpecificOutput:{hookEventName:"PreToolUse",permissionDecision:"deny",permissionDecisionReason:$m}}'
|
|
16
|
-
exit 0
|
|
17
|
-
;;
|
|
18
|
-
esac
|
|
19
|
-
exit 0
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# PreToolUse hook (experiment): deny Read of homegraph-indexed source files and
|
|
3
|
+
# steer the agent to homegraph_explore/homegraph_node instead. Tests whether
|
|
4
|
+
# homegraph can FULLY replace Read for code-understanding once the escape hatch
|
|
5
|
+
# is removed. Non-source reads (config, .env, markdown, new files) pass through.
|
|
6
|
+
#
|
|
7
|
+
# Wire via: claude ... --settings scripts/agent-eval/hook-settings.json
|
|
8
|
+
set -uo pipefail
|
|
9
|
+
input="$(cat)"
|
|
10
|
+
fp="$(printf '%s' "$input" | jq -r '.tool_input.file_path // empty' 2>/dev/null)"
|
|
11
|
+
|
|
12
|
+
case "$fp" in
|
|
13
|
+
*.ts|*.tsx|*.js|*.jsx|*.mjs|*.cjs|*.py|*.go|*.rs|*.java|*.rb|*.php|*.swift|*.kt|*.kts|*.c|*.cc|*.cpp|*.h|*.hpp|*.cs|*.lua|*.vue|*.svelte)
|
|
14
|
+
msg="Read is disabled for source files in this session — homegraph already has this file indexed (with line numbers, kept in sync on every change). Use homegraph_explore (several related symbols at once) or homegraph_node (one symbol's full source). If a symbol you need wasn't in a prior explore, run ANOTHER homegraph_explore with its exact name instead of reading the file."
|
|
15
|
+
jq -n --arg m "$msg" '{reason:$m, hookSpecificOutput:{hookEventName:"PreToolUse",permissionDecision:"deny",permissionDecisionReason:$m}}'
|
|
16
|
+
exit 0
|
|
17
|
+
;;
|
|
18
|
+
esac
|
|
19
|
+
exit 0
|
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
{
|
|
2
|
-
"hooks": {
|
|
3
|
-
"PreToolUse": [
|
|
4
|
-
{
|
|
5
|
-
"matcher": "Read",
|
|
6
|
-
"hooks": [
|
|
7
|
-
{
|
|
8
|
-
"type": "command",
|
|
9
|
-
"command": "bash /Users/colby/Development/Personal/homegraph/scripts/agent-eval/block-read-hook.sh"
|
|
10
|
-
}
|
|
11
|
-
]
|
|
12
|
-
}
|
|
13
|
-
]
|
|
14
|
-
}
|
|
15
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"hooks": {
|
|
3
|
+
"PreToolUse": [
|
|
4
|
+
{
|
|
5
|
+
"matcher": "Read",
|
|
6
|
+
"hooks": [
|
|
7
|
+
{
|
|
8
|
+
"type": "command",
|
|
9
|
+
"command": "bash /Users/colby/Development/Personal/homegraph/scripts/agent-eval/block-read-hook.sh"
|
|
10
|
+
}
|
|
11
|
+
]
|
|
12
|
+
}
|
|
13
|
+
]
|
|
14
|
+
}
|
|
15
|
+
}
|
|
@@ -1,120 +1,120 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# Drive an INTERACTIVE Claude Code session in tmux, send a prompt, wait for the
|
|
3
|
-
# agent to finish, then print the tool-call breakdown from the session logs.
|
|
4
|
-
#
|
|
5
|
-
# Why interactive (not `claude -p`): headless print-mode picks the
|
|
6
|
-
# general-purpose subagent, while real interactive sessions delegate to the
|
|
7
|
-
# Explore subagent (or drive homegraph from the main thread). Only the
|
|
8
|
-
# interactive TUI reproduces the behavior users actually see. (Idle-detection
|
|
9
|
-
# technique borrowed from devpit's WaitForIdle.)
|
|
10
|
-
#
|
|
11
|
-
# Usage: itrun.sh <repo-path> <label> "<prompt>"
|
|
12
|
-
# Output dir: $AGENT_EVAL_OUT (default /tmp/agent-eval)
|
|
13
|
-
# Requires: tmux 3.0+, a logged-in `claude` CLI, homegraph MCP configured.
|
|
14
|
-
set -uo pipefail
|
|
15
|
-
REPO="$1"; LABEL="$2"; PROMPT="$3"
|
|
16
|
-
SESSION="cgt_${LABEL}"
|
|
17
|
-
OUT_DIR="${AGENT_EVAL_OUT:-/tmp/agent-eval}"; mkdir -p "$OUT_DIR"
|
|
18
|
-
OUT="$OUT_DIR/itrun-${LABEL}.txt"
|
|
19
|
-
HERE="$(cd "$(dirname "$0")" && pwd)"
|
|
20
|
-
|
|
21
|
-
cap() { tmux capture-pane -p -t "$SESSION" -S -40; }
|
|
22
|
-
|
|
23
|
-
tmux kill-session -t "$SESSION" 2>/dev/null
|
|
24
|
-
|
|
25
|
-
# Wide pane so the TUI doesn't hard-wrap tool lines.
|
|
26
|
-
tmux new-session -d -s "$SESSION" -x 230 -y 60
|
|
27
|
-
tmux send-keys -t "$SESSION" "cd $REPO && claude --dangerously-skip-permissions ${CLAUDE_EXTRA_ARGS:-}" Enter
|
|
28
|
-
|
|
29
|
-
# Wait for the ❯ prompt (claude drew its UI), up to 60s. NOTE: ❯ appears on the
|
|
30
|
-
# welcome screen seconds before the input actually accepts keystrokes, so this is
|
|
31
|
-
# necessary but NOT sufficient — the type-and-verify loop below is what proves
|
|
32
|
-
# the input is live.
|
|
33
|
-
ready=0
|
|
34
|
-
for _ in $(seq 1 120); do
|
|
35
|
-
cap | grep -q "❯" && { ready=1; break; }
|
|
36
|
-
sleep 0.5
|
|
37
|
-
done
|
|
38
|
-
[ "$ready" = 1 ] || { echo "claude never drew its UI"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
|
|
39
|
-
|
|
40
|
-
# Accept the per-folder "Is this a project you trust?" dialog if it shows (first
|
|
41
|
-
# time claude opens a given repo). Option 1 ("Yes, I trust this folder") is
|
|
42
|
-
# pre-selected, so Enter accepts. This dialog also contains ❯, so it must be
|
|
43
|
-
# cleared before the type-and-verify loop or keystrokes land on the menu.
|
|
44
|
-
for _ in $(seq 1 20); do
|
|
45
|
-
cap | grep -q "trust this folder" || break
|
|
46
|
-
tmux send-keys -t "$SESSION" Enter
|
|
47
|
-
sleep 1
|
|
48
|
-
done
|
|
49
|
-
|
|
50
|
-
# Type-and-verify: send the prompt, confirm a distinctive chunk of it actually
|
|
51
|
-
# landed in the input box, retry if it didn't (handles the early-❯ race where
|
|
52
|
-
# the welcome screen shows the prompt glyph but MCP init is still eating keys).
|
|
53
|
-
needle="${PROMPT:0:24}"
|
|
54
|
-
typed=0
|
|
55
|
-
for _ in $(seq 1 30); do
|
|
56
|
-
tmux send-keys -l -t "$SESSION" "$PROMPT"
|
|
57
|
-
sleep 1
|
|
58
|
-
if cap | grep -Fq "$needle"; then typed=1; break; fi
|
|
59
|
-
# Clear whatever partial text may have landed, then retry.
|
|
60
|
-
tmux send-keys -t "$SESSION" C-u
|
|
61
|
-
sleep 1
|
|
62
|
-
done
|
|
63
|
-
[ "$typed" = 1 ] || { echo "prompt never landed in the input box"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
|
|
64
|
-
sleep 0.5
|
|
65
|
-
tmux send-keys -t "$SESSION" Enter
|
|
66
|
-
|
|
67
|
-
# Busy signals. The robust one is the spinner's elapsed-time-in-parens, which
|
|
68
|
-
# EVERY working state shows — both the pre-stream thinking phase
|
|
69
|
-
# "(8s · thinking with max effort)" and the streaming phase
|
|
70
|
-
# "(24s · ↑ 2.5k tokens · …)", and it survives the 32s→"1m 3s" rollover. We OR
|
|
71
|
-
# in the token arrows, "esc to interrupt", and "Initializing" as belt-and-braces
|
|
72
|
-
# (some TUI versions/states show one but not the others).
|
|
73
|
-
BUSY_RE='esc to interrupt|↓ [0-9]|↑ [0-9]|Initializing|\(([0-9]+m )?[0-9]+s ·'
|
|
74
|
-
|
|
75
|
-
# Wait for work to START (busy indicator appears), up to 60s. If it never starts,
|
|
76
|
-
# fail loudly rather than silently reporting an empty run.
|
|
77
|
-
started=0
|
|
78
|
-
for _ in $(seq 1 120); do
|
|
79
|
-
cap | grep -qE "$BUSY_RE" && { started=1; break; }
|
|
80
|
-
sleep 0.5
|
|
81
|
-
done
|
|
82
|
-
[ "$started" = 1 ] || { echo "agent never started working"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
|
|
83
|
-
|
|
84
|
-
# Poll for idle. CRITICAL: Opus 4.8 (extended thinking) renders NO spinner /
|
|
85
|
-
# "esc to interrupt" / timer while it STREAMS its final answer — those appear
|
|
86
|
-
# only during the thinking + tool-use phases ("✻ Marinating… (32s · ↓ 1.3k
|
|
87
|
-
# tokens · thinking with max effort)"). So BUSY_RE reads "not busy" for the whole
|
|
88
|
-
# 10-30s answer stream, and any short not-busy threshold kills the run mid-answer
|
|
89
|
-
# (the truncation bug). We therefore detect "done" by CONTENT STABILITY, not by a
|
|
90
|
-
# spinner string: while the agent streams, the captured pane changes every poll,
|
|
91
|
-
# so stability never accrues; it accrues only once the agent has finished and the
|
|
92
|
-
# static "✻ Brewed for 1m 9s" summary is all that is left. BUSY_RE still hard-
|
|
93
|
-
# resets stability (covers thinking/tool-use/live-timer, where text can briefly
|
|
94
|
-
# sit still). Need STABLE_NEEDED polls (~8s) of zero pane change + ❯ present.
|
|
95
|
-
# Content-stability is model-agnostic — it survives future spinner re-wordings.
|
|
96
|
-
STABLE_NEEDED=16
|
|
97
|
-
prev=""; stable=0
|
|
98
|
-
for _ in $(seq 1 2400); do # up to ~20 min
|
|
99
|
-
pane="$(cap)"
|
|
100
|
-
sig="$(printf '%s' "$pane" | tr -s '[:space:]' ' ')"
|
|
101
|
-
if printf '%s' "$pane" | grep -qE "$BUSY_RE"; then
|
|
102
|
-
stable=0 # thinking / tool use / live timer → busy
|
|
103
|
-
elif [ -n "$sig" ] && [ "$sig" = "$prev" ] && printf '%s' "$pane" | grep -q "❯"; then
|
|
104
|
-
stable=$((stable+1)); [ "$stable" -ge "$STABLE_NEEDED" ] && break
|
|
105
|
-
else
|
|
106
|
-
stable=0 # answer still streaming → pane changing
|
|
107
|
-
fi
|
|
108
|
-
prev="$sig"
|
|
109
|
-
sleep 0.5
|
|
110
|
-
done
|
|
111
|
-
sleep 1
|
|
112
|
-
|
|
113
|
-
tmux capture-pane -p -t "$SESSION" -S - > "$OUT"
|
|
114
|
-
echo "captured $(wc -l < "$OUT") lines -> $OUT"
|
|
115
|
-
grep -oE "Done \([^)]*\)|[A-Z][a-z]+ for ([0-9]+m )?[0-9]+s" "$OUT" | tail -1
|
|
116
|
-
grep -oE "[0-9.]+k?/[0-9.]+M" "$OUT" | tail -1 | sed 's/^/Context /'
|
|
117
|
-
tmux kill-session -t "$SESSION" 2>/dev/null
|
|
118
|
-
|
|
119
|
-
# Clean tool breakdown from the session logs (main + subagents).
|
|
120
|
-
node "$HERE/parse-session.mjs" "$REPO" 2>/dev/null || true
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Drive an INTERACTIVE Claude Code session in tmux, send a prompt, wait for the
|
|
3
|
+
# agent to finish, then print the tool-call breakdown from the session logs.
|
|
4
|
+
#
|
|
5
|
+
# Why interactive (not `claude -p`): headless print-mode picks the
|
|
6
|
+
# general-purpose subagent, while real interactive sessions delegate to the
|
|
7
|
+
# Explore subagent (or drive homegraph from the main thread). Only the
|
|
8
|
+
# interactive TUI reproduces the behavior users actually see. (Idle-detection
|
|
9
|
+
# technique borrowed from devpit's WaitForIdle.)
|
|
10
|
+
#
|
|
11
|
+
# Usage: itrun.sh <repo-path> <label> "<prompt>"
|
|
12
|
+
# Output dir: $AGENT_EVAL_OUT (default /tmp/agent-eval)
|
|
13
|
+
# Requires: tmux 3.0+, a logged-in `claude` CLI, homegraph MCP configured.
|
|
14
|
+
set -uo pipefail
|
|
15
|
+
REPO="$1"; LABEL="$2"; PROMPT="$3"
|
|
16
|
+
SESSION="cgt_${LABEL}"
|
|
17
|
+
OUT_DIR="${AGENT_EVAL_OUT:-/tmp/agent-eval}"; mkdir -p "$OUT_DIR"
|
|
18
|
+
OUT="$OUT_DIR/itrun-${LABEL}.txt"
|
|
19
|
+
HERE="$(cd "$(dirname "$0")" && pwd)"
|
|
20
|
+
|
|
21
|
+
cap() { tmux capture-pane -p -t "$SESSION" -S -40; }
|
|
22
|
+
|
|
23
|
+
tmux kill-session -t "$SESSION" 2>/dev/null
|
|
24
|
+
|
|
25
|
+
# Wide pane so the TUI doesn't hard-wrap tool lines.
|
|
26
|
+
tmux new-session -d -s "$SESSION" -x 230 -y 60
|
|
27
|
+
tmux send-keys -t "$SESSION" "cd $REPO && claude --dangerously-skip-permissions ${CLAUDE_EXTRA_ARGS:-}" Enter
|
|
28
|
+
|
|
29
|
+
# Wait for the ❯ prompt (claude drew its UI), up to 60s. NOTE: ❯ appears on the
|
|
30
|
+
# welcome screen seconds before the input actually accepts keystrokes, so this is
|
|
31
|
+
# necessary but NOT sufficient — the type-and-verify loop below is what proves
|
|
32
|
+
# the input is live.
|
|
33
|
+
ready=0
|
|
34
|
+
for _ in $(seq 1 120); do
|
|
35
|
+
cap | grep -q "❯" && { ready=1; break; }
|
|
36
|
+
sleep 0.5
|
|
37
|
+
done
|
|
38
|
+
[ "$ready" = 1 ] || { echo "claude never drew its UI"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
|
|
39
|
+
|
|
40
|
+
# Accept the per-folder "Is this a project you trust?" dialog if it shows (first
|
|
41
|
+
# time claude opens a given repo). Option 1 ("Yes, I trust this folder") is
|
|
42
|
+
# pre-selected, so Enter accepts. This dialog also contains ❯, so it must be
|
|
43
|
+
# cleared before the type-and-verify loop or keystrokes land on the menu.
|
|
44
|
+
for _ in $(seq 1 20); do
|
|
45
|
+
cap | grep -q "trust this folder" || break
|
|
46
|
+
tmux send-keys -t "$SESSION" Enter
|
|
47
|
+
sleep 1
|
|
48
|
+
done
|
|
49
|
+
|
|
50
|
+
# Type-and-verify: send the prompt, confirm a distinctive chunk of it actually
|
|
51
|
+
# landed in the input box, retry if it didn't (handles the early-❯ race where
|
|
52
|
+
# the welcome screen shows the prompt glyph but MCP init is still eating keys).
|
|
53
|
+
needle="${PROMPT:0:24}"
|
|
54
|
+
typed=0
|
|
55
|
+
for _ in $(seq 1 30); do
|
|
56
|
+
tmux send-keys -l -t "$SESSION" "$PROMPT"
|
|
57
|
+
sleep 1
|
|
58
|
+
if cap | grep -Fq "$needle"; then typed=1; break; fi
|
|
59
|
+
# Clear whatever partial text may have landed, then retry.
|
|
60
|
+
tmux send-keys -t "$SESSION" C-u
|
|
61
|
+
sleep 1
|
|
62
|
+
done
|
|
63
|
+
[ "$typed" = 1 ] || { echo "prompt never landed in the input box"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
|
|
64
|
+
sleep 0.5
|
|
65
|
+
tmux send-keys -t "$SESSION" Enter
|
|
66
|
+
|
|
67
|
+
# Busy signals. The robust one is the spinner's elapsed-time-in-parens, which
|
|
68
|
+
# EVERY working state shows — both the pre-stream thinking phase
|
|
69
|
+
# "(8s · thinking with max effort)" and the streaming phase
|
|
70
|
+
# "(24s · ↑ 2.5k tokens · …)", and it survives the 32s→"1m 3s" rollover. We OR
|
|
71
|
+
# in the token arrows, "esc to interrupt", and "Initializing" as belt-and-braces
|
|
72
|
+
# (some TUI versions/states show one but not the others).
|
|
73
|
+
BUSY_RE='esc to interrupt|↓ [0-9]|↑ [0-9]|Initializing|\(([0-9]+m )?[0-9]+s ·'
|
|
74
|
+
|
|
75
|
+
# Wait for work to START (busy indicator appears), up to 60s. If it never starts,
|
|
76
|
+
# fail loudly rather than silently reporting an empty run.
|
|
77
|
+
started=0
|
|
78
|
+
for _ in $(seq 1 120); do
|
|
79
|
+
cap | grep -qE "$BUSY_RE" && { started=1; break; }
|
|
80
|
+
sleep 0.5
|
|
81
|
+
done
|
|
82
|
+
[ "$started" = 1 ] || { echo "agent never started working"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
|
|
83
|
+
|
|
84
|
+
# Poll for idle. CRITICAL: Opus 4.8 (extended thinking) renders NO spinner /
|
|
85
|
+
# "esc to interrupt" / timer while it STREAMS its final answer — those appear
|
|
86
|
+
# only during the thinking + tool-use phases ("✻ Marinating… (32s · ↓ 1.3k
|
|
87
|
+
# tokens · thinking with max effort)"). So BUSY_RE reads "not busy" for the whole
|
|
88
|
+
# 10-30s answer stream, and any short not-busy threshold kills the run mid-answer
|
|
89
|
+
# (the truncation bug). We therefore detect "done" by CONTENT STABILITY, not by a
|
|
90
|
+
# spinner string: while the agent streams, the captured pane changes every poll,
|
|
91
|
+
# so stability never accrues; it accrues only once the agent has finished and the
|
|
92
|
+
# static "✻ Brewed for 1m 9s" summary is all that is left. BUSY_RE still hard-
|
|
93
|
+
# resets stability (covers thinking/tool-use/live-timer, where text can briefly
|
|
94
|
+
# sit still). Need STABLE_NEEDED polls (~8s) of zero pane change + ❯ present.
|
|
95
|
+
# Content-stability is model-agnostic — it survives future spinner re-wordings.
|
|
96
|
+
STABLE_NEEDED=16
|
|
97
|
+
prev=""; stable=0
|
|
98
|
+
for _ in $(seq 1 2400); do # up to ~20 min
|
|
99
|
+
pane="$(cap)"
|
|
100
|
+
sig="$(printf '%s' "$pane" | tr -s '[:space:]' ' ')"
|
|
101
|
+
if printf '%s' "$pane" | grep -qE "$BUSY_RE"; then
|
|
102
|
+
stable=0 # thinking / tool use / live timer → busy
|
|
103
|
+
elif [ -n "$sig" ] && [ "$sig" = "$prev" ] && printf '%s' "$pane" | grep -q "❯"; then
|
|
104
|
+
stable=$((stable+1)); [ "$stable" -ge "$STABLE_NEEDED" ] && break
|
|
105
|
+
else
|
|
106
|
+
stable=0 # answer still streaming → pane changing
|
|
107
|
+
fi
|
|
108
|
+
prev="$sig"
|
|
109
|
+
sleep 0.5
|
|
110
|
+
done
|
|
111
|
+
sleep 1
|
|
112
|
+
|
|
113
|
+
tmux capture-pane -p -t "$SESSION" -S - > "$OUT"
|
|
114
|
+
echo "captured $(wc -l < "$OUT") lines -> $OUT"
|
|
115
|
+
grep -oE "Done \([^)]*\)|[A-Z][a-z]+ for ([0-9]+m )?[0-9]+s" "$OUT" | tail -1
|
|
116
|
+
grep -oE "[0-9.]+k?/[0-9.]+M" "$OUT" | tail -1 | sed 's/^/Context /'
|
|
117
|
+
tmux kill-session -t "$SESSION" 2>/dev/null
|
|
118
|
+
|
|
119
|
+
# Clean tool breakdown from the session logs (main + subagents).
|
|
120
|
+
node "$HERE/parse-session.mjs" "$REPO" 2>/dev/null || true
|
|
@@ -1,72 +1,72 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# 3-arm offload eval for ONE indexed repo + ONE question, n reps each.
|
|
3
|
-
# ARM offload : homegraph attached, managed offload ON (per-run AI usage log)
|
|
4
|
-
# ARM raw : homegraph attached, HOMEGRAPH_OFFLOAD_DISABLE=1 (raw source)
|
|
5
|
-
# ARM nocg : no homegraph (empty MCP config) -> Read/Grep baseline
|
|
6
|
-
# All arms: claude -p sonnet --effort high. One JSON metrics line/run -> $RESULTS.
|
|
7
|
-
#
|
|
8
|
-
# Usage: offload-eval-3arm.sh <indexed-repo> <tier> <reps> "<question>"
|
|
9
|
-
# Env: MODEL=sonnet EFFORT=high RESULTS=<file> AGENT_EVAL_OUT=<scratch dir>
|
|
10
|
-
set -uo pipefail
|
|
11
|
-
HERE="$(cd "$(dirname "$0")" && pwd)"
|
|
12
|
-
ENGINE="$(cd "$HERE/../.." && pwd)"
|
|
13
|
-
BIN="$ENGINE/dist/bin/homegraph.js"
|
|
14
|
-
OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
|
|
15
|
-
TARGET="${1:?usage: offload-eval-3arm.sh <indexed-repo> <tier> <reps> \"<question>\"}"
|
|
16
|
-
TIER="${2:?tier}"; REPS="${3:?reps}"; Q="${4:?question}"
|
|
17
|
-
RUNS="$OUT/runs"
|
|
18
|
-
EXTRACT="$HERE/offload-eval-metrics.mjs"
|
|
19
|
-
RESULTS="${RESULTS:-$OUT/results.jsonl}"
|
|
20
|
-
REPO=$(basename "$TARGET")
|
|
21
|
-
mkdir -p "$RUNS"
|
|
22
|
-
command -v claude >/dev/null || { echo "no claude on PATH"; exit 1; }
|
|
23
|
-
[ -d "$TARGET/.homegraph" ] || { echo "not indexed: $TARGET (run offload-eval-setup.sh first)"; exit 1; }
|
|
24
|
-
# Physical path so pkill matches the daemon's real cmdline (macOS /tmp->/private/tmp symlink
|
|
25
|
-
# otherwise makes the kill miss the daemon, and the next arm connects to the SURVIVING daemon
|
|
26
|
-
# — contaminating the raw arm with offload).
|
|
27
|
-
TARGET=$(cd "$TARGET" && pwd -P)
|
|
28
|
-
|
|
29
|
-
prewarm() { # path extra-env (e.g. "FOO=bar")
|
|
30
|
-
pkill -9 -f "serve --mcp --path $1" 2>/dev/null; rm -f "$1/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
|
|
31
|
-
env ${2:-} HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 node "$BIN" serve --mcp --path "$1" </dev/null >/dev/null 2>&1 &
|
|
32
|
-
node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$1" \
|
|
33
|
-
&& echo " daemon warm" || echo " WARN daemon never bound"
|
|
34
|
-
}
|
|
35
|
-
|
|
36
|
-
run() { # arm rep mcp-config usage-log-or-dash
|
|
37
|
-
local arm="$1" rep="$2" cfg="$3" usage="$4" tag="$REPO-$1-$2"
|
|
38
|
-
[ "$usage" != "-" ] && : > "$usage"
|
|
39
|
-
# DISALLOW (optional): block sub-agent delegation across all arms so the A/B
|
|
40
|
-
# measures the retrieval mode, not whether Sonnet decides to spawn a homegraph-blind
|
|
41
|
-
# Explore subagent (which thrashes regardless and adds huge variance).
|
|
42
|
-
( cd "$TARGET" && claude -p "$Q" \
|
|
43
|
-
--output-format stream-json --verbose --permission-mode bypassPermissions \
|
|
44
|
-
--model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
|
|
45
|
-
${DISALLOW:+--disallowedTools "$DISALLOW"} \
|
|
46
|
-
--strict-mcp-config --mcp-config "$cfg" \
|
|
47
|
-
</dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
|
|
48
|
-
node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$usage" --arm "$arm" --rep "$rep" \
|
|
49
|
-
--repo "$REPO" --tier "$TIER" --q "$Q" >> "$RESULTS"
|
|
50
|
-
node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | main $${o.costUsdMain} ${o.tokBillable} tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}call/${o.ai.totalTokens}tok/$${o.ai.costUsd.toFixed(4)} | ok=${o.ok}`)' "$RESULTS"
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
CFG_OFF="$RUNS/mcp-offload-$REPO.json"; CFG_RAW="$RUNS/mcp-raw-$REPO.json"; CFG_NOCG="$RUNS/mcp-nocg.json"
|
|
54
|
-
USAGE="$RUNS/$REPO-usage.jsonl"
|
|
55
|
-
printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_USAGE_LOG=%s","node","%s","serve","--mcp","--path","%s"]}}}' "$USAGE" "$BIN" "$TARGET" > "$CFG_OFF"
|
|
56
|
-
printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_DISABLE=1","node","%s","serve","--mcp","--path","%s"]}}}' "$BIN" "$TARGET" > "$CFG_RAW"
|
|
57
|
-
printf '{"mcpServers":{}}' > "$CFG_NOCG"
|
|
58
|
-
|
|
59
|
-
# REP_START lets a later batch ADD reps without clobbering earlier jsonls
|
|
60
|
-
# (e.g. REP_START=4 REPS=3 -> reps 4,5,6; default starts at 1).
|
|
61
|
-
START="${REP_START:-1}"; END=$((START + REPS - 1))
|
|
62
|
-
echo "###### repo=$REPO tier=$TIER reps=$START..$END model=${MODEL:-sonnet}/${EFFORT:-high}"
|
|
63
|
-
echo "###### Q=$Q"
|
|
64
|
-
echo "== ARM offload =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
|
|
65
|
-
for r in $(seq "$START" "$END"); do run offload "$r" "$CFG_OFF" "$USAGE"; done
|
|
66
|
-
pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1
|
|
67
|
-
echo "== ARM raw =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_DISABLE=1"
|
|
68
|
-
for r in $(seq "$START" "$END"); do run raw "$r" "$CFG_RAW" "-"; done
|
|
69
|
-
pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1
|
|
70
|
-
echo "== ARM nocg =="
|
|
71
|
-
for r in $(seq "$START" "$END"); do run nocg "$r" "$CFG_NOCG" "-"; done
|
|
72
|
-
echo "###### DONE $REPO"
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# 3-arm offload eval for ONE indexed repo + ONE question, n reps each.
|
|
3
|
+
# ARM offload : homegraph attached, managed offload ON (per-run AI usage log)
|
|
4
|
+
# ARM raw : homegraph attached, HOMEGRAPH_OFFLOAD_DISABLE=1 (raw source)
|
|
5
|
+
# ARM nocg : no homegraph (empty MCP config) -> Read/Grep baseline
|
|
6
|
+
# All arms: claude -p sonnet --effort high. One JSON metrics line/run -> $RESULTS.
|
|
7
|
+
#
|
|
8
|
+
# Usage: offload-eval-3arm.sh <indexed-repo> <tier> <reps> "<question>"
|
|
9
|
+
# Env: MODEL=sonnet EFFORT=high RESULTS=<file> AGENT_EVAL_OUT=<scratch dir>
|
|
10
|
+
set -uo pipefail
|
|
11
|
+
HERE="$(cd "$(dirname "$0")" && pwd)"
|
|
12
|
+
ENGINE="$(cd "$HERE/../.." && pwd)"
|
|
13
|
+
BIN="$ENGINE/dist/bin/homegraph.js"
|
|
14
|
+
OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
|
|
15
|
+
TARGET="${1:?usage: offload-eval-3arm.sh <indexed-repo> <tier> <reps> \"<question>\"}"
|
|
16
|
+
TIER="${2:?tier}"; REPS="${3:?reps}"; Q="${4:?question}"
|
|
17
|
+
RUNS="$OUT/runs"
|
|
18
|
+
EXTRACT="$HERE/offload-eval-metrics.mjs"
|
|
19
|
+
RESULTS="${RESULTS:-$OUT/results.jsonl}"
|
|
20
|
+
REPO=$(basename "$TARGET")
|
|
21
|
+
mkdir -p "$RUNS"
|
|
22
|
+
command -v claude >/dev/null || { echo "no claude on PATH"; exit 1; }
|
|
23
|
+
[ -d "$TARGET/.homegraph" ] || { echo "not indexed: $TARGET (run offload-eval-setup.sh first)"; exit 1; }
|
|
24
|
+
# Physical path so pkill matches the daemon's real cmdline (macOS /tmp->/private/tmp symlink
|
|
25
|
+
# otherwise makes the kill miss the daemon, and the next arm connects to the SURVIVING daemon
|
|
26
|
+
# — contaminating the raw arm with offload).
|
|
27
|
+
TARGET=$(cd "$TARGET" && pwd -P)
|
|
28
|
+
|
|
29
|
+
prewarm() { # path extra-env (e.g. "FOO=bar")
|
|
30
|
+
pkill -9 -f "serve --mcp --path $1" 2>/dev/null; rm -f "$1/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
|
|
31
|
+
env ${2:-} HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 node "$BIN" serve --mcp --path "$1" </dev/null >/dev/null 2>&1 &
|
|
32
|
+
node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$1" \
|
|
33
|
+
&& echo " daemon warm" || echo " WARN daemon never bound"
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
run() { # arm rep mcp-config usage-log-or-dash
|
|
37
|
+
local arm="$1" rep="$2" cfg="$3" usage="$4" tag="$REPO-$1-$2"
|
|
38
|
+
[ "$usage" != "-" ] && : > "$usage"
|
|
39
|
+
# DISALLOW (optional): block sub-agent delegation across all arms so the A/B
|
|
40
|
+
# measures the retrieval mode, not whether Sonnet decides to spawn a homegraph-blind
|
|
41
|
+
# Explore subagent (which thrashes regardless and adds huge variance).
|
|
42
|
+
( cd "$TARGET" && claude -p "$Q" \
|
|
43
|
+
--output-format stream-json --verbose --permission-mode bypassPermissions \
|
|
44
|
+
--model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
|
|
45
|
+
${DISALLOW:+--disallowedTools "$DISALLOW"} \
|
|
46
|
+
--strict-mcp-config --mcp-config "$cfg" \
|
|
47
|
+
</dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
|
|
48
|
+
node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$usage" --arm "$arm" --rep "$rep" \
|
|
49
|
+
--repo "$REPO" --tier "$TIER" --q "$Q" >> "$RESULTS"
|
|
50
|
+
node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | main $${o.costUsdMain} ${o.tokBillable} tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}call/${o.ai.totalTokens}tok/$${o.ai.costUsd.toFixed(4)} | ok=${o.ok}`)' "$RESULTS"
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
CFG_OFF="$RUNS/mcp-offload-$REPO.json"; CFG_RAW="$RUNS/mcp-raw-$REPO.json"; CFG_NOCG="$RUNS/mcp-nocg.json"
|
|
54
|
+
USAGE="$RUNS/$REPO-usage.jsonl"
|
|
55
|
+
printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_USAGE_LOG=%s","node","%s","serve","--mcp","--path","%s"]}}}' "$USAGE" "$BIN" "$TARGET" > "$CFG_OFF"
|
|
56
|
+
printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_DISABLE=1","node","%s","serve","--mcp","--path","%s"]}}}' "$BIN" "$TARGET" > "$CFG_RAW"
|
|
57
|
+
printf '{"mcpServers":{}}' > "$CFG_NOCG"
|
|
58
|
+
|
|
59
|
+
# REP_START lets a later batch ADD reps without clobbering earlier jsonls
|
|
60
|
+
# (e.g. REP_START=4 REPS=3 -> reps 4,5,6; default starts at 1).
|
|
61
|
+
START="${REP_START:-1}"; END=$((START + REPS - 1))
|
|
62
|
+
echo "###### repo=$REPO tier=$TIER reps=$START..$END model=${MODEL:-sonnet}/${EFFORT:-high}"
|
|
63
|
+
echo "###### Q=$Q"
|
|
64
|
+
echo "== ARM offload =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
|
|
65
|
+
for r in $(seq "$START" "$END"); do run offload "$r" "$CFG_OFF" "$USAGE"; done
|
|
66
|
+
pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1
|
|
67
|
+
echo "== ARM raw =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_DISABLE=1"
|
|
68
|
+
for r in $(seq "$START" "$END"); do run raw "$r" "$CFG_RAW" "-"; done
|
|
69
|
+
pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1
|
|
70
|
+
echo "== ARM nocg =="
|
|
71
|
+
for r in $(seq "$START" "$END"); do run nocg "$r" "$CFG_NOCG" "-"; done
|
|
72
|
+
echo "###### DONE $REPO"
|