homegraph 1.5.0 → 1.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +305 -305
  3. package/dist/bin/command-supervision.d.ts.map +1 -1
  4. package/dist/bin/command-supervision.js +7 -4
  5. package/dist/bin/command-supervision.js.map +1 -1
  6. package/dist/bin/homegraph.js +9 -9
  7. package/dist/db/index.js +36 -36
  8. package/dist/db/migrations.js +37 -37
  9. package/dist/db/queries.js +156 -156
  10. package/dist/db/schema.sql +203 -203
  11. package/dist/directory.js +5 -5
  12. package/dist/extraction/languages/arkts-viewtree.d.ts +2 -4
  13. package/dist/extraction/languages/arkts-viewtree.d.ts.map +1 -1
  14. package/dist/extraction/languages/arkts-viewtree.js +6 -21
  15. package/dist/extraction/languages/arkts-viewtree.js.map +1 -1
  16. package/dist/extraction/languages/arkts.d.ts +16 -6
  17. package/dist/extraction/languages/arkts.d.ts.map +1 -1
  18. package/dist/extraction/languages/arkts.js +174 -21
  19. package/dist/extraction/languages/arkts.js.map +1 -1
  20. package/dist/extraction/wasm/tree-sitter-c_sharp.wasm +0 -0
  21. package/dist/extraction/wasm/tree-sitter-cfml.wasm +0 -0
  22. package/dist/extraction/wasm/tree-sitter-cfquery.wasm +0 -0
  23. package/dist/extraction/wasm/tree-sitter-cfscript.wasm +0 -0
  24. package/dist/extraction/wasm/tree-sitter-cobol.wasm +0 -0
  25. package/dist/extraction/wasm/tree-sitter-erlang.wasm +0 -0
  26. package/dist/extraction/wasm/tree-sitter-nix.wasm +0 -0
  27. package/dist/extraction/wasm/tree-sitter-pascal.wasm +0 -0
  28. package/dist/extraction/wasm/tree-sitter-vbnet.wasm +0 -0
  29. package/dist/installer/instructions-template.js +9 -9
  30. package/dist/mcp/liveness-watchdog.d.ts +18 -0
  31. package/dist/mcp/liveness-watchdog.d.ts.map +1 -1
  32. package/dist/mcp/liveness-watchdog.js +185 -73
  33. package/dist/mcp/liveness-watchdog.js.map +1 -1
  34. package/dist/mcp/server-instructions.js +47 -47
  35. package/dist/reasoning/reasoner.js +32 -32
  36. package/dist/spec/db/commit-node.js +4 -4
  37. package/dist/spec/db/fragment-node.js +10 -10
  38. package/dist/spec/db/fts.js +8 -8
  39. package/dist/spec/db/relations.js +88 -88
  40. package/dist/spec/db/schema.js +6 -6
  41. package/dist/spec/db/schema.sql +121 -121
  42. package/dist/spec/db/spec-node.js +4 -4
  43. package/dist/spec/llm/prompts.js +53 -53
  44. package/dist/spec/utils.d.ts +16 -4
  45. package/dist/spec/utils.d.ts.map +1 -1
  46. package/dist/spec/utils.js +58 -6
  47. package/dist/spec/utils.js.map +1 -1
  48. package/package.json +62 -62
  49. package/scripts/_tmp-cfwk-resolve.log +0 -0
  50. package/scripts/_tmp-cfwk-sig.log +0 -0
  51. package/scripts/_tmp-cfwk-vt.log +0 -0
  52. package/scripts/add-lang/bench.sh +60 -60
  53. package/scripts/add-lang/check-grammar.mjs +75 -75
  54. package/scripts/add-lang/dump-ast.mjs +103 -103
  55. package/scripts/add-lang/verify-extraction.mjs +70 -70
  56. package/scripts/agent-eval/ab-adoption.sh +91 -91
  57. package/scripts/agent-eval/ab-hook.sh +86 -86
  58. package/scripts/agent-eval/ab-impl.sh +78 -78
  59. package/scripts/agent-eval/ab-new-vs-baseline.sh +102 -102
  60. package/scripts/agent-eval/ab-sufficiency.sh +78 -78
  61. package/scripts/agent-eval/arms-F.sh +21 -21
  62. package/scripts/agent-eval/arms-matrix.sh +37 -37
  63. package/scripts/agent-eval/audit.sh +68 -68
  64. package/scripts/agent-eval/bench-readme.sh +28 -28
  65. package/scripts/agent-eval/bench-why-repo.sh +22 -22
  66. package/scripts/agent-eval/block-read-hook.sh +19 -19
  67. package/scripts/agent-eval/hook-settings.json +15 -15
  68. package/scripts/agent-eval/itrun.sh +120 -120
  69. package/scripts/agent-eval/offload-eval-3arm.sh +72 -72
  70. package/scripts/agent-eval/offload-eval-cost.mjs +133 -133
  71. package/scripts/agent-eval/offload-eval-effort.mjs +108 -108
  72. package/scripts/agent-eval/offload-eval-frontload-matrix.sh +25 -25
  73. package/scripts/agent-eval/offload-eval-frontload.sh +47 -47
  74. package/scripts/agent-eval/offload-eval-ground-truth.json +18 -18
  75. package/scripts/agent-eval/offload-eval-hook.mjs +84 -84
  76. package/scripts/agent-eval/offload-eval-judge.mjs +103 -103
  77. package/scripts/agent-eval/offload-eval-matrix.sh +20 -20
  78. package/scripts/agent-eval/offload-eval-metrics.mjs +94 -94
  79. package/scripts/agent-eval/offload-eval-refs1.sh +50 -50
  80. package/scripts/agent-eval/offload-eval-setup.sh +24 -24
  81. package/scripts/agent-eval/offload-eval-styles.sh +71 -71
  82. package/scripts/agent-eval/offload-eval-summarize.mjs +68 -68
  83. package/scripts/agent-eval/offload-eval.md +76 -76
  84. package/scripts/agent-eval/parse-arms.mjs +116 -116
  85. package/scripts/agent-eval/parse-bench-readme.mjs +84 -84
  86. package/scripts/agent-eval/parse-run.mjs +45 -45
  87. package/scripts/agent-eval/parse-session.mjs +93 -93
  88. package/scripts/agent-eval/probe-context.mjs +21 -21
  89. package/scripts/agent-eval/probe-explore.mjs +40 -40
  90. package/scripts/agent-eval/probe-node.mjs +20 -20
  91. package/scripts/agent-eval/probe-sweep.mjs +119 -119
  92. package/scripts/agent-eval/probe-trace.mjs +20 -20
  93. package/scripts/agent-eval/redirect-read-hook.sh +38 -38
  94. package/scripts/agent-eval/repro-concurrent-explore.mjs +119 -119
  95. package/scripts/agent-eval/repro-daemon-clients.mjs +125 -125
  96. package/scripts/agent-eval/run-agent.sh +34 -34
  97. package/scripts/agent-eval/run-all.sh +75 -75
  98. package/scripts/agent-eval/run-arms.sh +56 -56
  99. package/scripts/agent-eval/seq-matrix.mjs +137 -137
  100. package/scripts/bench-arkts-init-rss.log +0 -0
  101. package/scripts/build-bundle.sh +123 -123
  102. package/scripts/exp_boundary_eval/README.md +247 -247
  103. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-310.pyc +0 -0
  104. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-38.pyc +0 -0
  105. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-310.pyc +0 -0
  106. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-38.pyc +0 -0
  107. package/scripts/exp_boundary_eval/__pycache__/deveco_arm.cpython-38.pyc +0 -0
  108. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-310.pyc +0 -0
  109. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-38.pyc +0 -0
  110. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-310.pyc +0 -0
  111. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-38.pyc +0 -0
  112. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-310.pyc +0 -0
  113. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-38.pyc +0 -0
  114. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-310.pyc +0 -0
  115. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-38.pyc +0 -0
  116. package/scripts/exp_boundary_eval/__pycache__/win_mcp_launcher.cpython-38.pyc +0 -0
  117. package/scripts/exp_boundary_eval/_test_mcp_chain.py +78 -78
  118. package/scripts/exp_boundary_eval/_test_stdin.py +8 -8
  119. package/scripts/exp_boundary_eval/_utils.py +1116 -1116
  120. package/scripts/exp_boundary_eval/analyze.py +1313 -1313
  121. package/scripts/exp_boundary_eval/deveco_arm.py +519 -519
  122. package/scripts/exp_boundary_eval/run_all.py +378 -378
  123. package/scripts/exp_boundary_eval/run_one.py +165 -165
  124. package/scripts/exp_boundary_eval/run_session.py +158 -158
  125. package/scripts/exp_boundary_eval/setup.py +120 -120
  126. package/scripts/exp_boundary_eval/win_mcp_launcher.py +73 -73
  127. package/scripts/exp_boundary_eval/win_mcp_stdio_wrap.js +36 -36
  128. package/scripts/exp_boundary_eval/win_node_launcher.py +24 -24
  129. package/scripts/extract-release-notes.mjs +130 -130
  130. package/scripts/local-install.sh +41 -41
  131. package/scripts/npm-sdk.js +75 -75
  132. package/scripts/npm-shim.js +275 -275
  133. package/scripts/ohos-sdk-publish.mjs +133 -133
  134. package/scripts/pack-npm.sh +119 -119
  135. package/scripts/prepare-release.mjs +270 -270
  136. package/scripts/probe-arkts-mem-why-run.log +0 -0
  137. package/scripts/probe-banner-livecard-ir.log +0 -0
  138. package/scripts/probe-cfwk-ast.stderr.log +0 -0
  139. package/scripts/probe-cfwk-ast.stdout.log +0 -0
  140. package/scripts/probe-cfwk-attr-shape.log +0 -0
  141. package/scripts/probe-cfwk-cfgdump.stderr.log +0 -0
  142. package/scripts/probe-cfwk-cfgdump.stdout.log +0 -0
  143. package/scripts/probe-cfwk-cvc.stderr.log +0 -0
  144. package/scripts/probe-cfwk-cvc.stdout.log +0 -0
  145. package/scripts/probe-cfwk-diag2.log +0 -0
  146. package/scripts/probe-cfwk-diag3.log +0 -0
  147. package/scripts/probe-cfwk-fedbg.stderr.log +0 -0
  148. package/scripts/probe-cfwk-fedbg.stdout.log +0 -0
  149. package/scripts/probe-cfwk-fileresult.log +0 -0
  150. package/scripts/probe-cfwk-fix.stderr.log +0 -0
  151. package/scripts/probe-cfwk-fix.stdout.log +0 -0
  152. package/scripts/probe-cfwk-fix2.stderr.log +0 -0
  153. package/scripts/probe-cfwk-fix2.stdout.log +0 -0
  154. package/scripts/probe-cfwk-fix3.stderr.log +0 -0
  155. package/scripts/probe-cfwk-fix3.stdout.log +0 -0
  156. package/scripts/probe-cfwk-foreach.log +0 -0
  157. package/scripts/probe-cfwk-getmethod-throw.log +0 -0
  158. package/scripts/probe-cfwk-hg-extract.log +0 -0
  159. package/scripts/probe-cfwk-pr1003-noprior.stderr.log +0 -0
  160. package/scripts/probe-cfwk-pr1003-noprior.stdout.log +0 -0
  161. package/scripts/probe-cfwk-pr1003.stderr.log +0 -0
  162. package/scripts/probe-cfwk-pr1003.stdout.log +0 -0
  163. package/scripts/probe-cfwk-preroot-noprior.stderr.log +0 -0
  164. package/scripts/probe-cfwk-preroot-noprior.stdout.log +0 -0
  165. package/scripts/probe-cfwk-preroot-prior.stderr.log +0 -0
  166. package/scripts/probe-cfwk-preroot-prior.stdout.log +0 -0
  167. package/scripts/probe-cfwk-preroot-skipstate.stderr.log +0 -0
  168. package/scripts/probe-cfwk-preroot-skipstate.stdout.log +0 -0
  169. package/scripts/probe-cfwk-resolve-sim.log +0 -0
  170. package/scripts/probe-cfwk-tree-shape.log +0 -0
  171. package/scripts/probe-cfwk-walk-abort.log +0 -0
  172. package/scripts/probe-cfwk.log +0 -0
  173. package/scripts/probe-force-index.log +0 -0
  174. package/scripts/probe-no-force-index.log +0 -0
  175. package/scripts/probe-samefile-ir.stderr.log +0 -0
  176. package/scripts/probe-samefile-ir.stdout.log +224 -0
  177. package/scripts/probe-sdk-vs-project.log +0 -0
  178. package/scripts/probe-viewtree-downgrade.log +0 -0
  179. package/dist/arkts/ohos-api-index.d.ts +0 -15
  180. package/dist/arkts/ohos-api-index.d.ts.map +0 -1
  181. package/dist/arkts/ohos-api-index.js +0 -190
  182. package/dist/arkts/ohos-api-index.js.map +0 -1
  183. package/dist/arkts/ohos-sdk-input.d.ts +0 -36
  184. package/dist/arkts/ohos-sdk-input.d.ts.map +0 -1
  185. package/dist/arkts/ohos-sdk-input.js +0 -214
  186. package/dist/arkts/ohos-sdk-input.js.map +0 -1
  187. package/dist/extraction/languages/arkts-state-decorators.d.ts +0 -13
  188. package/dist/extraction/languages/arkts-state-decorators.d.ts.map +0 -1
  189. package/dist/extraction/languages/arkts-state-decorators.js +0 -26
  190. package/dist/extraction/languages/arkts-state-decorators.js.map +0 -1
  191. package/dist/extraction/languages/ohos-api-consumer.d.ts +0 -34
  192. package/dist/extraction/languages/ohos-api-consumer.d.ts.map +0 -1
  193. package/dist/extraction/languages/ohos-api-consumer.js +0 -283
  194. package/dist/extraction/languages/ohos-api-consumer.js.map +0 -1
  195. package/dist/spec/build/git-scanner.d.ts +0 -93
  196. package/dist/spec/build/git-scanner.d.ts.map +0 -1
  197. package/dist/spec/build/git-scanner.js +0 -254
  198. package/dist/spec/build/git-scanner.js.map +0 -1
  199. package/dist/spec/git-utils.d.ts +0 -8
  200. package/dist/spec/git-utils.d.ts.map +0 -1
  201. package/dist/spec/git-utils.js +0 -14
  202. package/dist/spec/git-utils.js.map +0 -1
  203. package/dist/spec/mine/clusterer.d.ts +0 -63
  204. package/dist/spec/mine/clusterer.d.ts.map +0 -1
  205. package/dist/spec/mine/clusterer.js +0 -904
  206. package/dist/spec/mine/clusterer.js.map +0 -1
  207. package/dist/spec/mine/progress-handler.d.ts +0 -22
  208. package/dist/spec/mine/progress-handler.d.ts.map +0 -1
  209. package/dist/spec/mine/progress-handler.js +0 -108
  210. package/dist/spec/mine/progress-handler.js.map +0 -1
  211. package/dist/spec/mine/progress.d.ts +0 -23
  212. package/dist/spec/mine/progress.d.ts.map +0 -1
  213. package/dist/spec/mine/progress.js +0 -12
  214. package/dist/spec/mine/progress.js.map +0 -1
@@ -1,165 +1,165 @@
1
- #!/usr/bin/env python3
2
- """Run a single one-shot experiment. Supports multiple agents via --agent."""
3
-
4
- import os
5
- import subprocess
6
- import sys
7
- from pathlib import Path
8
- from typing import Optional
9
-
10
- import setup as setup_mod
11
- from _utils import (OUTPUT_DIR, append_run_log, build_agent_cmd, current_time_ms,
12
- error, find_experiment, get_agent, info, header,
13
- parse_output, resolve_agent_model, run_monitored_subprocess,
14
- subprocess_no_window_kwargs, warn, write_json)
15
- from deveco_arm import analyze_homegraph_stream
16
-
17
-
18
- def run_experiment(exp_id: str, skip_permissions: bool = True,
19
- agent_name: str = "",
20
- results_dir: Optional[Path] = None,
21
- state_dir: Optional[Path] = None) -> bool:
22
- exp = find_experiment(exp_id)
23
- if exp is None:
24
- error(f"Experiment {exp_id} not found in config")
25
- return False
26
- if exp.get("type") == "session":
27
- error(f"Experiment {exp_id} is session type. Use run_session.py.")
28
- return False
29
-
30
- agent = get_agent(agent_name)
31
- clone_path = os.environ.get("REPO_CLONE", "")
32
- local_repo = os.environ.get("REPO_PATH", "")
33
- if clone_path:
34
- repo_root = Path(clone_path)
35
- elif local_repo:
36
- repo_root = Path(local_repo)
37
- else:
38
- error("No repo. Use run_all.py to clone, or set REPO_PATH / REPO_CLONE.")
39
- return False
40
-
41
- title = exp["title"]
42
- prompt = exp["prompt"]
43
- max_turns = exp.get("max_turns", 15)
44
-
45
- header(f"Experiment {exp_id}: {title}")
46
- info(f"Agent: {agent['name']} ({agent['binary']}) | Repo: {repo_root}")
47
- info(f"Max turns: {max_turns} | Permissions: {'skip' if skip_permissions else 'ask'}")
48
-
49
- session_id, _ = setup_mod.prepare_environment(exp_id, "full", agent_name,
50
- state_dir=state_dir)
51
-
52
- result_dir = results_dir / exp_id if results_dir else OUTPUT_DIR / exp_id
53
- result_dir.mkdir(parents=True, exist_ok=True)
54
- (result_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
55
- write_json(result_dir / "experiment_config.json", {**exp, "agent": agent_name or "default"})
56
-
57
- info("Starting agent...")
58
- start_ms = current_time_ms()
59
- os.chdir(repo_root)
60
-
61
- cmd = build_agent_cmd(agent, session_id, max_turns, prompt,
62
- resume=False, skip_permissions=skip_permissions,
63
- repo_root=repo_root)
64
-
65
- stream_path = result_dir / "stream_output.jsonl"
66
- text_path = result_dir / "raw_output.txt"
67
- track_hg = os.environ.get("EVAL_ARM", "") == "homegraph"
68
-
69
- with open(stream_path, "w", encoding="utf-8") as stream_f:
70
- result, mem = run_monitored_subprocess(
71
- cmd, stdout=stream_f, stderr=subprocess.STDOUT, text=True,
72
- track_homegraph=track_hg,
73
- )
74
- exit_code = result.returncode
75
-
76
- end_ms = current_time_ms()
77
- dur_ms = end_ms - start_ms
78
- dur_s = f"{dur_ms / 1000:.1f}"
79
- info(f"Completed in {dur_s}s (exit={exit_code})")
80
-
81
- # Parse agent output
82
- parser_name = agent.get("parser", "claude_stream_json")
83
- stats = parse_output(stream_path, parser_name)
84
-
85
- if exit_code != 0:
86
- error(f"Agent 退出码 {exit_code}")
87
- for err in stats.errors[:3]:
88
- error(f" {err}")
89
- if "Unexpected server error" in " ".join(stats.errors):
90
- warn("DevEco 服务端报错 — 通常为 API/网络/配额问题,与评测脚本无关。"
91
- "请检查 DevEco 登录状态、模型配置,稍后重试 `deveco run`。")
92
- elif stats.tool_calls == 0 and not stats.errors:
93
- warn("无工具调用且无错误事件 — 检查 stream_output.jsonl 是否为空。")
94
-
95
- # Human-readable text
96
- text_content = "\n".join(stats.text_lines)
97
- text_path.write_text(text_content or "(no text output)", encoding="utf-8")
98
-
99
- # Modified files
100
- r2 = subprocess.run(
101
- ["git", "diff", "--name-only"], capture_output=True, text=True, cwd=repo_root,
102
- **subprocess_no_window_kwargs(),
103
- )
104
- mod_files = len([l for l in r2.stdout.strip().split("\n") if l])
105
-
106
- hg_stream = analyze_homegraph_stream(stream_path)
107
-
108
- results = {
109
- "experiment_id": exp_id, "title": title, "agent": agent["name"],
110
- "agent_binary": agent["binary"], "session_id": session_id,
111
- "start_time_ms": start_ms, "end_time_ms": end_ms,
112
- "duration_ms": dur_ms, "duration_s": dur_s,
113
- "exit_code": exit_code, "max_turns": max_turns,
114
- "skip_permissions": skip_permissions,
115
- "tool_calls": stats.tool_calls, "tool_names": stats.tool_names,
116
- "homegraph_tool_calls": stats.homegraph_tool_calls,
117
- "used_homegraph": stats.used_homegraph,
118
- **hg_stream,
119
- "eval_arm": os.environ.get("EVAL_ARM", ""),
120
- "files_read": list(set(stats.files_read)),
121
- "files_edited": list(set(stats.files_edited)),
122
- "model": resolve_agent_model(agent, stats),
123
- "deveco_session_id": stats.deveco_session_id,
124
- "input_tokens": stats.total_input_tokens,
125
- "output_tokens": stats.total_output_tokens,
126
- "assistant_messages": stats.assistant_messages,
127
- "max_turns_hit": stats.max_turns_hit,
128
- "errors": stats.errors[:5],
129
- "modified_files": mod_files,
130
- "memory": mem.to_dict(),
131
- }
132
- write_json(result_dir / "results.json", results)
133
-
134
- if state_dir:
135
- append_run_log(state_dir, f"Exp {exp_id} finished in {dur_s}s, tools={stats.tool_calls}, exit={exit_code}")
136
- info(f"Results saved: {result_dir}/")
137
- print(f" Duration: {dur_s}s | Tool calls: {stats.tool_calls} | "
138
- f"HG tools: {stats.homegraph_tool_calls} | "
139
- f"Files read: {len(set(stats.files_read))} | Edited: {mod_files} | "
140
- f"Peak mem: {mem.peak_combined_rss_mb:.0f} MB")
141
- if stats.max_turns_hit:
142
- print(f" ⚠️ Max turns ({max_turns}) hit")
143
- return exit_code == 0
144
-
145
-
146
- if __name__ == "__main__":
147
- # Parse: python run_one.py <id> [--agent <name>] [--no-skip-permissions]
148
- args = sys.argv[1:]
149
- exp_id = args[0] if args else ""
150
- agent_name = ""
151
- skip = True
152
-
153
- i = 1
154
- while i < len(args):
155
- if args[i] == "--agent" and i + 1 < len(args):
156
- agent_name = args[i + 1]; i += 2
157
- elif args[i] == "--no-skip-permissions":
158
- skip = False; i += 1
159
- else:
160
- i += 1
161
-
162
- if not exp_id:
163
- error("Usage: python run_one.py <experiment_id> [--agent <name>]")
164
- sys.exit(1)
165
- sys.exit(0 if run_experiment(exp_id, skip, agent_name) else 1)
1
+ #!/usr/bin/env python3
2
+ """Run a single one-shot experiment. Supports multiple agents via --agent."""
3
+
4
+ import os
5
+ import subprocess
6
+ import sys
7
+ from pathlib import Path
8
+ from typing import Optional
9
+
10
+ import setup as setup_mod
11
+ from _utils import (OUTPUT_DIR, append_run_log, build_agent_cmd, current_time_ms,
12
+ error, find_experiment, get_agent, info, header,
13
+ parse_output, resolve_agent_model, run_monitored_subprocess,
14
+ subprocess_no_window_kwargs, warn, write_json)
15
+ from deveco_arm import analyze_homegraph_stream
16
+
17
+
18
+ def run_experiment(exp_id: str, skip_permissions: bool = True,
19
+ agent_name: str = "",
20
+ results_dir: Optional[Path] = None,
21
+ state_dir: Optional[Path] = None) -> bool:
22
+ exp = find_experiment(exp_id)
23
+ if exp is None:
24
+ error(f"Experiment {exp_id} not found in config")
25
+ return False
26
+ if exp.get("type") == "session":
27
+ error(f"Experiment {exp_id} is session type. Use run_session.py.")
28
+ return False
29
+
30
+ agent = get_agent(agent_name)
31
+ clone_path = os.environ.get("REPO_CLONE", "")
32
+ local_repo = os.environ.get("REPO_PATH", "")
33
+ if clone_path:
34
+ repo_root = Path(clone_path)
35
+ elif local_repo:
36
+ repo_root = Path(local_repo)
37
+ else:
38
+ error("No repo. Use run_all.py to clone, or set REPO_PATH / REPO_CLONE.")
39
+ return False
40
+
41
+ title = exp["title"]
42
+ prompt = exp["prompt"]
43
+ max_turns = exp.get("max_turns", 15)
44
+
45
+ header(f"Experiment {exp_id}: {title}")
46
+ info(f"Agent: {agent['name']} ({agent['binary']}) | Repo: {repo_root}")
47
+ info(f"Max turns: {max_turns} | Permissions: {'skip' if skip_permissions else 'ask'}")
48
+
49
+ session_id, _ = setup_mod.prepare_environment(exp_id, "full", agent_name,
50
+ state_dir=state_dir)
51
+
52
+ result_dir = results_dir / exp_id if results_dir else OUTPUT_DIR / exp_id
53
+ result_dir.mkdir(parents=True, exist_ok=True)
54
+ (result_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
55
+ write_json(result_dir / "experiment_config.json", {**exp, "agent": agent_name or "default"})
56
+
57
+ info("Starting agent...")
58
+ start_ms = current_time_ms()
59
+ os.chdir(repo_root)
60
+
61
+ cmd = build_agent_cmd(agent, session_id, max_turns, prompt,
62
+ resume=False, skip_permissions=skip_permissions,
63
+ repo_root=repo_root)
64
+
65
+ stream_path = result_dir / "stream_output.jsonl"
66
+ text_path = result_dir / "raw_output.txt"
67
+ track_hg = os.environ.get("EVAL_ARM", "") == "homegraph"
68
+
69
+ with open(stream_path, "w", encoding="utf-8") as stream_f:
70
+ result, mem = run_monitored_subprocess(
71
+ cmd, stdout=stream_f, stderr=subprocess.STDOUT, text=True,
72
+ track_homegraph=track_hg,
73
+ )
74
+ exit_code = result.returncode
75
+
76
+ end_ms = current_time_ms()
77
+ dur_ms = end_ms - start_ms
78
+ dur_s = f"{dur_ms / 1000:.1f}"
79
+ info(f"Completed in {dur_s}s (exit={exit_code})")
80
+
81
+ # Parse agent output
82
+ parser_name = agent.get("parser", "claude_stream_json")
83
+ stats = parse_output(stream_path, parser_name)
84
+
85
+ if exit_code != 0:
86
+ error(f"Agent 退出码 {exit_code}")
87
+ for err in stats.errors[:3]:
88
+ error(f" {err}")
89
+ if "Unexpected server error" in " ".join(stats.errors):
90
+ warn("DevEco 服务端报错 — 通常为 API/网络/配额问题,与评测脚本无关。"
91
+ "请检查 DevEco 登录状态、模型配置,稍后重试 `deveco run`。")
92
+ elif stats.tool_calls == 0 and not stats.errors:
93
+ warn("无工具调用且无错误事件 — 检查 stream_output.jsonl 是否为空。")
94
+
95
+ # Human-readable text
96
+ text_content = "\n".join(stats.text_lines)
97
+ text_path.write_text(text_content or "(no text output)", encoding="utf-8")
98
+
99
+ # Modified files
100
+ r2 = subprocess.run(
101
+ ["git", "diff", "--name-only"], capture_output=True, text=True, cwd=repo_root,
102
+ **subprocess_no_window_kwargs(),
103
+ )
104
+ mod_files = len([l for l in r2.stdout.strip().split("\n") if l])
105
+
106
+ hg_stream = analyze_homegraph_stream(stream_path)
107
+
108
+ results = {
109
+ "experiment_id": exp_id, "title": title, "agent": agent["name"],
110
+ "agent_binary": agent["binary"], "session_id": session_id,
111
+ "start_time_ms": start_ms, "end_time_ms": end_ms,
112
+ "duration_ms": dur_ms, "duration_s": dur_s,
113
+ "exit_code": exit_code, "max_turns": max_turns,
114
+ "skip_permissions": skip_permissions,
115
+ "tool_calls": stats.tool_calls, "tool_names": stats.tool_names,
116
+ "homegraph_tool_calls": stats.homegraph_tool_calls,
117
+ "used_homegraph": stats.used_homegraph,
118
+ **hg_stream,
119
+ "eval_arm": os.environ.get("EVAL_ARM", ""),
120
+ "files_read": list(set(stats.files_read)),
121
+ "files_edited": list(set(stats.files_edited)),
122
+ "model": resolve_agent_model(agent, stats),
123
+ "deveco_session_id": stats.deveco_session_id,
124
+ "input_tokens": stats.total_input_tokens,
125
+ "output_tokens": stats.total_output_tokens,
126
+ "assistant_messages": stats.assistant_messages,
127
+ "max_turns_hit": stats.max_turns_hit,
128
+ "errors": stats.errors[:5],
129
+ "modified_files": mod_files,
130
+ "memory": mem.to_dict(),
131
+ }
132
+ write_json(result_dir / "results.json", results)
133
+
134
+ if state_dir:
135
+ append_run_log(state_dir, f"Exp {exp_id} finished in {dur_s}s, tools={stats.tool_calls}, exit={exit_code}")
136
+ info(f"Results saved: {result_dir}/")
137
+ print(f" Duration: {dur_s}s | Tool calls: {stats.tool_calls} | "
138
+ f"HG tools: {stats.homegraph_tool_calls} | "
139
+ f"Files read: {len(set(stats.files_read))} | Edited: {mod_files} | "
140
+ f"Peak mem: {mem.peak_combined_rss_mb:.0f} MB")
141
+ if stats.max_turns_hit:
142
+ print(f" ⚠️ Max turns ({max_turns}) hit")
143
+ return exit_code == 0
144
+
145
+
146
+ if __name__ == "__main__":
147
+ # Parse: python run_one.py <id> [--agent <name>] [--no-skip-permissions]
148
+ args = sys.argv[1:]
149
+ exp_id = args[0] if args else ""
150
+ agent_name = ""
151
+ skip = True
152
+
153
+ i = 1
154
+ while i < len(args):
155
+ if args[i] == "--agent" and i + 1 < len(args):
156
+ agent_name = args[i + 1]; i += 2
157
+ elif args[i] == "--no-skip-permissions":
158
+ skip = False; i += 1
159
+ else:
160
+ i += 1
161
+
162
+ if not exp_id:
163
+ error("Usage: python run_one.py <experiment_id> [--agent <name>]")
164
+ sys.exit(1)
165
+ sys.exit(0 if run_experiment(exp_id, skip, agent_name) else 1)
@@ -1,158 +1,158 @@
1
- #!/usr/bin/env python3
2
- """Run multi-turn session experiment (5). Supports multiple agents via --agent."""
3
-
4
- import os
5
- import subprocess
6
- import sys
7
- import time
8
- from pathlib import Path
9
- from typing import Optional
10
-
11
- import setup as setup_mod
12
- from _utils import (OUTPUT_DIR, append_run_log, build_agent_cmd,
13
- current_time_ms, find_experiment, get_agent, info, header,
14
- log, parse_output, resolve_agent_model, run_monitored_subprocess,
15
- subprocess_no_window_kwargs, write_json, warn)
16
-
17
-
18
- def run_session(exp_id: str = "5", skip_permissions: bool = True,
19
- agent_name: str = "",
20
- results_dir: Optional[Path] = None,
21
- state_dir: Optional[Path] = None) -> bool:
22
- exp = find_experiment(exp_id)
23
- if exp is None:
24
- print(f"ERROR: Experiment {exp_id} not found")
25
- return False
26
-
27
- agent = get_agent(agent_name)
28
- clone_path = os.environ.get("REPO_CLONE", "")
29
- local_repo = os.environ.get("REPO_PATH", "")
30
- if clone_path:
31
- repo_root = Path(clone_path)
32
- elif local_repo:
33
- repo_root = Path(local_repo)
34
- else:
35
- print("ERROR: No repo. Use run_all.py to clone, or set REPO_PATH / REPO_CLONE.")
36
- return False
37
- max_turns = exp.get("max_turns_per_round", 25)
38
-
39
- header("Experiment 5: Session Persistence & Context Decay")
40
- info(f"Agent: {agent['name']} ({agent['binary']}) | Repo: {repo_root}")
41
- info(f"Max turns/round: {max_turns} | Permissions: {'skip' if skip_permissions else 'ask'}")
42
-
43
- session_id, _ = setup_mod.prepare_environment("5", "full", agent_name,
44
- state_dir=state_dir)
45
- log(f"Shared Session ID: {session_id}")
46
-
47
- result_dir = results_dir / "5" if results_dir else OUTPUT_DIR / "5"
48
- result_dir.mkdir(parents=True, exist_ok=True)
49
-
50
- rounds_data = []
51
- session_model = ""
52
- deveco_session_id = ""
53
- for round_num in range(1, 6):
54
- ri = next((r for r in exp["rounds"] if r["round"] == round_num), None)
55
- if not ri: continue
56
-
57
- header(f"Round {round_num} of 5")
58
- info(f"Title: {ri['title']}")
59
-
60
- round_dir = result_dir / f"round_{round_num}"
61
- round_dir.mkdir(parents=True, exist_ok=True)
62
- (round_dir / "prompt.txt").write_text(ri["prompt"], encoding="utf-8")
63
-
64
- start_ms = current_time_ms()
65
- is_resume = round_num > 1
66
-
67
- cmd = build_agent_cmd(agent, session_id, max_turns, ri["prompt"],
68
- resume=is_resume, skip_permissions=skip_permissions,
69
- repo_root=repo_root)
70
-
71
- stream_path = round_dir / "stream_output.jsonl"
72
- text_path = round_dir / "raw_output.txt"
73
- track_hg = os.environ.get("EVAL_ARM", "") == "homegraph"
74
-
75
- with open(stream_path, "w", encoding="utf-8") as stream_f:
76
- result, mem = run_monitored_subprocess(
77
- cmd, stdout=stream_f, stderr=subprocess.STDOUT, text=True,
78
- track_homegraph=track_hg,
79
- )
80
- exit_code = result.returncode
81
-
82
- end_ms = current_time_ms()
83
- dur_ms = end_ms - start_ms
84
- dur_s = f"{dur_ms / 1000:.1f}"
85
-
86
- parser_name = agent.get("parser", "claude_stream_json")
87
- stats = parse_output(stream_path, parser_name)
88
- if not session_model:
89
- session_model = resolve_agent_model(agent, stats)
90
- if not deveco_session_id and stats.deveco_session_id:
91
- deveco_session_id = stats.deveco_session_id
92
-
93
- text_content = "\n".join(stats.text_lines)
94
- text_path.write_text(text_content or "(no text output)", encoding="utf-8")
95
-
96
- r2 = subprocess.run(
97
- ["git", "diff", "--name-only"], capture_output=True, text=True, cwd=repo_root,
98
- **subprocess_no_window_kwargs(),
99
- )
100
- mod_files = len([l for l in r2.stdout.strip().split("\n") if l])
101
-
102
- rr = {"round": round_num, "title": ri["title"], "duration_ms": dur_ms,
103
- "duration_s": dur_s, "exit_code": exit_code,
104
- "tool_calls": stats.tool_calls, "tool_names": stats.tool_names,
105
- "files_read": list(set(stats.files_read)),
106
- "files_edited": list(set(stats.files_edited)),
107
- "assistant_messages": stats.assistant_messages,
108
- "max_turns_hit": stats.max_turns_hit,
109
- "modified_files": mod_files, "max_turns": max_turns,
110
- "memory": mem.to_dict()}
111
- write_json(round_dir / "round_results.json", rr)
112
- rounds_data.append(rr)
113
-
114
- info(f"Round {round_num} done: {dur_s}s | Tool calls: {stats.tool_calls} | "
115
- f"Files read: {len(set(stats.files_read))} | Edited: {mod_files}")
116
- if stats.max_turns_hit:
117
- warn(f" Max turns ({max_turns}) hit in round {round_num}")
118
- if round_num < 5:
119
- time.sleep(2)
120
-
121
- summary = {
122
- "experiment_id": "5", "session_id": session_id, "agent": agent["name"],
123
- "model": session_model,
124
- "deveco_session_id": deveco_session_id,
125
- "total_rounds": len(rounds_data),
126
- "total_duration_ms": sum(r["duration_ms"] for r in rounds_data),
127
- "total_tool_calls": sum(int(r["tool_calls"]) for r in rounds_data),
128
- "memory_retention_curve": [
129
- {"round": r["round"], "tool_calls": int(r["tool_calls"]),
130
- "files_read": r.get("files_read", [])} for r in rounds_data
131
- ],
132
- }
133
- write_json(result_dir / "session_results.json", summary)
134
-
135
- print("\nMemory Retention Curve:")
136
- for r in rounds_data:
137
- tc = int(r["tool_calls"])
138
- fr = len(r.get("files_read", []))
139
- print(f" Round {r['round']}: {'#' * (tc // 2)} ({tc} tool calls, {fr} files read)")
140
-
141
- log(f"Results: {result_dir}/session_results.json")
142
- return True
143
-
144
-
145
- if __name__ == "__main__":
146
- args = sys.argv[1:]
147
- exp_id = args[0] if args else "5"
148
- agent_name = ""
149
- skip = True
150
- i = 1
151
- while i < len(args):
152
- if args[i] == "--agent" and i + 1 < len(args):
153
- agent_name = args[i + 1]; i += 2
154
- elif args[i] == "--no-skip-permissions":
155
- skip = False; i += 1
156
- else:
157
- i += 1
158
- sys.exit(0 if run_session(exp_id, skip, agent_name) else 1)
1
+ #!/usr/bin/env python3
2
+ """Run multi-turn session experiment (5). Supports multiple agents via --agent."""
3
+
4
+ import os
5
+ import subprocess
6
+ import sys
7
+ import time
8
+ from pathlib import Path
9
+ from typing import Optional
10
+
11
+ import setup as setup_mod
12
+ from _utils import (OUTPUT_DIR, append_run_log, build_agent_cmd,
13
+ current_time_ms, find_experiment, get_agent, info, header,
14
+ log, parse_output, resolve_agent_model, run_monitored_subprocess,
15
+ subprocess_no_window_kwargs, write_json, warn)
16
+
17
+
18
+ def run_session(exp_id: str = "5", skip_permissions: bool = True,
19
+ agent_name: str = "",
20
+ results_dir: Optional[Path] = None,
21
+ state_dir: Optional[Path] = None) -> bool:
22
+ exp = find_experiment(exp_id)
23
+ if exp is None:
24
+ print(f"ERROR: Experiment {exp_id} not found")
25
+ return False
26
+
27
+ agent = get_agent(agent_name)
28
+ clone_path = os.environ.get("REPO_CLONE", "")
29
+ local_repo = os.environ.get("REPO_PATH", "")
30
+ if clone_path:
31
+ repo_root = Path(clone_path)
32
+ elif local_repo:
33
+ repo_root = Path(local_repo)
34
+ else:
35
+ print("ERROR: No repo. Use run_all.py to clone, or set REPO_PATH / REPO_CLONE.")
36
+ return False
37
+ max_turns = exp.get("max_turns_per_round", 25)
38
+
39
+ header("Experiment 5: Session Persistence & Context Decay")
40
+ info(f"Agent: {agent['name']} ({agent['binary']}) | Repo: {repo_root}")
41
+ info(f"Max turns/round: {max_turns} | Permissions: {'skip' if skip_permissions else 'ask'}")
42
+
43
+ session_id, _ = setup_mod.prepare_environment("5", "full", agent_name,
44
+ state_dir=state_dir)
45
+ log(f"Shared Session ID: {session_id}")
46
+
47
+ result_dir = results_dir / "5" if results_dir else OUTPUT_DIR / "5"
48
+ result_dir.mkdir(parents=True, exist_ok=True)
49
+
50
+ rounds_data = []
51
+ session_model = ""
52
+ deveco_session_id = ""
53
+ for round_num in range(1, 6):
54
+ ri = next((r for r in exp["rounds"] if r["round"] == round_num), None)
55
+ if not ri: continue
56
+
57
+ header(f"Round {round_num} of 5")
58
+ info(f"Title: {ri['title']}")
59
+
60
+ round_dir = result_dir / f"round_{round_num}"
61
+ round_dir.mkdir(parents=True, exist_ok=True)
62
+ (round_dir / "prompt.txt").write_text(ri["prompt"], encoding="utf-8")
63
+
64
+ start_ms = current_time_ms()
65
+ is_resume = round_num > 1
66
+
67
+ cmd = build_agent_cmd(agent, session_id, max_turns, ri["prompt"],
68
+ resume=is_resume, skip_permissions=skip_permissions,
69
+ repo_root=repo_root)
70
+
71
+ stream_path = round_dir / "stream_output.jsonl"
72
+ text_path = round_dir / "raw_output.txt"
73
+ track_hg = os.environ.get("EVAL_ARM", "") == "homegraph"
74
+
75
+ with open(stream_path, "w", encoding="utf-8") as stream_f:
76
+ result, mem = run_monitored_subprocess(
77
+ cmd, stdout=stream_f, stderr=subprocess.STDOUT, text=True,
78
+ track_homegraph=track_hg,
79
+ )
80
+ exit_code = result.returncode
81
+
82
+ end_ms = current_time_ms()
83
+ dur_ms = end_ms - start_ms
84
+ dur_s = f"{dur_ms / 1000:.1f}"
85
+
86
+ parser_name = agent.get("parser", "claude_stream_json")
87
+ stats = parse_output(stream_path, parser_name)
88
+ if not session_model:
89
+ session_model = resolve_agent_model(agent, stats)
90
+ if not deveco_session_id and stats.deveco_session_id:
91
+ deveco_session_id = stats.deveco_session_id
92
+
93
+ text_content = "\n".join(stats.text_lines)
94
+ text_path.write_text(text_content or "(no text output)", encoding="utf-8")
95
+
96
+ r2 = subprocess.run(
97
+ ["git", "diff", "--name-only"], capture_output=True, text=True, cwd=repo_root,
98
+ **subprocess_no_window_kwargs(),
99
+ )
100
+ mod_files = len([l for l in r2.stdout.strip().split("\n") if l])
101
+
102
+ rr = {"round": round_num, "title": ri["title"], "duration_ms": dur_ms,
103
+ "duration_s": dur_s, "exit_code": exit_code,
104
+ "tool_calls": stats.tool_calls, "tool_names": stats.tool_names,
105
+ "files_read": list(set(stats.files_read)),
106
+ "files_edited": list(set(stats.files_edited)),
107
+ "assistant_messages": stats.assistant_messages,
108
+ "max_turns_hit": stats.max_turns_hit,
109
+ "modified_files": mod_files, "max_turns": max_turns,
110
+ "memory": mem.to_dict()}
111
+ write_json(round_dir / "round_results.json", rr)
112
+ rounds_data.append(rr)
113
+
114
+ info(f"Round {round_num} done: {dur_s}s | Tool calls: {stats.tool_calls} | "
115
+ f"Files read: {len(set(stats.files_read))} | Edited: {mod_files}")
116
+ if stats.max_turns_hit:
117
+ warn(f" Max turns ({max_turns}) hit in round {round_num}")
118
+ if round_num < 5:
119
+ time.sleep(2)
120
+
121
+ summary = {
122
+ "experiment_id": "5", "session_id": session_id, "agent": agent["name"],
123
+ "model": session_model,
124
+ "deveco_session_id": deveco_session_id,
125
+ "total_rounds": len(rounds_data),
126
+ "total_duration_ms": sum(r["duration_ms"] for r in rounds_data),
127
+ "total_tool_calls": sum(int(r["tool_calls"]) for r in rounds_data),
128
+ "memory_retention_curve": [
129
+ {"round": r["round"], "tool_calls": int(r["tool_calls"]),
130
+ "files_read": r.get("files_read", [])} for r in rounds_data
131
+ ],
132
+ }
133
+ write_json(result_dir / "session_results.json", summary)
134
+
135
+ print("\nMemory Retention Curve:")
136
+ for r in rounds_data:
137
+ tc = int(r["tool_calls"])
138
+ fr = len(r.get("files_read", []))
139
+ print(f" Round {r['round']}: {'#' * (tc // 2)} ({tc} tool calls, {fr} files read)")
140
+
141
+ log(f"Results: {result_dir}/session_results.json")
142
+ return True
143
+
144
+
145
+ if __name__ == "__main__":
146
+ args = sys.argv[1:]
147
+ exp_id = args[0] if args else "5"
148
+ agent_name = ""
149
+ skip = True
150
+ i = 1
151
+ while i < len(args):
152
+ if args[i] == "--agent" and i + 1 < len(args):
153
+ agent_name = args[i + 1]; i += 2
154
+ elif args[i] == "--no-skip-permissions":
155
+ skip = False; i += 1
156
+ else:
157
+ i += 1
158
+ sys.exit(0 if run_session(exp_id, skip, agent_name) else 1)