sr-harness 1.0.0rc1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. sr_harness-1.0.0rc1/LICENSE +7 -0
  2. sr_harness-1.0.0rc1/PKG-INFO +391 -0
  3. sr_harness-1.0.0rc1/README.md +342 -0
  4. sr_harness-1.0.0rc1/pyproject.toml +97 -0
  5. sr_harness-1.0.0rc1/setup.cfg +4 -0
  6. sr_harness-1.0.0rc1/src/sr_harness/README.md +102 -0
  7. sr_harness-1.0.0rc1/src/sr_harness/__init__.py +31 -0
  8. sr_harness-1.0.0rc1/src/sr_harness/_vendor/README.zh.md +13 -0
  9. sr_harness-1.0.0rc1/src/sr_harness/_vendor/__init__.py +0 -0
  10. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/__init__.py +1 -0
  11. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/__init__.py +32 -0
  12. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/__init__.py +747 -0
  13. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/call_tool_template.py +63 -0
  14. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/readme_template.md +83 -0
  15. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/utils.py +41 -0
  16. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/functionevolve.py +288 -0
  17. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/linear.py +37 -0
  18. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/my_igsr.py +414 -0
  19. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/my_pysr.py +102 -0
  20. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/polynomial.py +57 -0
  21. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/pysr.py +105 -0
  22. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/sr_harness.py +212 -0
  23. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/sr_scientist.py +425 -0
  24. sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/core.py +81 -0
  25. sr_harness-1.0.0rc1/src/sr_harness/agents/__init__.py +14 -0
  26. sr_harness-1.0.0rc1/src/sr_harness/agents/agent.py +136 -0
  27. sr_harness-1.0.0rc1/src/sr_harness/agents/data_preparation_agent.py +315 -0
  28. sr_harness-1.0.0rc1/src/sr_harness/agents/evaluator_construction_agent.py +317 -0
  29. sr_harness-1.0.0rc1/src/sr_harness/agents/sr_agent.py +1370 -0
  30. sr_harness-1.0.0rc1/src/sr_harness/agents/sr_agent_interactive.py +655 -0
  31. sr_harness-1.0.0rc1/src/sr_harness/api/__init__.py +24 -0
  32. sr_harness-1.0.0rc1/src/sr_harness/api/base_api.py +188 -0
  33. sr_harness-1.0.0rc1/src/sr_harness/api/deepseek_api.py +102 -0
  34. sr_harness-1.0.0rc1/src/sr_harness/api/gemini_api.py +103 -0
  35. sr_harness-1.0.0rc1/src/sr_harness/api/lmstudio_api.py +161 -0
  36. sr_harness-1.0.0rc1/src/sr_harness/api/manual_api.py +60 -0
  37. sr_harness-1.0.0rc1/src/sr_harness/api/openai_api.py +349 -0
  38. sr_harness-1.0.0rc1/src/sr_harness/api/openrouter_api.py +297 -0
  39. sr_harness-1.0.0rc1/src/sr_harness/api/siliconflow_api.py +199 -0
  40. sr_harness-1.0.0rc1/src/sr_harness/cli/README.zh.md +5 -0
  41. sr_harness-1.0.0rc1/src/sr_harness/cli/__init__.py +78 -0
  42. sr_harness-1.0.0rc1/src/sr_harness/cli/benchmark.py +705 -0
  43. sr_harness-1.0.0rc1/src/sr_harness/cli/run.py +142 -0
  44. sr_harness-1.0.0rc1/src/sr_harness/cli/synthetic.py +356 -0
  45. sr_harness-1.0.0rc1/src/sr_harness/cli/tool.py +194 -0
  46. sr_harness-1.0.0rc1/src/sr_harness/core/__init__.py +40 -0
  47. sr_harness-1.0.0rc1/src/sr_harness/core/api.py +73 -0
  48. sr_harness-1.0.0rc1/src/sr_harness/core/context.py +319 -0
  49. sr_harness-1.0.0rc1/src/sr_harness/core/context_data.py +458 -0
  50. sr_harness-1.0.0rc1/src/sr_harness/core/search.py +668 -0
  51. sr_harness-1.0.0rc1/src/sr_harness/core/tool.py +59 -0
  52. sr_harness-1.0.0rc1/src/sr_harness/evaluator/__init__.py +15 -0
  53. sr_harness-1.0.0rc1/src/sr_harness/evaluator/default_evaluator.py +90 -0
  54. sr_harness-1.0.0rc1/src/sr_harness/evaluator/graph_evaluator.py +99 -0
  55. sr_harness-1.0.0rc1/src/sr_harness/evaluator/load_custom_evaluator.py +170 -0
  56. sr_harness-1.0.0rc1/src/sr_harness/evaluator/utils/__init__.py +342 -0
  57. sr_harness-1.0.0rc1/src/sr_harness/parser/__init__.py +12 -0
  58. sr_harness-1.0.0rc1/src/sr_harness/parser/base_parser.py +103 -0
  59. sr_harness-1.0.0rc1/src/sr_harness/parser/json_parser.py +110 -0
  60. sr_harness-1.0.0rc1/src/sr_harness/parser/openai_parser.py +72 -0
  61. sr_harness-1.0.0rc1/src/sr_harness/parser/text_parser.py +231 -0
  62. sr_harness-1.0.0rc1/src/sr_harness/parser/xml_parser.py +18 -0
  63. sr_harness-1.0.0rc1/src/sr_harness/runtime/__init__.py +24 -0
  64. sr_harness-1.0.0rc1/src/sr_harness/runtime/interaction_manager.py +559 -0
  65. sr_harness-1.0.0rc1/src/sr_harness/runtime/model_router.py +120 -0
  66. sr_harness-1.0.0rc1/src/sr_harness/skills/__init__.py +1 -0
  67. sr_harness-1.0.0rc1/src/sr_harness/skills/discover-symbolic-laws/SKILL.md +343 -0
  68. sr_harness-1.0.0rc1/src/sr_harness/skills/skill_manager.py +289 -0
  69. sr_harness-1.0.0rc1/src/sr_harness/tools/README.md +160 -0
  70. sr_harness-1.0.0rc1/src/sr_harness/tools/__init__.py +50 -0
  71. sr_harness-1.0.0rc1/src/sr_harness/tools/base_tool.py +907 -0
  72. sr_harness-1.0.0rc1/src/sr_harness/tools/call_llm.py +36 -0
  73. sr_harness-1.0.0rc1/src/sr_harness/tools/call_pysr.py +274 -0
  74. sr_harness-1.0.0rc1/src/sr_harness/tools/call_sindy.py +275 -0
  75. sr_harness-1.0.0rc1/src/sr_harness/tools/code_executor.py +810 -0
  76. sr_harness-1.0.0rc1/src/sr_harness/tools/constant_fit.py +225 -0
  77. sr_harness-1.0.0rc1/src/sr_harness/tools/create_skill.py +293 -0
  78. sr_harness-1.0.0rc1/src/sr_harness/tools/edit_skill.py +133 -0
  79. sr_harness-1.0.0rc1/src/sr_harness/tools/edit_tool.py +113 -0
  80. sr_harness-1.0.0rc1/src/sr_harness/tools/eic.py +320 -0
  81. sr_harness-1.0.0rc1/src/sr_harness/tools/evaluate_code.py +353 -0
  82. sr_harness-1.0.0rc1/src/sr_harness/tools/evaluate_formula.py +98 -0
  83. sr_harness-1.0.0rc1/src/sr_harness/tools/harmonic_interaction_fit.py +155 -0
  84. sr_harness-1.0.0rc1/src/sr_harness/tools/model_test.py +36 -0
  85. sr_harness-1.0.0rc1/src/sr_harness/tools/nd2.py +383 -0
  86. sr_harness-1.0.0rc1/src/sr_harness/tools/polynomial_fit.py +385 -0
  87. sr_harness-1.0.0rc1/src/sr_harness/tools/power_law_fit.py +320 -0
  88. sr_harness-1.0.0rc1/src/sr_harness/tools/predict_property.py +398 -0
  89. sr_harness-1.0.0rc1/src/sr_harness/tools/rational_fit.py +301 -0
  90. sr_harness-1.0.0rc1/src/sr_harness/tools/read_pdf.py +114 -0
  91. sr_harness-1.0.0rc1/src/sr_harness/tools/read_skill.py +119 -0
  92. sr_harness-1.0.0rc1/src/sr_harness/tools/read_source.py +209 -0
  93. sr_harness-1.0.0rc1/src/sr_harness/tools/relationship_analysis.py +329 -0
  94. sr_harness-1.0.0rc1/src/sr_harness/tools/sr4mdl.py +345 -0
  95. sr_harness-1.0.0rc1/src/sr_harness/tools/statistics_analysis.py +205 -0
  96. sr_harness-1.0.0rc1/src/sr_harness/tools/subagent.py +196 -0
  97. sr_harness-1.0.0rc1/src/sr_harness/tools/validate_context_data.py +114 -0
  98. sr_harness-1.0.0rc1/src/sr_harness/tools/validate_evaluator.py +67 -0
  99. sr_harness-1.0.0rc1/src/sr_harness/tools/web_research.py +191 -0
  100. sr_harness-1.0.0rc1/src/sr_harness/tools/workspace_code_executor.py +501 -0
  101. sr_harness-1.0.0rc1/src/sr_harness/tools/workspace_shell.py +1123 -0
  102. sr_harness-1.0.0rc1/src/sr_harness/utils/__init__.py +34 -0
  103. sr_harness-1.0.0rc1/src/sr_harness/utils/attr_dict.py +49 -0
  104. sr_harness-1.0.0rc1/src/sr_harness/utils/auto_gpu.py +81 -0
  105. sr_harness-1.0.0rc1/src/sr_harness/utils/classproperty.py +13 -0
  106. sr_harness-1.0.0rc1/src/sr_harness/utils/constant_optimizer.py +242 -0
  107. sr_harness-1.0.0rc1/src/sr_harness/utils/df_to_3line.py +48 -0
  108. sr_harness-1.0.0rc1/src/sr_harness/utils/factory_mixin.py +138 -0
  109. sr_harness-1.0.0rc1/src/sr_harness/utils/fix_parser.py +64 -0
  110. sr_harness-1.0.0rc1/src/sr_harness/utils/format_confusion_matrix.py +75 -0
  111. sr_harness-1.0.0rc1/src/sr_harness/utils/format_pareto_front.py +129 -0
  112. sr_harness-1.0.0rc1/src/sr_harness/utils/lazy_loader.py +35 -0
  113. sr_harness-1.0.0rc1/src/sr_harness/utils/log_exception.py +12 -0
  114. sr_harness-1.0.0rc1/src/sr_harness/utils/logger.py +268 -0
  115. sr_harness-1.0.0rc1/src/sr_harness/utils/metrics.py +30 -0
  116. sr_harness-1.0.0rc1/src/sr_harness/utils/model_store.py +154 -0
  117. sr_harness-1.0.0rc1/src/sr_harness/utils/nn/__init__.py +3 -0
  118. sr_harness-1.0.0rc1/src/sr_harness/utils/nn/gnn.py +113 -0
  119. sr_harness-1.0.0rc1/src/sr_harness/utils/nn/positional_encoding.py +22 -0
  120. sr_harness-1.0.0rc1/src/sr_harness/utils/parse_json_with_template.py +154 -0
  121. sr_harness-1.0.0rc1/src/sr_harness/utils/plot.py +369 -0
  122. sr_harness-1.0.0rc1/src/sr_harness/utils/render_markdown.py +23 -0
  123. sr_harness-1.0.0rc1/src/sr_harness/utils/render_python.py +18 -0
  124. sr_harness-1.0.0rc1/src/sr_harness/utils/sanitize_filename.py +6 -0
  125. sr_harness-1.0.0rc1/src/sr_harness/utils/save_args.py +77 -0
  126. sr_harness-1.0.0rc1/src/sr_harness/utils/serialize.py +35 -0
  127. sr_harness-1.0.0rc1/src/sr_harness/utils/symbolic_acc.py +291 -0
  128. sr_harness-1.0.0rc1/src/sr_harness/utils/tag2ansi.py +170 -0
  129. sr_harness-1.0.0rc1/src/sr_harness/utils/timing.py +283 -0
  130. sr_harness-1.0.0rc1/src/sr_harness/utils/utils.py +44 -0
  131. sr_harness-1.0.0rc1/src/sr_harness/web/__init__.py +2 -0
  132. sr_harness-1.0.0rc1/src/sr_harness/web/app.py +299 -0
  133. sr_harness-1.0.0rc1/src/sr_harness/web/conversations.py +478 -0
  134. sr_harness-1.0.0rc1/src/sr_harness/web/demo_data.py +139 -0
  135. sr_harness-1.0.0rc1/src/sr_harness/web/platform.py +1007 -0
  136. sr_harness-1.0.0rc1/src/sr_harness/web/session.py +1617 -0
  137. sr_harness-1.0.0rc1/src/sr_harness/web/static/context-data-guide.html +179 -0
  138. sr_harness-1.0.0rc1/src/sr_harness/web/static/data-agent-safety.html +109 -0
  139. sr_harness-1.0.0rc1/src/sr_harness/web/static/evaluator-guide.html +38 -0
  140. sr_harness-1.0.0rc1/src/sr_harness/web/static/favicon.svg +4 -0
  141. sr_harness-1.0.0rc1/src/sr_harness/web/static/index.html +2962 -0
  142. sr_harness-1.0.0rc1/src/sr_harness/web/static/platform.html +1327 -0
  143. sr_harness-1.0.0rc1/src/sr_harness.egg-info/PKG-INFO +391 -0
  144. sr_harness-1.0.0rc1/src/sr_harness.egg-info/SOURCES.txt +164 -0
  145. sr_harness-1.0.0rc1/src/sr_harness.egg-info/dependency_links.txt +1 -0
  146. sr_harness-1.0.0rc1/src/sr_harness.egg-info/entry_points.txt +2 -0
  147. sr_harness-1.0.0rc1/src/sr_harness.egg-info/requires.txt +42 -0
  148. sr_harness-1.0.0rc1/src/sr_harness.egg-info/top_level.txt +2 -0
  149. sr_harness-1.0.0rc1/src/sr_harness_engine/README.zh.md +282 -0
  150. sr_harness-1.0.0rc1/src/sr_harness_engine/__init__.py +104 -0
  151. sr_harness-1.0.0rc1/src/sr_harness_engine/analysis.py +127 -0
  152. sr_harness-1.0.0rc1/src/sr_harness_engine/context.py +37 -0
  153. sr_harness-1.0.0rc1/src/sr_harness_engine/desugar.py +90 -0
  154. sr_harness-1.0.0rc1/src/sr_harness_engine/evaluation.py +422 -0
  155. sr_harness-1.0.0rc1/src/sr_harness_engine/expression.py +530 -0
  156. sr_harness-1.0.0rc1/src/sr_harness_engine/indexed_evaluation.py +838 -0
  157. sr_harness-1.0.0rc1/src/sr_harness_engine/optimize.py +166 -0
  158. sr_harness-1.0.0rc1/src/sr_harness_engine/parser.py +281 -0
  159. sr_harness-1.0.0rc1/src/sr_harness_engine/render.py +132 -0
  160. sr_harness-1.0.0rc1/src/sr_harness_engine/tree.py +162 -0
  161. sr_harness-1.0.0rc1/tests/test_benchmark_default_tools.py +63 -0
  162. sr_harness-1.0.0rc1/tests/test_codex_usage.py +36 -0
  163. sr_harness-1.0.0rc1/tests/test_functionevolve.py +124 -0
  164. sr_harness-1.0.0rc1/tests/test_my_igsr.py +63 -0
  165. sr_harness-1.0.0rc1/tests/test_public_api_documentation.py +68 -0
  166. sr_harness-1.0.0rc1/tests/test_sr_agent_parallel.py +701 -0
@@ -0,0 +1,7 @@
1
+ Copyright (c) 2026-present, YuMeow.
2
+
3
+ Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
4
+
5
+ The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
6
+
7
+ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
@@ -0,0 +1,391 @@
1
+ Metadata-Version: 2.4
2
+ Name: sr-harness
3
+ Version: 1.0.0rc1
4
+ Summary: SRHarness: a harness for agentic symbolic regression
5
+ Author-email: YuMeow <yuzh19@tsinghua.org.cn>
6
+ License: MIT
7
+ Requires-Python: >=3.12
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: numpy>=1.24.0
11
+ Requires-Dist: scipy>=1.10.0
12
+ Requires-Dist: pandas>=2.0.0
13
+ Requires-Dist: openpyxl>=3.1.0
14
+ Requires-Dist: joblib>=1.3.0
15
+ Requires-Dist: docstring-parser>=0.16
16
+ Requires-Dist: pyyaml>=6.0
17
+ Requires-Dist: rich>=13.0.0
18
+ Requires-Dist: matplotlib>=3.7.0
19
+ Requires-Dist: openai>=1.0.0
20
+ Requires-Dist: google-genai>=1.0.0
21
+ Requires-Dist: requests>=2.28.0
22
+ Requires-Dist: python-dotenv>=1.0.0
23
+ Requires-Dist: h5py>=3.16.0
24
+ Requires-Dist: datasets>=4.8.0
25
+ Requires-Dist: textual>=1.0.0
26
+ Requires-Dist: prompt_toolkit>=3.0.0
27
+ Requires-Dist: fastapi>=0.115.0
28
+ Requires-Dist: uvicorn>=0.30.0
29
+ Provides-Extra: nn
30
+ Requires-Dist: torch>=2.0.0; extra == "nn"
31
+ Requires-Dist: torch-geometric>=2.4.0; extra == "nn"
32
+ Provides-Extra: tools
33
+ Requires-Dist: pysr>=1.5.10; extra == "tools"
34
+ Requires-Dist: pysindy>=2.1.0; extra == "tools"
35
+ Requires-Dist: pypdf>=5.0.0; extra == "tools"
36
+ Provides-Extra: dev
37
+ Requires-Dist: pytest>=8.0.0; extra == "dev"
38
+ Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
39
+ Requires-Dist: ipynbname>=2025.8.0.0; extra == "dev"
40
+ Requires-Dist: yumeow_plot>=0.1.2; extra == "dev"
41
+ Requires-Dist: mkdocs>=1.6; extra == "dev"
42
+ Requires-Dist: mkdocs-material>=9.5; extra == "dev"
43
+ Requires-Dist: mkdocstrings[python]>=0.27; extra == "dev"
44
+ Requires-Dist: mkdocs-static-i18n>=1.3; extra == "dev"
45
+ Requires-Dist: ruff>=0.8; extra == "dev"
46
+ Provides-Extra: all
47
+ Requires-Dist: sr-harness[dev,nn,tools]; extra == "all"
48
+ Dynamic: license-file
49
+
50
+ # SRHarness: A Harness for Agentic Symbolic Regression
51
+
52
+ [English](README.md) | [简体中文](README.zh.md) | [Complete SRHarness documentation](docs/index.md) | [SRHarness-Engine documentation](docs/engine.md)
53
+
54
+ SRHarness is a domain-specific runtime for **agentic symbolic regression**. It lets a large language model inspect numerical observations, choose scientific operations, evaluate competing hypotheses, and refine a symbolic expression over a long search trajectory.
55
+
56
+ This repository contains the research code for **“SRHarness: A Harness for Agentic Symbolic Regression.”** The Python package is `sr_harness`, and its primary agent class remains `SRAgent`.
57
+
58
+ > **Research-code status:** the project is under active development. Experiment-scale runs can make many paid LLM requests and may invoke external solvers. Start with a small `R-C-L-K` configuration and inspect the generated logs before launching a benchmark campaign.
59
+
60
+ ## Why SRHarness?
61
+
62
+ SRHarness organizes agentic equation discovery around three mechanisms:
63
+
64
+ - **Composable scientific actions.** Analysis, fitting, evaluation, and search tools share a common interface. Actions can operate on raw variables, transformed expressions, residuals, and other candidate-derived views.
65
+ - **Persistent scientific state.** Candidate formulas, numerical metrics, complexity, evidence, and provenance survive beyond a single conversation. Compact Pareto and top-candidate views expose useful state back to the model.
66
+ - **Trajectory lifecycle management.** A configurable `R-C-L-K` scheduler coordinates restarts, independent branches, refinement steps, and local response sampling while preserving useful intermediate results.
67
+
68
+ The included action library covers statistical and relationship analysis, formula/code evaluation, constant and structured fitting, PySR and SINDy integration, code execution, skills, and final formula submission. Tools can be enabled, disabled, or extended without changing the main agent loop.
69
+
70
+ ## Results at a Glance
71
+
72
+ The accompanying paper evaluates SRHarness on LLM-SRBench, including LSR-Synth, LSR-Transform, and an anonymized LSR-Transform variant that removes scientific descriptions and variable semantics.
73
+
74
+ | Method / backbone | LSR-Transform SA | LSR-Transform-Anon SA |
75
+ |---|---:|---:|
76
+ | SRHarness + DeepSeek-v4-flash-0731 | **93.69%** | **72.97%** |
77
+ | SR-Scientist + DeepSeek-v4-flash-0731 | 62.16% | 39.64% |
78
+ | Codex + DeepSeek-v4-flash-0731 | — | 20.72% |
79
+
80
+ These are symbolic-accuracy results reported in the manuscript. See the paper for the complete numerical, symbolic, complexity, resource, and ablation results, as well as the exact evaluation protocol.
81
+
82
+ ## Installation
83
+
84
+ ### Requirements
85
+
86
+ - Linux is the primary tested platform.
87
+ - Python **3.12 or newer** is required.
88
+ - Git and a working C/C++ toolchain are recommended.
89
+ - Some optional actions have additional requirements, such as Julia for PySR or PyTorch for neural components.
90
+
91
+ The following setup mirrors [`scripts/install.sh`](scripts/install.sh) while using HTTPS clone URLs:
92
+
93
+ ```bash
94
+ git clone https://github.com/yuzhTHU/SRHarness.git SRHarness
95
+ cd SRHarness
96
+
97
+ conda create -p ./venv python=3.12 -y
98
+ conda activate ./venv
99
+
100
+ # Core package plus development/test dependencies.
101
+ pip install -e ".[dev]"
102
+ ```
103
+
104
+ Install optional components as needed:
105
+
106
+ ```bash
107
+ pip install -e ".[tools]" # PySR, PySINDy, and PDF integrations
108
+ pip install -e ".[nn]" # Experimental neural components
109
+ pip install -e ".[dev]" # Tests, documentation, and development tools
110
+ pip install -e ".[all]" # All optional components
111
+ ```
112
+
113
+ ## Provider Configuration
114
+
115
+ Copy the tracked environment template once, then fill in only the providers you use:
116
+
117
+ ```bash
118
+ test -f .env || cp .env.sample .env
119
+ ```
120
+
121
+ For example, OpenRouter requires:
122
+
123
+ ```dotenv
124
+ OPENROUTER_API_KEY="sk-or-v1-..."
125
+ ```
126
+
127
+ The code also contains adapters for DeepSeek, Gemini, OpenAI/Azure OpenAI, SiliconFlow, LM Studio, and manual interaction. See [`.env.sample`](.env.sample) for the corresponding variables. Never commit `.env`; it is ignored by Git.
128
+
129
+ ## Quick Start
130
+
131
+ Run a small synthetic problem:
132
+
133
+ ```bash
134
+ conda activate ./venv
135
+
136
+ sr-harness synthetic \
137
+ --equation "y = sin(x1 - x2)" \
138
+ --x_low -10 \
139
+ --x_high 10 \
140
+ --llm_provider openrouter \
141
+ --llm_model deepseek/deepseek-v4-flash-0731 \
142
+ --force_initial_diagnostics \
143
+ -R 1 -C 1 -L 3 -K 1
144
+ ```
145
+
146
+ This command performs paid API calls. Its search budget is controlled by:
147
+
148
+ | Symbol | Meaning |
149
+ |---|---|
150
+ | `R` | restart rounds initialized from persistent historical candidates |
151
+ | `C` | independent conversational branches per restart |
152
+ | `L` | refinement steps per branch |
153
+ | `K` | locally sampled responses per refinement step |
154
+
155
+ The nominal number of model responses is approximately `R × C × L × K`, although retries and provider behavior can affect actual usage.
156
+
157
+ ### Python API
158
+
159
+ ```python
160
+ import numpy as np
161
+ from sr_harness import SRAgent
162
+
163
+ x1 = np.linspace(-3.0, 3.0, 100)
164
+ x2 = np.linspace(3.0, -3.0, 100)
165
+
166
+ agent = SRAgent(
167
+ llm_provider="openrouter",
168
+ llm_model="deepseek/deepseek-v4-flash-0731",
169
+ max_restart_loop=1,
170
+ global_width=1,
171
+ max_refinement_depth=3,
172
+ local_sample_size=1,
173
+ save_path="logs/python_api_demo",
174
+ )
175
+
176
+ result = agent.run(
177
+ X={"x1": x1, "x2": x2},
178
+ y={"y": np.sin(x1 - x2)},
179
+ problem_description="Discover y as a function of x1 and x2.",
180
+ )
181
+ best = result["candidates"][result["best_candidate"]]
182
+ print(best["formula"])
183
+ ```
184
+
185
+ ## LLM-SRBench Evaluation
186
+
187
+ Download the benchmark data. Git LFS may be required:
188
+
189
+ ```bash
190
+ git lfs install
191
+ git clone https://huggingface.co/datasets/nnheui/llm-srbench \
192
+ ./data/llm-srbench-data
193
+ ```
194
+
195
+ Run one LSR-Transform problem before scaling up:
196
+
197
+ ```bash
198
+ sr-harness benchmark \
199
+ --algorithm sr_harness \
200
+ --datasets lsrtransform \
201
+ --problem_names II.6.15b_1_0 \
202
+ --exp_name smoke_lsrtransform \
203
+ --llm_provider openrouter \
204
+ --llm_model deepseek/deepseek-v4-flash-0731 \
205
+ -R 1 -C 1 -L 3 -K 1
206
+ ```
207
+
208
+ Add `--anonymize` to replace variable names and scientific descriptions with generic input/output labels while leaving the numerical observations unchanged:
209
+
210
+ ```bash
211
+ sr-harness benchmark \
212
+ --algorithm sr_harness \
213
+ --datasets lsrtransform \
214
+ --problem_names II.6.15b_1_0 \
215
+ --exp_name smoke_lsrtransform_anon \
216
+ --anonymize \
217
+ --llm_provider openrouter \
218
+ --llm_model deepseek/deepseek-v4-flash-0731 \
219
+ -R 1 -C 1 -L 3 -K 1
220
+ ```
221
+
222
+ The benchmark entry point also contains adapters for conventional and LLM-based baselines; `sr-harness benchmark --help` lists its general options, while each adapter defines its method-specific flags. Paper-scale reproduction requires the exact model, toolset, data split, token limit, seed, and `R-C-L-K` configuration reported with each experiment; the smoke commands above intentionally use a much smaller budget.
223
+
224
+ ## Logs and Web Visualization
225
+
226
+ `SearchRunState` always keeps the live run in memory. When `save_path` is enabled, it also writes:
227
+
228
+ - `run.json`: the globally unique run ID and agent metadata;
229
+ - `nodes.jsonl`: search nodes, parent relations, prompts, actions, results, and usage;
230
+ - `result.json`: candidates plus the Pareto-front and best-candidate indices;
231
+ - `response.jsonl`: raw model responses and token/cost accounting;
232
+ - `tool_calls.jsonl`: tool invocations and outputs;
233
+ - text logs and entry-point-specific result files.
234
+
235
+ With `save_path=None`, search identity, parent relations, candidates, and results remain fully
236
+ available through `agent.run_state`, while no search-state files are created.
237
+
238
+ Launch the web viewer (included in the default installation):
239
+
240
+ ```bash
241
+ sr-harness run --save-dir logs/run --host 127.0.0.1 --port 8000
242
+ ```
243
+
244
+ `--workspace-dir` stores the conversation registry and one persistent workspace per conversation.
245
+ Without it, an explicit save path is reused as the workspace directory; if neither is provided,
246
+ the workbench uses a temporary directory and warns that conversation records may be lost. Mount
247
+ existing files or directories into every new conversation as read-only inputs when needed:
248
+
249
+ ```bash
250
+ sr-harness run --workspace-dir ./workspaces --mount ./data.csv ./papers --port 8000
251
+ ```
252
+
253
+ Mounted inputs must have unique basenames. They remain readable by the data-preparation Agent and
254
+ preview APIs, while workspace uploads and tools cannot modify their source contents. Ordinary files
255
+ inside each conversation workspace are writable and may be changed or deleted by AI-operated tools.
256
+ By default all browsers share the conversation list; pass `--isolate-users` to isolate visible
257
+ conversations by a persistent browser cookie. When `--save-path` or `--save-dir` supplies a durable
258
+ save path, the server periodically snapshots each materialized interactive session. Restarting with
259
+ the same path restores its timeline, settings, prepared context, evaluator selection, and search
260
+ records. Model or tool work that was still active at shutdown is restored as interrupted rather than
261
+ as a misleading live task. Snapshots are stored under `<save-path>/sessions/`.
262
+
263
+ Then open <http://127.0.0.1:8000/>. During a process lifetime the Web API and search viewer read the
264
+ active session's in-memory `SearchRunState`; durable session snapshots rebuild that state after a
265
+ server restart.
266
+
267
+ ![SRHarness data workbench](docs/assets/web-data-workbench.png)
268
+
269
+ ![SRHarness execution timeline](docs/assets/web-timeline.png)
270
+
271
+ The workbench opens on **Data Preparation**, where files, demo data, and the data-preparation Agent
272
+ share one view. **Data Analysis** selects a CSV or Excel workbook, assigns one target and one or
273
+ more features, edits variable descriptions, and previews X/Y/Hue/Size relationships. The
274
+ data-preparation Agent can inspect the persistent workspace, clean or join tables, search and read
275
+ public Web sources, and atomically
276
+ publish a numeric target and aligned features into the shared `AgentContext`. Its conversation and
277
+ workspace survive later requests. The direct structured-data workflow remains available without
278
+ using this Agent. SRHarness generates the initial system and user prompts from the resulting
279
+ configuration; either prompt remains editable before the run starts. The included `demo.csv`
280
+ contains three input columns (including one categorical column) and one numeric target.
281
+
282
+ During a run, **Timeline** shows model reasoning, tool calls, results, token/cost usage, and control
283
+ events, while **Current Context** exposes the messages associated with each R-C-L node. The search
284
+ tree and candidate panel stay linked to those nodes and can switch between all ranked candidates
285
+ and the Pareto front. Guidance, model changes, and pause requests take effect at safe operation
286
+ boundaries; a second pause request interrupts the active model or tool operation so the Agent can
287
+ reach that boundary sooner. A tool-free assistant reply naturally yields control to the user. The interface supports Chinese/English text, light/dark
288
+ themes, and resizable or collapsible side panels.
289
+
290
+ The data-preparation Agent and `SRAgentInteractive` keep separate message histories while sharing
291
+ one `AgentContext`. To add features during search, pause symbolic regression, ask the preparation
292
+ Agent to create and commit the aligned columns, then resume. At the next safe iteration boundary,
293
+ the SR Agent detects the new data revision, rebuilds its train/validation split, tells the existing
294
+ conversation which variables were added, and continues with its prior evidence and candidates.
295
+ While a search is active, data commits may add features but cannot alter the target, row alignment,
296
+ or previously used values; those changes require a new run because old candidate metrics would no
297
+ longer be comparable.
298
+
299
+ ### Research backends, subagents, and live control
300
+
301
+ The default tool set includes recursive per-subtree EIC diagnostics (`evaluate_eic`), an actual
302
+ MDLformer-guided SR4MDL search (`sr4mdl`), NDformer-guided network-dynamics search (`nd2`), bounded
303
+ symbolic-regression hypothesis/critique delegation (`delegate_subagent`), web search, and PDF
304
+ reading. Configure heavyweight external projects with `SR4MDL_HOME` and `ND2_HOME`, and point
305
+ `SR4MDL_CHECKPOINT` to the trained MDLformer checkpoint. Repositories placed at
306
+ `third-party/SR4MDL` and `third-party/ND2` are discovered automatically. When `evaluate_eic` is
307
+ enabled, each newly generated scalar candidate receives a lightweight structural audit whose
308
+ diagnostics are retained in candidate state. Documentation for EIC, SR4MDL, and ND2 is exposed as
309
+ runtime read-only skills by each tool's `get_doc()` method.
310
+
311
+ `Agent` contains the common API, parser, and tool-execution mechanics used by
312
+ `DataPreparationAgent` and `SRAgent`; `SRAgentInteractive` specializes the shared `SRAgent` search
313
+ loop with human control and frontend events. `AgentContext` owns the structured data and metadata,
314
+ evaluator, runtime arguments, and workspace shared by cooperating agents. Train/evaluation mappings
315
+ are lazily produced by the evaluator and cached by the context.
316
+
317
+ `SRAgentInteractive` accepts an `SRInteractionManager` that connects its search loop to an
318
+ interface. Data preparation and evaluator construction each use their own `InteractionManager`, so
319
+ their controls and timelines remain isolated. These managers own queued guidance, pause and
320
+ interrupt requests, safe-boundary coordination, and observable events; they do not own the
321
+ scientific search state or duplicate the R-C-L loop. Model auto-routing can use a cheap base backend for simple/early requests and an optional
322
+ strong backend for complex or stagnated searches. Configure
323
+ `strong_llm_provider`/`strong_llm_model`, or pass `auto_routing=False` to keep every request on the
324
+ base backend.
325
+
326
+ ## Evaluation and Reproducibility Notes
327
+
328
+ - Benchmark test observations are not exposed during search or candidate selection.
329
+ - The agent can reserve part of the visible training data for random or OOD-style validation using `--validation_fraction` and `--split_by`.
330
+ - Numerical predictions are evaluated through the shared benchmark pipeline. Symbolic equivalence is implemented in [`src/sr_harness/utils/symbolic_acc.py`](src/sr_harness/utils/symbolic_acc.py).
331
+ - Logs preserve prompts, model responses, tool calls, candidate provenance, token usage, and recorded cost so that a run can be audited after completion.
332
+ - API behavior, model aliases, prices, and stochastic outputs can change over time. Record the exact provider model identifier, source revision, arguments, and environment for serious comparisons.
333
+
334
+ ## Extending SRHarness
335
+
336
+ New scientific actions inherit `BaseTool`, declare stable metadata, and return a serializable result. Candidate-producing actions should use the shared evaluation contract so their formulas, train/validation metrics, complexity, diagnostics, and provenance can enter persistent scientific state consistently.
337
+
338
+ See:
339
+
340
+ - [`src/sr_harness/README.md`](src/sr_harness/README.md) for the agent loop and internal architecture;
341
+ - [`src/sr_harness/tools/README.md`](src/sr_harness/tools/README.md) for the action API and custom-tool guide;
342
+ - [`tests/README.md`](tests/README.md) for testing conventions.
343
+
344
+ ## Project Layout
345
+
346
+ ```text
347
+ ├── src/sr_harness/ # Python package
348
+ │ ├── agents/ # Batch and interactive SRAgent implementations
349
+ │ ├── api/ # BaseAPI and LLM provider adapters
350
+ │ ├── core/ # API, tool, candidate, node, and run-state structures
351
+ │ ├── interaction/ # Terminal and Web interaction managers
352
+ │ ├── runtime/ # Model routing and interaction control
353
+ │ ├── cli/ # sr-harness subcommands
354
+ │ ├── parser/ # Native/text/JSON/XML tool-call parsing
355
+ │ ├── tools/ # Scientific actions and shared evaluation contract
356
+ │ ├── skills/ # Reusable agent-facing scientific instructions
357
+ │ ├── web/ # Interactive workbench backend and static UI
358
+ │ ├── utils/ # Metrics, symbolic accuracy, logging, and utilities
359
+ │ └── _vendor/ # Integrated benchmark/baseline adapters
360
+ ├── tests/ # Unit and integration tests
361
+ ├── scripts/ # Experiment and analysis utilities
362
+ ├── analysis/ # Analysis notebooks
363
+ ├── data/ # Local datasets; ignored by Git
364
+ ├── logs/ # Run artifacts; ignored by Git
365
+ └── playground/ # Temporary experiments; ignored by Git
366
+ ```
367
+
368
+ Repository conventions:
369
+
370
+ - Add user-facing commands as `sr-harness` subcommands under `src/sr_harness/cli/`.
371
+ - Put experiment and analysis utilities under `scripts/`.
372
+ - Name analysis notebooks as `YYMMDD_description.ipynb` and avoid committing large outputs.
373
+ - Treat `data/`, `logs/`, and `playground/` as local working directories.
374
+
375
+ ## Testing
376
+
377
+ The default test configuration excludes tests marked `slow` or `paid`:
378
+
379
+ ```bash
380
+ python -m pytest tests/ -v
381
+ ```
382
+
383
+ Run paid or slow integration tests only when the required services and budget are available.
384
+
385
+ ## Citation
386
+
387
+ If you use this code, please cite **“SRHarness: A Harness for Agentic Symbolic Regression.”** A copy-ready BibTeX entry and public paper link will be added when the paper record becomes publicly available.
388
+
389
+ ## License
390
+
391
+ SRHarness is released under the [MIT License](LICENSE).