sr-harness 1.0.0rc1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sr_harness-1.0.0rc1/LICENSE +7 -0
- sr_harness-1.0.0rc1/PKG-INFO +391 -0
- sr_harness-1.0.0rc1/README.md +342 -0
- sr_harness-1.0.0rc1/pyproject.toml +97 -0
- sr_harness-1.0.0rc1/setup.cfg +4 -0
- sr_harness-1.0.0rc1/src/sr_harness/README.md +102 -0
- sr_harness-1.0.0rc1/src/sr_harness/__init__.py +31 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/README.zh.md +13 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/__init__.py +0 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/__init__.py +1 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/__init__.py +32 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/__init__.py +747 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/call_tool_template.py +63 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/readme_template.md +83 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/codex/utils.py +41 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/functionevolve.py +288 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/linear.py +37 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/my_igsr.py +414 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/my_pysr.py +102 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/polynomial.py +57 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/pysr.py +105 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/sr_harness.py +212 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/algorithms/sr_scientist.py +425 -0
- sr_harness-1.0.0rc1/src/sr_harness/_vendor/llmsr_bench/core.py +81 -0
- sr_harness-1.0.0rc1/src/sr_harness/agents/__init__.py +14 -0
- sr_harness-1.0.0rc1/src/sr_harness/agents/agent.py +136 -0
- sr_harness-1.0.0rc1/src/sr_harness/agents/data_preparation_agent.py +315 -0
- sr_harness-1.0.0rc1/src/sr_harness/agents/evaluator_construction_agent.py +317 -0
- sr_harness-1.0.0rc1/src/sr_harness/agents/sr_agent.py +1370 -0
- sr_harness-1.0.0rc1/src/sr_harness/agents/sr_agent_interactive.py +655 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/__init__.py +24 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/base_api.py +188 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/deepseek_api.py +102 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/gemini_api.py +103 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/lmstudio_api.py +161 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/manual_api.py +60 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/openai_api.py +349 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/openrouter_api.py +297 -0
- sr_harness-1.0.0rc1/src/sr_harness/api/siliconflow_api.py +199 -0
- sr_harness-1.0.0rc1/src/sr_harness/cli/README.zh.md +5 -0
- sr_harness-1.0.0rc1/src/sr_harness/cli/__init__.py +78 -0
- sr_harness-1.0.0rc1/src/sr_harness/cli/benchmark.py +705 -0
- sr_harness-1.0.0rc1/src/sr_harness/cli/run.py +142 -0
- sr_harness-1.0.0rc1/src/sr_harness/cli/synthetic.py +356 -0
- sr_harness-1.0.0rc1/src/sr_harness/cli/tool.py +194 -0
- sr_harness-1.0.0rc1/src/sr_harness/core/__init__.py +40 -0
- sr_harness-1.0.0rc1/src/sr_harness/core/api.py +73 -0
- sr_harness-1.0.0rc1/src/sr_harness/core/context.py +319 -0
- sr_harness-1.0.0rc1/src/sr_harness/core/context_data.py +458 -0
- sr_harness-1.0.0rc1/src/sr_harness/core/search.py +668 -0
- sr_harness-1.0.0rc1/src/sr_harness/core/tool.py +59 -0
- sr_harness-1.0.0rc1/src/sr_harness/evaluator/__init__.py +15 -0
- sr_harness-1.0.0rc1/src/sr_harness/evaluator/default_evaluator.py +90 -0
- sr_harness-1.0.0rc1/src/sr_harness/evaluator/graph_evaluator.py +99 -0
- sr_harness-1.0.0rc1/src/sr_harness/evaluator/load_custom_evaluator.py +170 -0
- sr_harness-1.0.0rc1/src/sr_harness/evaluator/utils/__init__.py +342 -0
- sr_harness-1.0.0rc1/src/sr_harness/parser/__init__.py +12 -0
- sr_harness-1.0.0rc1/src/sr_harness/parser/base_parser.py +103 -0
- sr_harness-1.0.0rc1/src/sr_harness/parser/json_parser.py +110 -0
- sr_harness-1.0.0rc1/src/sr_harness/parser/openai_parser.py +72 -0
- sr_harness-1.0.0rc1/src/sr_harness/parser/text_parser.py +231 -0
- sr_harness-1.0.0rc1/src/sr_harness/parser/xml_parser.py +18 -0
- sr_harness-1.0.0rc1/src/sr_harness/runtime/__init__.py +24 -0
- sr_harness-1.0.0rc1/src/sr_harness/runtime/interaction_manager.py +559 -0
- sr_harness-1.0.0rc1/src/sr_harness/runtime/model_router.py +120 -0
- sr_harness-1.0.0rc1/src/sr_harness/skills/__init__.py +1 -0
- sr_harness-1.0.0rc1/src/sr_harness/skills/discover-symbolic-laws/SKILL.md +343 -0
- sr_harness-1.0.0rc1/src/sr_harness/skills/skill_manager.py +289 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/README.md +160 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/__init__.py +50 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/base_tool.py +907 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/call_llm.py +36 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/call_pysr.py +274 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/call_sindy.py +275 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/code_executor.py +810 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/constant_fit.py +225 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/create_skill.py +293 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/edit_skill.py +133 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/edit_tool.py +113 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/eic.py +320 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/evaluate_code.py +353 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/evaluate_formula.py +98 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/harmonic_interaction_fit.py +155 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/model_test.py +36 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/nd2.py +383 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/polynomial_fit.py +385 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/power_law_fit.py +320 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/predict_property.py +398 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/rational_fit.py +301 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/read_pdf.py +114 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/read_skill.py +119 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/read_source.py +209 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/relationship_analysis.py +329 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/sr4mdl.py +345 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/statistics_analysis.py +205 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/subagent.py +196 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/validate_context_data.py +114 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/validate_evaluator.py +67 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/web_research.py +191 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/workspace_code_executor.py +501 -0
- sr_harness-1.0.0rc1/src/sr_harness/tools/workspace_shell.py +1123 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/__init__.py +34 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/attr_dict.py +49 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/auto_gpu.py +81 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/classproperty.py +13 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/constant_optimizer.py +242 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/df_to_3line.py +48 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/factory_mixin.py +138 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/fix_parser.py +64 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/format_confusion_matrix.py +75 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/format_pareto_front.py +129 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/lazy_loader.py +35 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/log_exception.py +12 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/logger.py +268 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/metrics.py +30 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/model_store.py +154 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/nn/__init__.py +3 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/nn/gnn.py +113 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/nn/positional_encoding.py +22 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/parse_json_with_template.py +154 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/plot.py +369 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/render_markdown.py +23 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/render_python.py +18 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/sanitize_filename.py +6 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/save_args.py +77 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/serialize.py +35 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/symbolic_acc.py +291 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/tag2ansi.py +170 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/timing.py +283 -0
- sr_harness-1.0.0rc1/src/sr_harness/utils/utils.py +44 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/__init__.py +2 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/app.py +299 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/conversations.py +478 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/demo_data.py +139 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/platform.py +1007 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/session.py +1617 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/static/context-data-guide.html +179 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/static/data-agent-safety.html +109 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/static/evaluator-guide.html +38 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/static/favicon.svg +4 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/static/index.html +2962 -0
- sr_harness-1.0.0rc1/src/sr_harness/web/static/platform.html +1327 -0
- sr_harness-1.0.0rc1/src/sr_harness.egg-info/PKG-INFO +391 -0
- sr_harness-1.0.0rc1/src/sr_harness.egg-info/SOURCES.txt +164 -0
- sr_harness-1.0.0rc1/src/sr_harness.egg-info/dependency_links.txt +1 -0
- sr_harness-1.0.0rc1/src/sr_harness.egg-info/entry_points.txt +2 -0
- sr_harness-1.0.0rc1/src/sr_harness.egg-info/requires.txt +42 -0
- sr_harness-1.0.0rc1/src/sr_harness.egg-info/top_level.txt +2 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/README.zh.md +282 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/__init__.py +104 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/analysis.py +127 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/context.py +37 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/desugar.py +90 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/evaluation.py +422 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/expression.py +530 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/indexed_evaluation.py +838 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/optimize.py +166 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/parser.py +281 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/render.py +132 -0
- sr_harness-1.0.0rc1/src/sr_harness_engine/tree.py +162 -0
- sr_harness-1.0.0rc1/tests/test_benchmark_default_tools.py +63 -0
- sr_harness-1.0.0rc1/tests/test_codex_usage.py +36 -0
- sr_harness-1.0.0rc1/tests/test_functionevolve.py +124 -0
- sr_harness-1.0.0rc1/tests/test_my_igsr.py +63 -0
- sr_harness-1.0.0rc1/tests/test_public_api_documentation.py +68 -0
- sr_harness-1.0.0rc1/tests/test_sr_agent_parallel.py +701 -0
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
Copyright (c) 2026-present, YuMeow.
|
|
2
|
+
|
|
3
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
|
4
|
+
|
|
5
|
+
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
|
|
6
|
+
|
|
7
|
+
THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sr-harness
|
|
3
|
+
Version: 1.0.0rc1
|
|
4
|
+
Summary: SRHarness: a harness for agentic symbolic regression
|
|
5
|
+
Author-email: YuMeow <yuzh19@tsinghua.org.cn>
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.12
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: numpy>=1.24.0
|
|
11
|
+
Requires-Dist: scipy>=1.10.0
|
|
12
|
+
Requires-Dist: pandas>=2.0.0
|
|
13
|
+
Requires-Dist: openpyxl>=3.1.0
|
|
14
|
+
Requires-Dist: joblib>=1.3.0
|
|
15
|
+
Requires-Dist: docstring-parser>=0.16
|
|
16
|
+
Requires-Dist: pyyaml>=6.0
|
|
17
|
+
Requires-Dist: rich>=13.0.0
|
|
18
|
+
Requires-Dist: matplotlib>=3.7.0
|
|
19
|
+
Requires-Dist: openai>=1.0.0
|
|
20
|
+
Requires-Dist: google-genai>=1.0.0
|
|
21
|
+
Requires-Dist: requests>=2.28.0
|
|
22
|
+
Requires-Dist: python-dotenv>=1.0.0
|
|
23
|
+
Requires-Dist: h5py>=3.16.0
|
|
24
|
+
Requires-Dist: datasets>=4.8.0
|
|
25
|
+
Requires-Dist: textual>=1.0.0
|
|
26
|
+
Requires-Dist: prompt_toolkit>=3.0.0
|
|
27
|
+
Requires-Dist: fastapi>=0.115.0
|
|
28
|
+
Requires-Dist: uvicorn>=0.30.0
|
|
29
|
+
Provides-Extra: nn
|
|
30
|
+
Requires-Dist: torch>=2.0.0; extra == "nn"
|
|
31
|
+
Requires-Dist: torch-geometric>=2.4.0; extra == "nn"
|
|
32
|
+
Provides-Extra: tools
|
|
33
|
+
Requires-Dist: pysr>=1.5.10; extra == "tools"
|
|
34
|
+
Requires-Dist: pysindy>=2.1.0; extra == "tools"
|
|
35
|
+
Requires-Dist: pypdf>=5.0.0; extra == "tools"
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: pytest>=8.0.0; extra == "dev"
|
|
38
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
39
|
+
Requires-Dist: ipynbname>=2025.8.0.0; extra == "dev"
|
|
40
|
+
Requires-Dist: yumeow_plot>=0.1.2; extra == "dev"
|
|
41
|
+
Requires-Dist: mkdocs>=1.6; extra == "dev"
|
|
42
|
+
Requires-Dist: mkdocs-material>=9.5; extra == "dev"
|
|
43
|
+
Requires-Dist: mkdocstrings[python]>=0.27; extra == "dev"
|
|
44
|
+
Requires-Dist: mkdocs-static-i18n>=1.3; extra == "dev"
|
|
45
|
+
Requires-Dist: ruff>=0.8; extra == "dev"
|
|
46
|
+
Provides-Extra: all
|
|
47
|
+
Requires-Dist: sr-harness[dev,nn,tools]; extra == "all"
|
|
48
|
+
Dynamic: license-file
|
|
49
|
+
|
|
50
|
+
# SRHarness: A Harness for Agentic Symbolic Regression
|
|
51
|
+
|
|
52
|
+
[English](README.md) | [简体中文](README.zh.md) | [Complete SRHarness documentation](docs/index.md) | [SRHarness-Engine documentation](docs/engine.md)
|
|
53
|
+
|
|
54
|
+
SRHarness is a domain-specific runtime for **agentic symbolic regression**. It lets a large language model inspect numerical observations, choose scientific operations, evaluate competing hypotheses, and refine a symbolic expression over a long search trajectory.
|
|
55
|
+
|
|
56
|
+
This repository contains the research code for **“SRHarness: A Harness for Agentic Symbolic Regression.”** The Python package is `sr_harness`, and its primary agent class remains `SRAgent`.
|
|
57
|
+
|
|
58
|
+
> **Research-code status:** the project is under active development. Experiment-scale runs can make many paid LLM requests and may invoke external solvers. Start with a small `R-C-L-K` configuration and inspect the generated logs before launching a benchmark campaign.
|
|
59
|
+
|
|
60
|
+
## Why SRHarness?
|
|
61
|
+
|
|
62
|
+
SRHarness organizes agentic equation discovery around three mechanisms:
|
|
63
|
+
|
|
64
|
+
- **Composable scientific actions.** Analysis, fitting, evaluation, and search tools share a common interface. Actions can operate on raw variables, transformed expressions, residuals, and other candidate-derived views.
|
|
65
|
+
- **Persistent scientific state.** Candidate formulas, numerical metrics, complexity, evidence, and provenance survive beyond a single conversation. Compact Pareto and top-candidate views expose useful state back to the model.
|
|
66
|
+
- **Trajectory lifecycle management.** A configurable `R-C-L-K` scheduler coordinates restarts, independent branches, refinement steps, and local response sampling while preserving useful intermediate results.
|
|
67
|
+
|
|
68
|
+
The included action library covers statistical and relationship analysis, formula/code evaluation, constant and structured fitting, PySR and SINDy integration, code execution, skills, and final formula submission. Tools can be enabled, disabled, or extended without changing the main agent loop.
|
|
69
|
+
|
|
70
|
+
## Results at a Glance
|
|
71
|
+
|
|
72
|
+
The accompanying paper evaluates SRHarness on LLM-SRBench, including LSR-Synth, LSR-Transform, and an anonymized LSR-Transform variant that removes scientific descriptions and variable semantics.
|
|
73
|
+
|
|
74
|
+
| Method / backbone | LSR-Transform SA | LSR-Transform-Anon SA |
|
|
75
|
+
|---|---:|---:|
|
|
76
|
+
| SRHarness + DeepSeek-v4-flash-0731 | **93.69%** | **72.97%** |
|
|
77
|
+
| SR-Scientist + DeepSeek-v4-flash-0731 | 62.16% | 39.64% |
|
|
78
|
+
| Codex + DeepSeek-v4-flash-0731 | — | 20.72% |
|
|
79
|
+
|
|
80
|
+
These are symbolic-accuracy results reported in the manuscript. See the paper for the complete numerical, symbolic, complexity, resource, and ablation results, as well as the exact evaluation protocol.
|
|
81
|
+
|
|
82
|
+
## Installation
|
|
83
|
+
|
|
84
|
+
### Requirements
|
|
85
|
+
|
|
86
|
+
- Linux is the primary tested platform.
|
|
87
|
+
- Python **3.12 or newer** is required.
|
|
88
|
+
- Git and a working C/C++ toolchain are recommended.
|
|
89
|
+
- Some optional actions have additional requirements, such as Julia for PySR or PyTorch for neural components.
|
|
90
|
+
|
|
91
|
+
The following setup mirrors [`scripts/install.sh`](scripts/install.sh) while using HTTPS clone URLs:
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
git clone https://github.com/yuzhTHU/SRHarness.git SRHarness
|
|
95
|
+
cd SRHarness
|
|
96
|
+
|
|
97
|
+
conda create -p ./venv python=3.12 -y
|
|
98
|
+
conda activate ./venv
|
|
99
|
+
|
|
100
|
+
# Core package plus development/test dependencies.
|
|
101
|
+
pip install -e ".[dev]"
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Install optional components as needed:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
pip install -e ".[tools]" # PySR, PySINDy, and PDF integrations
|
|
108
|
+
pip install -e ".[nn]" # Experimental neural components
|
|
109
|
+
pip install -e ".[dev]" # Tests, documentation, and development tools
|
|
110
|
+
pip install -e ".[all]" # All optional components
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
## Provider Configuration
|
|
114
|
+
|
|
115
|
+
Copy the tracked environment template once, then fill in only the providers you use:
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
test -f .env || cp .env.sample .env
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
For example, OpenRouter requires:
|
|
122
|
+
|
|
123
|
+
```dotenv
|
|
124
|
+
OPENROUTER_API_KEY="sk-or-v1-..."
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
The code also contains adapters for DeepSeek, Gemini, OpenAI/Azure OpenAI, SiliconFlow, LM Studio, and manual interaction. See [`.env.sample`](.env.sample) for the corresponding variables. Never commit `.env`; it is ignored by Git.
|
|
128
|
+
|
|
129
|
+
## Quick Start
|
|
130
|
+
|
|
131
|
+
Run a small synthetic problem:
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
conda activate ./venv
|
|
135
|
+
|
|
136
|
+
sr-harness synthetic \
|
|
137
|
+
--equation "y = sin(x1 - x2)" \
|
|
138
|
+
--x_low -10 \
|
|
139
|
+
--x_high 10 \
|
|
140
|
+
--llm_provider openrouter \
|
|
141
|
+
--llm_model deepseek/deepseek-v4-flash-0731 \
|
|
142
|
+
--force_initial_diagnostics \
|
|
143
|
+
-R 1 -C 1 -L 3 -K 1
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
This command performs paid API calls. Its search budget is controlled by:
|
|
147
|
+
|
|
148
|
+
| Symbol | Meaning |
|
|
149
|
+
|---|---|
|
|
150
|
+
| `R` | restart rounds initialized from persistent historical candidates |
|
|
151
|
+
| `C` | independent conversational branches per restart |
|
|
152
|
+
| `L` | refinement steps per branch |
|
|
153
|
+
| `K` | locally sampled responses per refinement step |
|
|
154
|
+
|
|
155
|
+
The nominal number of model responses is approximately `R × C × L × K`, although retries and provider behavior can affect actual usage.
|
|
156
|
+
|
|
157
|
+
### Python API
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
import numpy as np
|
|
161
|
+
from sr_harness import SRAgent
|
|
162
|
+
|
|
163
|
+
x1 = np.linspace(-3.0, 3.0, 100)
|
|
164
|
+
x2 = np.linspace(3.0, -3.0, 100)
|
|
165
|
+
|
|
166
|
+
agent = SRAgent(
|
|
167
|
+
llm_provider="openrouter",
|
|
168
|
+
llm_model="deepseek/deepseek-v4-flash-0731",
|
|
169
|
+
max_restart_loop=1,
|
|
170
|
+
global_width=1,
|
|
171
|
+
max_refinement_depth=3,
|
|
172
|
+
local_sample_size=1,
|
|
173
|
+
save_path="logs/python_api_demo",
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
result = agent.run(
|
|
177
|
+
X={"x1": x1, "x2": x2},
|
|
178
|
+
y={"y": np.sin(x1 - x2)},
|
|
179
|
+
problem_description="Discover y as a function of x1 and x2.",
|
|
180
|
+
)
|
|
181
|
+
best = result["candidates"][result["best_candidate"]]
|
|
182
|
+
print(best["formula"])
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
## LLM-SRBench Evaluation
|
|
186
|
+
|
|
187
|
+
Download the benchmark data. Git LFS may be required:
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
git lfs install
|
|
191
|
+
git clone https://huggingface.co/datasets/nnheui/llm-srbench \
|
|
192
|
+
./data/llm-srbench-data
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Run one LSR-Transform problem before scaling up:
|
|
196
|
+
|
|
197
|
+
```bash
|
|
198
|
+
sr-harness benchmark \
|
|
199
|
+
--algorithm sr_harness \
|
|
200
|
+
--datasets lsrtransform \
|
|
201
|
+
--problem_names II.6.15b_1_0 \
|
|
202
|
+
--exp_name smoke_lsrtransform \
|
|
203
|
+
--llm_provider openrouter \
|
|
204
|
+
--llm_model deepseek/deepseek-v4-flash-0731 \
|
|
205
|
+
-R 1 -C 1 -L 3 -K 1
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
Add `--anonymize` to replace variable names and scientific descriptions with generic input/output labels while leaving the numerical observations unchanged:
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
sr-harness benchmark \
|
|
212
|
+
--algorithm sr_harness \
|
|
213
|
+
--datasets lsrtransform \
|
|
214
|
+
--problem_names II.6.15b_1_0 \
|
|
215
|
+
--exp_name smoke_lsrtransform_anon \
|
|
216
|
+
--anonymize \
|
|
217
|
+
--llm_provider openrouter \
|
|
218
|
+
--llm_model deepseek/deepseek-v4-flash-0731 \
|
|
219
|
+
-R 1 -C 1 -L 3 -K 1
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
The benchmark entry point also contains adapters for conventional and LLM-based baselines; `sr-harness benchmark --help` lists its general options, while each adapter defines its method-specific flags. Paper-scale reproduction requires the exact model, toolset, data split, token limit, seed, and `R-C-L-K` configuration reported with each experiment; the smoke commands above intentionally use a much smaller budget.
|
|
223
|
+
|
|
224
|
+
## Logs and Web Visualization
|
|
225
|
+
|
|
226
|
+
`SearchRunState` always keeps the live run in memory. When `save_path` is enabled, it also writes:
|
|
227
|
+
|
|
228
|
+
- `run.json`: the globally unique run ID and agent metadata;
|
|
229
|
+
- `nodes.jsonl`: search nodes, parent relations, prompts, actions, results, and usage;
|
|
230
|
+
- `result.json`: candidates plus the Pareto-front and best-candidate indices;
|
|
231
|
+
- `response.jsonl`: raw model responses and token/cost accounting;
|
|
232
|
+
- `tool_calls.jsonl`: tool invocations and outputs;
|
|
233
|
+
- text logs and entry-point-specific result files.
|
|
234
|
+
|
|
235
|
+
With `save_path=None`, search identity, parent relations, candidates, and results remain fully
|
|
236
|
+
available through `agent.run_state`, while no search-state files are created.
|
|
237
|
+
|
|
238
|
+
Launch the web viewer (included in the default installation):
|
|
239
|
+
|
|
240
|
+
```bash
|
|
241
|
+
sr-harness run --save-dir logs/run --host 127.0.0.1 --port 8000
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
`--workspace-dir` stores the conversation registry and one persistent workspace per conversation.
|
|
245
|
+
Without it, an explicit save path is reused as the workspace directory; if neither is provided,
|
|
246
|
+
the workbench uses a temporary directory and warns that conversation records may be lost. Mount
|
|
247
|
+
existing files or directories into every new conversation as read-only inputs when needed:
|
|
248
|
+
|
|
249
|
+
```bash
|
|
250
|
+
sr-harness run --workspace-dir ./workspaces --mount ./data.csv ./papers --port 8000
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
Mounted inputs must have unique basenames. They remain readable by the data-preparation Agent and
|
|
254
|
+
preview APIs, while workspace uploads and tools cannot modify their source contents. Ordinary files
|
|
255
|
+
inside each conversation workspace are writable and may be changed or deleted by AI-operated tools.
|
|
256
|
+
By default all browsers share the conversation list; pass `--isolate-users` to isolate visible
|
|
257
|
+
conversations by a persistent browser cookie. When `--save-path` or `--save-dir` supplies a durable
|
|
258
|
+
save path, the server periodically snapshots each materialized interactive session. Restarting with
|
|
259
|
+
the same path restores its timeline, settings, prepared context, evaluator selection, and search
|
|
260
|
+
records. Model or tool work that was still active at shutdown is restored as interrupted rather than
|
|
261
|
+
as a misleading live task. Snapshots are stored under `<save-path>/sessions/`.
|
|
262
|
+
|
|
263
|
+
Then open <http://127.0.0.1:8000/>. During a process lifetime the Web API and search viewer read the
|
|
264
|
+
active session's in-memory `SearchRunState`; durable session snapshots rebuild that state after a
|
|
265
|
+
server restart.
|
|
266
|
+
|
|
267
|
+

|
|
268
|
+
|
|
269
|
+

|
|
270
|
+
|
|
271
|
+
The workbench opens on **Data Preparation**, where files, demo data, and the data-preparation Agent
|
|
272
|
+
share one view. **Data Analysis** selects a CSV or Excel workbook, assigns one target and one or
|
|
273
|
+
more features, edits variable descriptions, and previews X/Y/Hue/Size relationships. The
|
|
274
|
+
data-preparation Agent can inspect the persistent workspace, clean or join tables, search and read
|
|
275
|
+
public Web sources, and atomically
|
|
276
|
+
publish a numeric target and aligned features into the shared `AgentContext`. Its conversation and
|
|
277
|
+
workspace survive later requests. The direct structured-data workflow remains available without
|
|
278
|
+
using this Agent. SRHarness generates the initial system and user prompts from the resulting
|
|
279
|
+
configuration; either prompt remains editable before the run starts. The included `demo.csv`
|
|
280
|
+
contains three input columns (including one categorical column) and one numeric target.
|
|
281
|
+
|
|
282
|
+
During a run, **Timeline** shows model reasoning, tool calls, results, token/cost usage, and control
|
|
283
|
+
events, while **Current Context** exposes the messages associated with each R-C-L node. The search
|
|
284
|
+
tree and candidate panel stay linked to those nodes and can switch between all ranked candidates
|
|
285
|
+
and the Pareto front. Guidance, model changes, and pause requests take effect at safe operation
|
|
286
|
+
boundaries; a second pause request interrupts the active model or tool operation so the Agent can
|
|
287
|
+
reach that boundary sooner. A tool-free assistant reply naturally yields control to the user. The interface supports Chinese/English text, light/dark
|
|
288
|
+
themes, and resizable or collapsible side panels.
|
|
289
|
+
|
|
290
|
+
The data-preparation Agent and `SRAgentInteractive` keep separate message histories while sharing
|
|
291
|
+
one `AgentContext`. To add features during search, pause symbolic regression, ask the preparation
|
|
292
|
+
Agent to create and commit the aligned columns, then resume. At the next safe iteration boundary,
|
|
293
|
+
the SR Agent detects the new data revision, rebuilds its train/validation split, tells the existing
|
|
294
|
+
conversation which variables were added, and continues with its prior evidence and candidates.
|
|
295
|
+
While a search is active, data commits may add features but cannot alter the target, row alignment,
|
|
296
|
+
or previously used values; those changes require a new run because old candidate metrics would no
|
|
297
|
+
longer be comparable.
|
|
298
|
+
|
|
299
|
+
### Research backends, subagents, and live control
|
|
300
|
+
|
|
301
|
+
The default tool set includes recursive per-subtree EIC diagnostics (`evaluate_eic`), an actual
|
|
302
|
+
MDLformer-guided SR4MDL search (`sr4mdl`), NDformer-guided network-dynamics search (`nd2`), bounded
|
|
303
|
+
symbolic-regression hypothesis/critique delegation (`delegate_subagent`), web search, and PDF
|
|
304
|
+
reading. Configure heavyweight external projects with `SR4MDL_HOME` and `ND2_HOME`, and point
|
|
305
|
+
`SR4MDL_CHECKPOINT` to the trained MDLformer checkpoint. Repositories placed at
|
|
306
|
+
`third-party/SR4MDL` and `third-party/ND2` are discovered automatically. When `evaluate_eic` is
|
|
307
|
+
enabled, each newly generated scalar candidate receives a lightweight structural audit whose
|
|
308
|
+
diagnostics are retained in candidate state. Documentation for EIC, SR4MDL, and ND2 is exposed as
|
|
309
|
+
runtime read-only skills by each tool's `get_doc()` method.
|
|
310
|
+
|
|
311
|
+
`Agent` contains the common API, parser, and tool-execution mechanics used by
|
|
312
|
+
`DataPreparationAgent` and `SRAgent`; `SRAgentInteractive` specializes the shared `SRAgent` search
|
|
313
|
+
loop with human control and frontend events. `AgentContext` owns the structured data and metadata,
|
|
314
|
+
evaluator, runtime arguments, and workspace shared by cooperating agents. Train/evaluation mappings
|
|
315
|
+
are lazily produced by the evaluator and cached by the context.
|
|
316
|
+
|
|
317
|
+
`SRAgentInteractive` accepts an `SRInteractionManager` that connects its search loop to an
|
|
318
|
+
interface. Data preparation and evaluator construction each use their own `InteractionManager`, so
|
|
319
|
+
their controls and timelines remain isolated. These managers own queued guidance, pause and
|
|
320
|
+
interrupt requests, safe-boundary coordination, and observable events; they do not own the
|
|
321
|
+
scientific search state or duplicate the R-C-L loop. Model auto-routing can use a cheap base backend for simple/early requests and an optional
|
|
322
|
+
strong backend for complex or stagnated searches. Configure
|
|
323
|
+
`strong_llm_provider`/`strong_llm_model`, or pass `auto_routing=False` to keep every request on the
|
|
324
|
+
base backend.
|
|
325
|
+
|
|
326
|
+
## Evaluation and Reproducibility Notes
|
|
327
|
+
|
|
328
|
+
- Benchmark test observations are not exposed during search or candidate selection.
|
|
329
|
+
- The agent can reserve part of the visible training data for random or OOD-style validation using `--validation_fraction` and `--split_by`.
|
|
330
|
+
- Numerical predictions are evaluated through the shared benchmark pipeline. Symbolic equivalence is implemented in [`src/sr_harness/utils/symbolic_acc.py`](src/sr_harness/utils/symbolic_acc.py).
|
|
331
|
+
- Logs preserve prompts, model responses, tool calls, candidate provenance, token usage, and recorded cost so that a run can be audited after completion.
|
|
332
|
+
- API behavior, model aliases, prices, and stochastic outputs can change over time. Record the exact provider model identifier, source revision, arguments, and environment for serious comparisons.
|
|
333
|
+
|
|
334
|
+
## Extending SRHarness
|
|
335
|
+
|
|
336
|
+
New scientific actions inherit `BaseTool`, declare stable metadata, and return a serializable result. Candidate-producing actions should use the shared evaluation contract so their formulas, train/validation metrics, complexity, diagnostics, and provenance can enter persistent scientific state consistently.
|
|
337
|
+
|
|
338
|
+
See:
|
|
339
|
+
|
|
340
|
+
- [`src/sr_harness/README.md`](src/sr_harness/README.md) for the agent loop and internal architecture;
|
|
341
|
+
- [`src/sr_harness/tools/README.md`](src/sr_harness/tools/README.md) for the action API and custom-tool guide;
|
|
342
|
+
- [`tests/README.md`](tests/README.md) for testing conventions.
|
|
343
|
+
|
|
344
|
+
## Project Layout
|
|
345
|
+
|
|
346
|
+
```text
|
|
347
|
+
├── src/sr_harness/ # Python package
|
|
348
|
+
│ ├── agents/ # Batch and interactive SRAgent implementations
|
|
349
|
+
│ ├── api/ # BaseAPI and LLM provider adapters
|
|
350
|
+
│ ├── core/ # API, tool, candidate, node, and run-state structures
|
|
351
|
+
│ ├── interaction/ # Terminal and Web interaction managers
|
|
352
|
+
│ ├── runtime/ # Model routing and interaction control
|
|
353
|
+
│ ├── cli/ # sr-harness subcommands
|
|
354
|
+
│ ├── parser/ # Native/text/JSON/XML tool-call parsing
|
|
355
|
+
│ ├── tools/ # Scientific actions and shared evaluation contract
|
|
356
|
+
│ ├── skills/ # Reusable agent-facing scientific instructions
|
|
357
|
+
│ ├── web/ # Interactive workbench backend and static UI
|
|
358
|
+
│ ├── utils/ # Metrics, symbolic accuracy, logging, and utilities
|
|
359
|
+
│ └── _vendor/ # Integrated benchmark/baseline adapters
|
|
360
|
+
├── tests/ # Unit and integration tests
|
|
361
|
+
├── scripts/ # Experiment and analysis utilities
|
|
362
|
+
├── analysis/ # Analysis notebooks
|
|
363
|
+
├── data/ # Local datasets; ignored by Git
|
|
364
|
+
├── logs/ # Run artifacts; ignored by Git
|
|
365
|
+
└── playground/ # Temporary experiments; ignored by Git
|
|
366
|
+
```
|
|
367
|
+
|
|
368
|
+
Repository conventions:
|
|
369
|
+
|
|
370
|
+
- Add user-facing commands as `sr-harness` subcommands under `src/sr_harness/cli/`.
|
|
371
|
+
- Put experiment and analysis utilities under `scripts/`.
|
|
372
|
+
- Name analysis notebooks as `YYMMDD_description.ipynb` and avoid committing large outputs.
|
|
373
|
+
- Treat `data/`, `logs/`, and `playground/` as local working directories.
|
|
374
|
+
|
|
375
|
+
## Testing
|
|
376
|
+
|
|
377
|
+
The default test configuration excludes tests marked `slow` or `paid`:
|
|
378
|
+
|
|
379
|
+
```bash
|
|
380
|
+
python -m pytest tests/ -v
|
|
381
|
+
```
|
|
382
|
+
|
|
383
|
+
Run paid or slow integration tests only when the required services and budget are available.
|
|
384
|
+
|
|
385
|
+
## Citation
|
|
386
|
+
|
|
387
|
+
If you use this code, please cite **“SRHarness: A Harness for Agentic Symbolic Regression.”** A copy-ready BibTeX entry and public paper link will be added when the paper record becomes publicly available.
|
|
388
|
+
|
|
389
|
+
## License
|
|
390
|
+
|
|
391
|
+
SRHarness is released under the [MIT License](LICENSE).
|