darwinagent 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. darwinagent-0.1.0/.gitignore +28 -0
  2. darwinagent-0.1.0/LICENSE +21 -0
  3. darwinagent-0.1.0/PKG-INFO +96 -0
  4. darwinagent-0.1.0/README.md +178 -0
  5. darwinagent-0.1.0/README.pypi.md +60 -0
  6. darwinagent-0.1.0/README.zh-CN.md +178 -0
  7. darwinagent-0.1.0/pyproject.toml +98 -0
  8. darwinagent-0.1.0/src/darwinagent/__init__.py +49 -0
  9. darwinagent-0.1.0/src/darwinagent/__main__.py +3 -0
  10. darwinagent-0.1.0/src/darwinagent/agents/__init__.py +4 -0
  11. darwinagent-0.1.0/src/darwinagent/agents/answer.py +441 -0
  12. darwinagent-0.1.0/src/darwinagent/agents/extraction.py +616 -0
  13. darwinagent-0.1.0/src/darwinagent/agents/protocol.py +202 -0
  14. darwinagent-0.1.0/src/darwinagent/cli.py +363 -0
  15. darwinagent-0.1.0/src/darwinagent/config.py +189 -0
  16. darwinagent-0.1.0/src/darwinagent/contracts.py +514 -0
  17. darwinagent-0.1.0/src/darwinagent/demo/__init__.py +232 -0
  18. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/C/subject.py +5 -0
  19. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/F/lookup.py +2 -0
  20. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/P/answer.txt +1 -0
  21. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/P/extract.txt +1 -0
  22. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/P/review.txt +1 -0
  23. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/P/tools.txt +1 -0
  24. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/S/schema.yaml +6 -0
  25. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/assets/index.yaml +15 -0
  26. darwinagent-0.1.0/src/darwinagent/demo/device_maintenance/task.yaml +10 -0
  27. darwinagent-0.1.0/src/darwinagent/engine/__init__.py +3 -0
  28. darwinagent-0.1.0/src/darwinagent/engine/pipeline.py +883 -0
  29. darwinagent-0.1.0/src/darwinagent/experiments/__init__.py +12 -0
  30. darwinagent-0.1.0/src/darwinagent/experiments/admission.py +171 -0
  31. darwinagent-0.1.0/src/darwinagent/experiments/admission_checks.py +181 -0
  32. darwinagent-0.1.0/src/darwinagent/experiments/admission_composition.py +95 -0
  33. darwinagent-0.1.0/src/darwinagent/experiments/admission_functions.py +363 -0
  34. darwinagent-0.1.0/src/darwinagent/experiments/admission_reporting.py +52 -0
  35. darwinagent-0.1.0/src/darwinagent/experiments/admission_rules.py +31 -0
  36. darwinagent-0.1.0/src/darwinagent/experiments/admission_samples.py +345 -0
  37. darwinagent-0.1.0/src/darwinagent/experiments/admission_worker.py +192 -0
  38. darwinagent-0.1.0/src/darwinagent/experiments/bootstrap.py +541 -0
  39. darwinagent-0.1.0/src/darwinagent/experiments/campaign.py +542 -0
  40. darwinagent-0.1.0/src/darwinagent/experiments/constants.py +6 -0
  41. darwinagent-0.1.0/src/darwinagent/experiments/control.py +349 -0
  42. darwinagent-0.1.0/src/darwinagent/experiments/feedback.py +526 -0
  43. darwinagent-0.1.0/src/darwinagent/experiments/graph_trials.py +186 -0
  44. darwinagent-0.1.0/src/darwinagent/experiments/lifecycle.py +340 -0
  45. darwinagent-0.1.0/src/darwinagent/experiments/optimization.py +605 -0
  46. darwinagent-0.1.0/src/darwinagent/experiments/policy.py +65 -0
  47. darwinagent-0.1.0/src/darwinagent/experiments/proposal.py +142 -0
  48. darwinagent-0.1.0/src/darwinagent/experiments/proposal_session.py +385 -0
  49. darwinagent-0.1.0/src/darwinagent/experiments/recovery.py +291 -0
  50. darwinagent-0.1.0/src/darwinagent/experiments/rounds.py +689 -0
  51. darwinagent-0.1.0/src/darwinagent/experiments/runner.py +636 -0
  52. darwinagent-0.1.0/src/darwinagent/experiments/selected_operations.py +485 -0
  53. darwinagent-0.1.0/src/darwinagent/experiments/snapshots.py +91 -0
  54. darwinagent-0.1.0/src/darwinagent/experiments/spec.py +152 -0
  55. darwinagent-0.1.0/src/darwinagent/experiments/stages.py +801 -0
  56. darwinagent-0.1.0/src/darwinagent/experiments/statistics.py +52 -0
  57. darwinagent-0.1.0/src/darwinagent/experiments/trials.py +121 -0
  58. darwinagent-0.1.0/src/darwinagent/experiments/wiki.py +404 -0
  59. darwinagent-0.1.0/src/darwinagent/experiments/wiki_context.py +168 -0
  60. darwinagent-0.1.0/src/darwinagent/experiments/wiki_evidence.py +306 -0
  61. darwinagent-0.1.0/src/darwinagent/experiments/wiki_lessons.py +258 -0
  62. darwinagent-0.1.0/src/darwinagent/experiments/wiki_service.py +613 -0
  63. darwinagent-0.1.0/src/darwinagent/kernel/__init__.py +6 -0
  64. darwinagent-0.1.0/src/darwinagent/kernel/assets.py +221 -0
  65. darwinagent-0.1.0/src/darwinagent/kernel/checks.py +355 -0
  66. darwinagent-0.1.0/src/darwinagent/kernel/counterexamples.py +239 -0
  67. darwinagent-0.1.0/src/darwinagent/kernel/execution.py +65 -0
  68. darwinagent-0.1.0/src/darwinagent/kernel/functions.py +131 -0
  69. darwinagent-0.1.0/src/darwinagent/kernel/registration.py +25 -0
  70. darwinagent-0.1.0/src/darwinagent/kernel/revision.py +175 -0
  71. darwinagent-0.1.0/src/darwinagent/kernel/spec.py +208 -0
  72. darwinagent-0.1.0/src/darwinagent/kernel/validation.py +272 -0
  73. darwinagent-0.1.0/src/darwinagent/kg/__init__.py +33 -0
  74. darwinagent-0.1.0/src/darwinagent/kg/assembler.py +580 -0
  75. darwinagent-0.1.0/src/darwinagent/kg/graph.py +308 -0
  76. darwinagent-0.1.0/src/darwinagent/llm/__init__.py +13 -0
  77. darwinagent-0.1.0/src/darwinagent/llm/client.py +512 -0
  78. darwinagent-0.1.0/src/darwinagent/llm/recorded.py +23 -0
  79. darwinagent-0.1.0/src/darwinagent/llm/registry.py +159 -0
  80. darwinagent-0.1.0/src/darwinagent/llm/settings.py +46 -0
  81. darwinagent-0.1.0/src/darwinagent/operators/__init__.py +4 -0
  82. darwinagent-0.1.0/src/darwinagent/operators/calendar.py +220 -0
  83. darwinagent-0.1.0/src/darwinagent/operators/data.py +205 -0
  84. darwinagent-0.1.0/src/darwinagent/operators/dates.py +318 -0
  85. darwinagent-0.1.0/src/darwinagent/operators/sandbox.py +536 -0
  86. darwinagent-0.1.0/src/darwinagent/presets.py +13 -0
  87. darwinagent-0.1.0/src/darwinagent/runtime/__init__.py +5 -0
  88. darwinagent-0.1.0/src/darwinagent/runtime/artifacts.py +40 -0
  89. darwinagent-0.1.0/src/darwinagent/runtime/budgets.py +39 -0
  90. darwinagent-0.1.0/src/darwinagent/runtime/continuation.py +218 -0
  91. darwinagent-0.1.0/src/darwinagent/runtime/deadline.py +25 -0
  92. darwinagent-0.1.0/src/darwinagent/runtime/execution.py +114 -0
  93. darwinagent-0.1.0/src/darwinagent/runtime/execution_cli.py +47 -0
  94. darwinagent-0.1.0/src/darwinagent/runtime/identity.py +67 -0
  95. darwinagent-0.1.0/src/darwinagent/runtime/journal_client.py +64 -0
  96. darwinagent-0.1.0/src/darwinagent/runtime/leases.py +46 -0
  97. darwinagent-0.1.0/src/darwinagent/runtime/migration.py +719 -0
  98. darwinagent-0.1.0/src/darwinagent/runtime/steps.py +184 -0
  99. darwinagent-0.1.0/src/darwinagent/runtime/workspace.py +600 -0
  100. darwinagent-0.1.0/src/darwinagent/schema/__init__.py +12 -0
  101. darwinagent-0.1.0/src/darwinagent/schema/graphcheck.py +69 -0
  102. darwinagent-0.1.0/src/darwinagent/schema/model.py +298 -0
  103. darwinagent-0.1.0/src/darwinagent/schema/owlcheck.py +322 -0
  104. darwinagent-0.1.0/src/darwinagent/vector/__init__.py +8 -0
  105. darwinagent-0.1.0/src/darwinagent/vector/embedder.py +128 -0
  106. darwinagent-0.1.0/src/darwinagent/vector/index.py +27 -0
  107. darwinagent-0.1.0/src/darwinagent/vector/store.py +118 -0
@@ -0,0 +1,28 @@
1
+ .env
2
+ __pycache__/
3
+ *.pyc
4
+ .venv/
5
+ third_party/
6
+ *.egg-info/
7
+ .DS_Store
8
+ .jdk/
9
+ .env.bak*
10
+
11
+ # ---- 框架净化(2026-10-07,操作者拍板):仓库只保留框架代码/文档/任务适配与基准数据 ----
12
+ # 一切实验运行产物与快照种子不入库(磁盘保留+HF 归档承载)
13
+ datasets/*/runs/
14
+ datasets/locomo/snapshots/
15
+ datasets/loomo_runs_typo_check/
16
+ runs/
17
+ embed_cache.json
18
+ embed_cache.tmp
19
+
20
+ # ---- 构建产物与本地参考资料 ----
21
+ dist/
22
+ papers/
23
+
24
+ # Local packaging and test artifacts
25
+ /build/
26
+ .coverage
27
+ .pytest_cache/
28
+ htmlcov/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DarwinAgent contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,96 @@
1
+ Metadata-Version: 2.5
2
+ Name: darwinagent
3
+ Version: 0.1.0
4
+ Summary: Evolution for the Agent Era: experience-driven recursive self-improvement for agents
5
+ Project-URL: Repository, https://github.com/gogoingai/DarwinAgent
6
+ Project-URL: Documentation, https://github.com/gogoingai/DarwinAgent#readme
7
+ Project-URL: Issues, https://github.com/gogoingai/DarwinAgent/issues
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: agent-memory,agents,darwinagent,recursive-self-improvement
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Requires-Python: >=3.11
17
+ Requires-Dist: httpx>=0.27
18
+ Requires-Dist: networkx>=3.6.1
19
+ Requires-Dist: openai>=1.40
20
+ Requires-Dist: python-dotenv>=1.0
21
+ Requires-Dist: pyyaml>=6.0
22
+ Provides-Extra: benchmarks
23
+ Requires-Dist: datasets>=2.20; extra == 'benchmarks'
24
+ Requires-Dist: gradio>=6.0.0; extra == 'benchmarks'
25
+ Requires-Dist: numpy>=1.26; extra == 'benchmarks'
26
+ Requires-Dist: pandas>=2.1; extra == 'benchmarks'
27
+ Requires-Dist: requests>=2.31; extra == 'benchmarks'
28
+ Requires-Dist: rich>=13.0; extra == 'benchmarks'
29
+ Requires-Dist: tqdm>=4.66; extra == 'benchmarks'
30
+ Provides-Extra: formal
31
+ Requires-Dist: owlready2>=0.46; extra == 'formal'
32
+ Provides-Extra: vector
33
+ Requires-Dist: numpy>=1.26; extra == 'vector'
34
+ Requires-Dist: requests>=2.31; extra == 'vector'
35
+ Description-Content-Type: text/markdown
36
+
37
+ # DarwinAgent
38
+
39
+ **Evolution for the Agent Era** · **Agent 时代的进化**
40
+
41
+ An open framework for experience-driven recursive self-improvement of AI agents.
42
+ DarwinAgent takes its name from Charles Darwin and the theory of evolution:
43
+ propose changes to task assets, select through independent evaluation, retain
44
+ effective versions, and use a persistent experience Wiki to inform subsequent
45
+ proposals.
46
+
47
+ **Version 0.1.0 is experimental.** Improvement is an outcome to measure, and a
48
+ candidate can be rejected. The framework does not train model weights or rewrite
49
+ its own optimizer.
50
+
51
+ ## Install and try the loop
52
+
53
+ Python 3.11 or newer is required. In your Python environment:
54
+
55
+ ```bash
56
+ python -m pip install darwinagent
57
+ darwinagent --version
58
+ darwinagent doctor --output runs/doctor
59
+ darwinagent demo --mode replay --rounds 2 --output runs/demo-replay
60
+ darwinagent demo --mode replay --rounds 2 --output runs/demo-replay --resume
61
+ ```
62
+
63
+ Replay requires no API key or network. It executes the actual controller, task
64
+ runtime, evaluation, admission, Wiki, and asset publication with scripted model
65
+ responses. In one tiny synthetic maintenance task, the first candidate is accepted
66
+ and the second is rejected on a tie; HTTP attempts are zero. This demonstrates
67
+ the mechanism and does not establish model learning or benchmark gains.
68
+
69
+ For real model execution, set `DARWINAGENT_API_KEY`, `DARWINAGENT_BASE_URL`, and
70
+ `DARWINAGENT_MODEL` to your OpenAI-compatible Chat Completions endpoint, then run:
71
+
72
+ ```bash
73
+ darwinagent demo --mode live --rounds 2 --output runs/demo-live \
74
+ --max-requests 40 --timeout 1800
75
+ ```
76
+
77
+ Live mode calls the configured model. The request cap counts dispatched HTTP
78
+ attempts, including retries, and the timeout covers the whole run.
79
+
80
+ ## Bring your own task
81
+
82
+ Register S/F/C/P assets: schemas, query functions, task checks, and role prompts.
83
+ Connect a dataset adapter and an independent evaluator to the shared `Pipeline`
84
+ and `ExperimentRunner`. The framework's executor, evaluator, permissions, and
85
+ adoption policy remain outside the proposal boundary.
86
+
87
+ ## Documentation
88
+
89
+ - [English README](https://github.com/gogoingai/DarwinAgent/blob/main/README.md)
90
+ - [中文说明](https://github.com/gogoingai/DarwinAgent/blob/main/README.zh-CN.md)
91
+ - [Quickstart](https://github.com/gogoingai/DarwinAgent/blob/main/docs/en/quickstart.md)
92
+ - [中文快速开始](https://github.com/gogoingai/DarwinAgent/blob/main/docs/zh-CN/quickstart.md)
93
+ - [Acceptance evidence](https://github.com/gogoingai/DarwinAgent/blob/main/docs/acceptance/2026-10-07.md)
94
+ - [Source and issues](https://github.com/gogoingai/DarwinAgent)
95
+
96
+ DarwinAgent is distributed under the MIT license.
@@ -0,0 +1,178 @@
1
+ <p align="center"><img src="docs/assets/hero-en.svg" alt="DarwinAgent — Evolution for the Agent Era" width="100%"></p>
2
+
3
+ # DarwinAgent
4
+
5
+ **An Open Framework for Experience-Driven Recursive Self-Improvement**
6
+
7
+ [English](README.md) · [简体中文](README.zh-CN.md)
8
+
9
+ [![Offline framework checks](https://github.com/gogoingai/DarwinAgent/actions/workflows/ci.yml/badge.svg)](https://github.com/gogoingai/DarwinAgent/actions/workflows/ci.yml)
10
+ [![Version](https://img.shields.io/badge/version-0.1.0%20experimental-45635c)](CHANGELOG.md)
11
+ [![Python](https://img.shields.io/badge/python-%3E%3D3.11-45635c)](pyproject.toml)
12
+ [![License: MIT](https://img.shields.io/badge/license-MIT-45635c)](LICENSE)
13
+
14
+ DarwinAgent takes its name from Charles Darwin and the theory of evolution. **Evolution for the Agent Era** is its mission: help agents adapt through experience and retain effective capabilities. In this framework, variation comes from proposed task assets, selection comes from independent evaluation and fixed admission rules, and retention comes from versioned assets and a persistent experience Wiki that informs subsequent proposals.
15
+
16
+ DarwinAgent **0.1.0 is experimental**. It implements an inspectable improvement loop around a shared graph-based agent runtime. Improvement is an outcome to measure; a candidate can be rejected. The current scope is task assets, with fixed framework execution and evaluation boundaries.
17
+
18
+ ## Why DarwinAgent
19
+
20
+ A useful answer is one run. A useful capability should survive the next run. DarwinAgent records what a candidate changed, which sources supported an answer, how it was evaluated, why it was accepted or rejected, and what experience reaches the next proposal. This makes adaptation a reproducible experiment rather than an untracked prompt edit.
21
+
22
+ ## The evolution loop
23
+
24
+ ![Evolution loop](docs/assets/evolution-loop-en.svg)
25
+
26
+ 1. **Run:** execute a baseline through the common `Pipeline` and score its actual outputs.
27
+ 2. **Vary:** propose bounded changes to S/F/C/P assets using training evidence and Wiki feedback.
28
+ 3. **Select:** validate contracts and capabilities, run the candidate, and apply the frozen adoption policy.
29
+ 4. **Retain:** publish accepted versions atomically; preserve rejection facts and feed experience into the next proposal.
30
+
31
+ The kernel, evaluator, permissions, and adoption rules stay outside the proposal boundary. DarwinAgent does not train model weights or rewrite its own optimizer. Its name describes the inspiration, not a claim to implement a genetic algorithm.
32
+
33
+ ## What is implemented
34
+
35
+ | Capability | v0.1.0 behavior |
36
+ | --- | --- |
37
+ | Shared task runtime | `ExtractionAgent` → attributed graph → `AnswerAgent`, through one `Pipeline` |
38
+ | Bounded task assets | **S** schemas, **F** query functions, **C** task checks, **P** role prompts |
39
+ | Experience | Wiki stores observations and accepted/rejected decisions for subsequent proposals |
40
+ | Reproducibility | Content identities, separate provenance, scoped resume and optional strict comparison |
41
+ | Evaluation | Separate `DatasetAdapter` and `Evaluator`; metric names supplied by the task |
42
+ | Experiment control | `ExperimentRunner`; separate `CampaignController` for train/validation/test protocols |
43
+ | Model connection | One explicit OpenAI-compatible Chat Completions endpoint by default; optional tier overrides |
44
+ | Entry points | Installed CLI, Python SDK, offline replay, live mode, maintenance task example |
45
+
46
+ ## Try the loop
47
+
48
+ Install **from this source checkout**. DarwinAgent 0.1.0 is available on `main` as an experimental source version; it has not been published to PyPI or as a GitHub Release.
49
+
50
+ ```bash
51
+ # Run from this source checkout; Python 3.11+ and uv are required.
52
+ uv sync --frozen
53
+ uv run darwinagent --version
54
+ uv run darwinagent doctor --output runs/doctor
55
+ uv run darwinagent demo --mode replay --rounds 2 --output runs/demo-replay
56
+ ```
57
+
58
+ Expected replay: `status=complete`, first candidate accepted, second rejected for `primary_not_strictly_improved`, and `http_attempts=0`. The independent evaluator scores requested technician/date fields **0.5 → 1.0 → 1.0** in one tiny synthetic task. Replay uses scripted model responses, a seeded B0, and **P-only** proposals through the real controller and runtime. It demonstrates mechanics, not model learning or benchmark gains; it has no held-out evaluation.
59
+
60
+ The output directory contains:
61
+
62
+ ```text
63
+ runs/demo-replay/
64
+ ├── experiment.json # original experiment declaration
65
+ ├── B0/evaluation/maintenance-demo.json # baseline score
66
+ ├── R1/evaluation/maintenance-demo.json # candidate score (also R2)
67
+ ├── R1/optimization/attempt-0/proposal-call.json # proposal input (also R2)
68
+ ├── optimization/wiki.json # durable experience
69
+ ├── published/current.json # accepted asset version
70
+ └── demo-summary.json # controller summary
71
+ ```
72
+
73
+ For live execution, configure your endpoint and use a separate output directory:
74
+
75
+ ```bash
76
+ cp .env.example .env
77
+ # Set DARWINAGENT_API_KEY, DARWINAGENT_BASE_URL, DARWINAGENT_MODEL in .env.
78
+ uv run darwinagent demo --mode live --rounds 2 --output runs/demo-live \
79
+ --max-requests 40 --timeout 1800
80
+ ```
81
+
82
+ The CLI reads the current directory's `.env` without overriding shell variables. Live uses the actual model, with no recorded fallback. The cap counts dispatched HTTP attempts, including retries; the timeout covers the whole run. A live baseline may already be correct, so ties are rejected. Success means the stages execute and decisions are recorded, not that the score must improve.
83
+
84
+ Daily continuation retains saved work across changes to models and execution controls. Source changes take effect after a safe restart; use `--strict-comparison` when requiring the original frozen comparison conditions:
85
+
86
+ ```bash
87
+ uv run darwinagent demo --mode replay --rounds 2 --output runs/demo-replay --resume
88
+ ```
89
+
90
+ Or call the demo from Python:
91
+
92
+ ```python
93
+ import asyncio
94
+ from pathlib import Path
95
+ from darwinagent.demo import run_demo
96
+
97
+ asyncio.run(run_demo(Path("runs/python-replay"), mode="replay", rounds=2))
98
+ ```
99
+
100
+ [Full quickstart](docs/en/quickstart.md) · [Configuration](docs/en/configuration.md) · [Dated acceptance evidence](docs/acceptance/2026-10-07.md)
101
+
102
+ ## Bring your own task
103
+
104
+ Implement the generation boundary and an independent evaluator, then register task assets in `task.yaml` and `assets/index.yaml`. The generation input carries records, questions, and source references; evaluator references stay in the evaluator.
105
+
106
+ ```python
107
+ from darwinagent import (
108
+ CaseInput, CorpusBlock, QuestionInput, SourceRef, EvaluationResult,
109
+ )
110
+
111
+ class Records:
112
+ def generation_input(self, case_id):
113
+ return CaseInput(case_id,
114
+ (CorpusBlock(SourceRef("maintenance_record", case_id, "row-1"),
115
+ "设备 D-17 于 2026-09-01 由林维护。"),),
116
+ (QuestionInput("q1", "谁在什么时候维护了 D-17?",
117
+ {"serial": "D-17"}),))
118
+
119
+ class Score:
120
+ async def evaluate(self, result):
121
+ correct = sum(a.status == "answered" and "林" in a.answer
122
+ and "2026-09-01" in a.answer for a in result.answers)
123
+ faults = sum(a.status == "execution_error" for a in result.answers)
124
+ return EvaluationResult({"correct": correct}, len(result.answers),
125
+ len(result.answers) - faults, faults, 0)
126
+ ```
127
+
128
+ Load the actual packaged declaration and asset registry:
129
+
130
+ ```python
131
+ from darwinagent import TaskSpec
132
+ from darwinagent.demo import TASK_ROOT
133
+ from darwinagent.kernel.registration import load_assets
134
+
135
+ assets = load_assets(TASK_ROOT) # task.yaml + assets/index.yaml + S/F/C/P files
136
+ spec = TaskSpec.load(TASK_ROOT / "task.yaml")
137
+ ```
138
+
139
+ These classes plug into the common runtime; they do not replace the agent execution flow. The [complete offline example](examples/third_domain.py) runs both interfaces with registered S/F/C/P assets:
140
+
141
+ ```bash
142
+ uv run python examples/third_domain.py
143
+ ```
144
+
145
+ See the [custom task guide](docs/en/custom-tasks.md) for a complete live `Pipeline` example and asset contracts. Arbitrary external agent plugins are a future extension, not a v0.1 capability.
146
+
147
+ ## Architecture and documentation
148
+
149
+ ![Architecture](docs/assets/architecture-en.svg)
150
+
151
+ | Guide | English | 简体中文 |
152
+ | --- | --- | --- |
153
+ | Quickstart | [Read](docs/en/quickstart.md) | [阅读](docs/zh-CN/quickstart.md) |
154
+ | Architecture | [Read](docs/en/architecture.md) | [阅读](docs/zh-CN/architecture.md) |
155
+ | Configuration | [Read](docs/en/configuration.md) | [阅读](docs/zh-CN/configuration.md) |
156
+ | Custom tasks | [Read](docs/en/custom-tasks.md) | [阅读](docs/zh-CN/custom-tasks.md) |
157
+ | Experiments | [Read](docs/en/experiments.md) | [阅读](docs/zh-CN/experiments.md) |
158
+ | Migration | [Read](docs/en/migration.md) | [阅读](docs/zh-CN/migration.md) |
159
+
160
+ [Historical evidence index](docs/history/README.md) · [Editable graphics and PNG exports](docs/assets/README.md) · [Project introduction](docs/launch/introduction.en.md)
161
+
162
+ ## Status and direction
163
+
164
+ The local Python 3.11/3.12/3.13 suites and installed-wheel acceptance are documented in the [dated report](docs/acceptance/2026-10-07.md). The workflow badge shows the latest GitHub CI status. Historical dataset scores belong to their original protocols and source revisions.
165
+
166
+ Next directions are broader independent task examples, controlled held-out studies, and a carefully specified external agent integration boundary. These are research and engineering plans, not shipped capabilities or promised quality gains.
167
+
168
+ ## Research provenance and community
169
+
170
+ The S/F kernel is inspired by [*Toward Effective and Reliable LLM Agents via Dynamic Ontology*](https://arxiv.org/abs/2608.22974) (OaK). C/P assets and the experience Wiki are engineering extensions in this project. Historical reproduction records remain separately indexed; no paper performance numbers are used as current DarwinAgent results.
171
+
172
+ [Contributing](CONTRIBUTING.md) · [中文贡献指南](CONTRIBUTING.zh-CN.md) · [Changelog](CHANGELOG.md) · [Software citation](CITATION.cff)
173
+
174
+ MIT © 2026 DarwinAgent contributors. See [LICENSE](LICENSE); third-party material retains its own license and provenance.
175
+
176
+ ## Durable continuation and Wiki queries
177
+
178
+ The Workspace APIs and offline CLI retain original evidence, request receipts, human selection history and scoped previews. Wiki queries can return raw evidence or create a resumable regroup task. See [the continuation guide](docs/workspace-continuation.md). Real model smoke remains pending explicit model selection and execution; these interfaces do not establish benchmark gains.
@@ -0,0 +1,60 @@
1
+ # DarwinAgent
2
+
3
+ **Evolution for the Agent Era** · **Agent 时代的进化**
4
+
5
+ An open framework for experience-driven recursive self-improvement of AI agents.
6
+ DarwinAgent takes its name from Charles Darwin and the theory of evolution:
7
+ propose changes to task assets, select through independent evaluation, retain
8
+ effective versions, and use a persistent experience Wiki to inform subsequent
9
+ proposals.
10
+
11
+ **Version 0.1.0 is experimental.** Improvement is an outcome to measure, and a
12
+ candidate can be rejected. The framework does not train model weights or rewrite
13
+ its own optimizer.
14
+
15
+ ## Install and try the loop
16
+
17
+ Python 3.11 or newer is required. In your Python environment:
18
+
19
+ ```bash
20
+ python -m pip install darwinagent
21
+ darwinagent --version
22
+ darwinagent doctor --output runs/doctor
23
+ darwinagent demo --mode replay --rounds 2 --output runs/demo-replay
24
+ darwinagent demo --mode replay --rounds 2 --output runs/demo-replay --resume
25
+ ```
26
+
27
+ Replay requires no API key or network. It executes the actual controller, task
28
+ runtime, evaluation, admission, Wiki, and asset publication with scripted model
29
+ responses. In one tiny synthetic maintenance task, the first candidate is accepted
30
+ and the second is rejected on a tie; HTTP attempts are zero. This demonstrates
31
+ the mechanism and does not establish model learning or benchmark gains.
32
+
33
+ For real model execution, set `DARWINAGENT_API_KEY`, `DARWINAGENT_BASE_URL`, and
34
+ `DARWINAGENT_MODEL` to your OpenAI-compatible Chat Completions endpoint, then run:
35
+
36
+ ```bash
37
+ darwinagent demo --mode live --rounds 2 --output runs/demo-live \
38
+ --max-requests 40 --timeout 1800
39
+ ```
40
+
41
+ Live mode calls the configured model. The request cap counts dispatched HTTP
42
+ attempts, including retries, and the timeout covers the whole run.
43
+
44
+ ## Bring your own task
45
+
46
+ Register S/F/C/P assets: schemas, query functions, task checks, and role prompts.
47
+ Connect a dataset adapter and an independent evaluator to the shared `Pipeline`
48
+ and `ExperimentRunner`. The framework's executor, evaluator, permissions, and
49
+ adoption policy remain outside the proposal boundary.
50
+
51
+ ## Documentation
52
+
53
+ - [English README](https://github.com/gogoingai/DarwinAgent/blob/main/README.md)
54
+ - [中文说明](https://github.com/gogoingai/DarwinAgent/blob/main/README.zh-CN.md)
55
+ - [Quickstart](https://github.com/gogoingai/DarwinAgent/blob/main/docs/en/quickstart.md)
56
+ - [中文快速开始](https://github.com/gogoingai/DarwinAgent/blob/main/docs/zh-CN/quickstart.md)
57
+ - [Acceptance evidence](https://github.com/gogoingai/DarwinAgent/blob/main/docs/acceptance/2026-10-07.md)
58
+ - [Source and issues](https://github.com/gogoingai/DarwinAgent)
59
+
60
+ DarwinAgent is distributed under the MIT license.
@@ -0,0 +1,178 @@
1
+ <p align="center"><img src="docs/assets/hero-zh-CN.svg" alt="DarwinAgent — Agent 时代的进化" width="100%"></p>
2
+
3
+ # DarwinAgent
4
+
5
+ **面向 Agent 经验驱动递归自改进的开源框架**
6
+
7
+ [English](README.md) · [简体中文](README.zh-CN.md)
8
+
9
+ [![Offline framework checks](https://github.com/gogoingai/DarwinAgent/actions/workflows/ci.yml/badge.svg)](https://github.com/gogoingai/DarwinAgent/actions/workflows/ci.yml)
10
+ [![Version](https://img.shields.io/badge/version-0.1.0%20experimental-45635c)](CHANGELOG.md)
11
+ [![Python](https://img.shields.io/badge/python-%3E%3D3.11-45635c)](pyproject.toml)
12
+ [![License: MIT](https://img.shields.io/badge/license-MIT-45635c)](LICENSE)
13
+
14
+ DarwinAgent 的名字来自查尔斯·达尔文与进化论。**Agent 时代的进化**是项目的使命:让 Agent 从经验中适应任务,并保留有效的能力。在这个框架里,资产提案产生候选变化,独立评测和固定准入规则负责筛选,版本化资产与持久化经验 Wiki 负责保留结果,并影响下一轮提案。
15
+
16
+ DarwinAgent **0.1.0 是实验版本**。它围绕基于图的共享 Agent 运行时,实现了可检查的迭代流程。改进是否发生需要测量,候选也可能被拒绝。当前优化范围是任务资产,框架的执行流程和评测边界保持固定。
17
+
18
+ ## 为什么做 DarwinAgent
19
+
20
+ 一次答对说明这次运行成功,有效的能力还应该在下一次运行中保留下来。DarwinAgent 记录候选改了什么、回答依据哪些来源、评测得到了什么结果、为什么采纳或拒绝,以及哪些经验进入下一轮提案。这样,适应过程就能成为可复查的实验,而不只是一次没有记录的提示词修改。
21
+
22
+ ## 进化循环
23
+
24
+ ![进化循环](docs/assets/evolution-loop-zh-CN.svg)
25
+
26
+ 1. **运行:**通过共同的 `Pipeline` 执行基线,按实际输出评分。
27
+ 2. **变化:**结合训练证据与 Wiki 反馈,提出有边界的 S/F/C/P 资产修改。
28
+ 3. **筛选:**验证契约与能力边界,运行候选,按冻结的采纳策略决定是否保留。
29
+ 4. **积累:**原子发布被采纳的版本;保留拒绝记录,让经验进入下一轮提案。
30
+
31
+ 内核、评测器、权限和采纳规则位于提案边界之外。DarwinAgent 不训练模型权重,也不重写自己的优化器。项目名称表达的是进化论带来的启发,并不表示实现了遗传算法。
32
+
33
+ ## 已实现的能力
34
+
35
+ | 能力 | v0.1.0 的实际行为 |
36
+ | --- | --- |
37
+ | 共享任务运行时 | `ExtractionAgent` → 带来源的图 → `AnswerAgent`,由一个 `Pipeline` 执行 |
38
+ | 有边界的任务资产 | **S** 本体/模式、**F** 查询工具函数、**C** 任务检查、**P** 角色提示词 |
39
+ | 经验积累 | Wiki 保存观察和采纳/拒绝决策,供后续提案使用 |
40
+ | 可复查运行 | 内容编号、独立来源、局部续跑与显式严格比较 |
41
+ | 独立评测 | 分开的 `DatasetAdapter` 和 `Evaluator`,任务自行定义指标名称 |
42
+ | 实验控制 | `ExperimentRunner`;另有 `CampaignController` 管理训练/验证/测试协议 |
43
+ | 模型连接 | 默认共用一个显式配置的 OpenAI 兼容 Chat Completions 端点,可按档位覆盖 |
44
+ | 使用入口 | 安装后的 CLI、Python SDK、离线回放、真实模型模式、设备维护示例 |
45
+
46
+ ## 跑通循环
47
+
48
+ **从当前源码检出目录安装**。DarwinAgent 0.1.0 已作为实验性源码版本合入 `main`,尚未发布到 PyPI 或创建 GitHub Release。
49
+
50
+ ```bash
51
+ # 在当前源码检出目录运行,需要 Python 3.11+ 和 uv。
52
+ uv sync --frozen
53
+ uv run darwinagent --version
54
+ uv run darwinagent doctor --output runs/doctor
55
+ uv run darwinagent demo --mode replay --rounds 2 --output runs/demo-replay
56
+ ```
57
+
58
+ 回放的预期结果:`status=complete`,第一轮采纳,第二轮因 `primary_not_strictly_improved` 拒绝,`http_attempts=0`。独立评测器在一个微型合成任务中,按维护人员/日期两个字段得到 **0.5 → 1.0 → 1.0**。回放使用脚本模型响应、预置 B0 和 **仅 P 资产**的提案,实际执行控制器与运行时。它验证机制,不证明模型学习或基准提升,也没有留出集评测。
59
+
60
+ 输出目录包括:
61
+
62
+ ```text
63
+ runs/demo-replay/
64
+ ├── experiment.json # original experiment declaration
65
+ ├── B0/evaluation/maintenance-demo.json # baseline score
66
+ ├── R1/evaluation/maintenance-demo.json # candidate score (also R2)
67
+ ├── R1/optimization/attempt-0/proposal-call.json # proposal input (also R2)
68
+ ├── optimization/wiki.json # durable experience
69
+ ├── published/current.json # accepted asset version
70
+ └── demo-summary.json # controller summary
71
+ ```
72
+
73
+ 真实模型运行需要配置端点,并使用另一个输出目录:
74
+
75
+ ```bash
76
+ cp .env.example .env
77
+ # 在 .env 中填写 DARWINAGENT_API_KEY、DARWINAGENT_BASE_URL、DARWINAGENT_MODEL。
78
+ uv run darwinagent demo --mode live --rounds 2 --output runs/demo-live \
79
+ --max-requests 40 --timeout 1800
80
+ ```
81
+
82
+ CLI 读取当前目录的 `.env`,不覆盖已有环境变量。真实模式调用实际模型,不回退到录制响应。请求上限按实际发出的 HTTP 尝试计数,包括重试;超时限制覆盖整个运行。真实模型可能在 B0 就全部答对,同分候选应当被拒绝。验收成功表示各阶段执行且决策留痕,不要求分数必然上升。
83
+
84
+ 日常续跑保留已有成果,模型和执行控制变化不会清空进度。源码修改在安全重启后生效;需要原冻结条件比较时显式添加 `--strict-comparison`:
85
+
86
+ ```bash
87
+ uv run darwinagent demo --mode replay --rounds 2 --output runs/demo-replay --resume
88
+ ```
89
+
90
+ 也可以从 Python 调用:
91
+
92
+ ```python
93
+ import asyncio
94
+ from pathlib import Path
95
+ from darwinagent.demo import run_demo
96
+
97
+ asyncio.run(run_demo(Path("runs/python-replay"), mode="replay", rounds=2))
98
+ ```
99
+
100
+ [完整快速开始](docs/zh-CN/quickstart.md) · [配置](docs/zh-CN/configuration.md) · [带日期的验收证据](docs/acceptance/2026-10-07.md)
101
+
102
+ ## 接入自己的任务
103
+
104
+ 实现生成输入接口和独立评测器,再通过 `task.yaml` 与 `assets/index.yaml` 登记任务资产。生成输入只携带记录、问题和来源;评测参考保留在评测器内部。
105
+
106
+ ```python
107
+ from darwinagent import (
108
+ CaseInput, CorpusBlock, QuestionInput, SourceRef, EvaluationResult,
109
+ )
110
+
111
+ class Records:
112
+ def generation_input(self, case_id):
113
+ return CaseInput(case_id,
114
+ (CorpusBlock(SourceRef("maintenance_record", case_id, "row-1"),
115
+ "设备 D-17 于 2026-09-01 由林维护。"),),
116
+ (QuestionInput("q1", "谁在什么时候维护了 D-17?",
117
+ {"serial": "D-17"}),))
118
+
119
+ class Score:
120
+ async def evaluate(self, result):
121
+ correct = sum(a.status == "answered" and "林" in a.answer
122
+ and "2026-09-01" in a.answer for a in result.answers)
123
+ faults = sum(a.status == "execution_error" for a in result.answers)
124
+ return EvaluationResult({"correct": correct}, len(result.answers),
125
+ len(result.answers) - faults, faults, 0)
126
+ ```
127
+
128
+ 加载实际打包的任务声明与资产登记:
129
+
130
+ ```python
131
+ from darwinagent import TaskSpec
132
+ from darwinagent.demo import TASK_ROOT
133
+ from darwinagent.kernel.registration import load_assets
134
+
135
+ assets = load_assets(TASK_ROOT) # task.yaml + assets/index.yaml + S/F/C/P files
136
+ spec = TaskSpec.load(TASK_ROOT / "task.yaml")
137
+ ```
138
+
139
+ 这两个类接入共同运行时,不替换 Agent 执行流程。[完整离线示例](examples/third_domain.py) 将这两个接口和已登记的 S/F/C/P 资产一起运行:
140
+
141
+ ```bash
142
+ uv run python examples/third_domain.py
143
+ ```
144
+
145
+ [自定义任务指南](docs/zh-CN/custom-tasks.md)提供完整的真实模型 `Pipeline` 示例及资产契约。任意外部 Agent 的插件接入是后续方向,不是 v0.1 已有能力。
146
+
147
+ ## 架构与文档
148
+
149
+ ![架构](docs/assets/architecture-zh-CN.svg)
150
+
151
+ | 指南 | English | 简体中文 |
152
+ | --- | --- | --- |
153
+ | 快速开始 | [Read](docs/en/quickstart.md) | [阅读](docs/zh-CN/quickstart.md) |
154
+ | 架构 | [Read](docs/en/architecture.md) | [阅读](docs/zh-CN/architecture.md) |
155
+ | 配置 | [Read](docs/en/configuration.md) | [阅读](docs/zh-CN/configuration.md) |
156
+ | 自定义任务 | [Read](docs/en/custom-tasks.md) | [阅读](docs/zh-CN/custom-tasks.md) |
157
+ | 实验 | [Read](docs/en/experiments.md) | [阅读](docs/zh-CN/experiments.md) |
158
+ | 迁移 | [Read](docs/en/migration.md) | [阅读](docs/zh-CN/migration.md) |
159
+
160
+ [历史证据索引](docs/history/README.md) · [可编辑图源与 PNG](docs/assets/README.md) · [项目介绍稿](docs/launch/introduction.zh-CN.md)
161
+
162
+ ## 状态与后续方向
163
+
164
+ 本地 Python 3.11/3.12/3.13 测试和安装后 wheel 验收记录见[带日期的报告](docs/acceptance/2026-10-07.md)。工作流徽章显示 GitHub CI 的最新状态。历史数据集成绩属于原来的协议和源码版本。
165
+
166
+ 后续计划包括更多独立任务示例、受控留出集实验,以及明确的外部 Agent 接入边界。这些是研究与工程方向,不是已交付能力,也不承诺质量提升。
167
+
168
+ ## 研究来源与社区
169
+
170
+ S/F 内核受 [*Toward Effective and Reliable LLM Agents via Dynamic Ontology*](https://arxiv.org/abs/2608.22974)(OaK)启发。C/P 资产与经验 Wiki 是本项目的工程扩展。历史复现记录另行索引;论文中的性能数字不作为当前 DarwinAgent 的结果。
171
+
172
+ [贡献指南](CONTRIBUTING.zh-CN.md) · [English contribution guide](CONTRIBUTING.md) · [变更记录](CHANGELOG.md) · [软件引用](CITATION.cff)
173
+
174
+ MIT © 2026 DarwinAgent contributors。见 [LICENSE](LICENSE),第三方材料保留各自的许可证与来源。
175
+
176
+ ## 人工干预、续跑与 Wiki 补查
177
+
178
+ Workspace 将原始内容、执行来源和分支选择分别保存,提供人工登记、执行预览、请求恢复及安全导入导出。Wiki 可以返回原件,并建立可续跑的重新归纳任务。操作说明见[续跑与补查指南](docs/workspace-continuation.zh-CN.md)。真实模型仍需操作者指定并实际执行五题冒烟;离线检查不代表真实冒烟通过或指标提升。
@@ -0,0 +1,98 @@
1
+ [project]
2
+ name = "darwinagent"
3
+ version = "0.1.0"
4
+ description = "Evolution for the Agent Era: experience-driven recursive self-improvement for agents"
5
+ readme = "README.pypi.md"
6
+ license = "MIT"
7
+ requires-python = ">=3.11"
8
+ keywords = ["agents", "recursive-self-improvement", "agent-memory", "darwinagent"]
9
+ classifiers = [
10
+ "Development Status :: 3 - Alpha",
11
+ "Programming Language :: Python :: 3",
12
+ "Programming Language :: Python :: 3.11",
13
+ "Programming Language :: Python :: 3.12",
14
+ "Programming Language :: Python :: 3.13",
15
+ ]
16
+ dependencies = [
17
+ "openai>=1.40",
18
+ "httpx>=0.27",
19
+ "networkx>=3.6.1",
20
+ "pyyaml>=6.0",
21
+ "python-dotenv>=1.0",
22
+ ]
23
+
24
+ [project.urls]
25
+ Repository = "https://github.com/gogoingai/DarwinAgent"
26
+ Documentation = "https://github.com/gogoingai/DarwinAgent#readme"
27
+ Issues = "https://github.com/gogoingai/DarwinAgent/issues"
28
+
29
+ [project.scripts]
30
+ darwinagent = "darwinagent.cli:main"
31
+
32
+ [project.optional-dependencies]
33
+ vector = ["numpy>=1.26", "requests>=2.31"]
34
+ formal = ["owlready2>=0.46"]
35
+ benchmarks = [
36
+ "pandas>=2.1",
37
+ "numpy>=1.26",
38
+ "datasets>=2.20",
39
+ "tqdm>=4.66",
40
+ "rich>=13.0",
41
+ "gradio>=6.0.0",
42
+ "requests>=2.31",
43
+ ]
44
+
45
+ [dependency-groups]
46
+ test = ["numpy>=1.26", "pandas>=2.1", "requests>=2.31", "ruff==0.16.10"]
47
+
48
+ [tool.ruff]
49
+ target-version = "py311"
50
+ line-length = 100
51
+ indent-width = 4
52
+ preview = false
53
+ force-exclude = true
54
+ include = ["src/**/*.py", "tests/**/*.py", "datasets/**/*.py", "examples/**/*.py", "scripts/check_installed.py"]
55
+ extend-exclude = [
56
+ "src/darwinagent/demo/device_maintenance/assets",
57
+ "tests/fixtures",
58
+ "tasks",
59
+ "datasets/*/scripts",
60
+ "datasets/locomo/data",
61
+ "datasets/travelplanner/data",
62
+ "datasets/*/runs",
63
+ "datasets/*/snapshots",
64
+ "docs/diagnostics",
65
+ "docs/history",
66
+ "mem0",
67
+ "third_party",
68
+ "oak",
69
+ ]
70
+
71
+ [tool.ruff.lint]
72
+ select = ["E4", "E7", "E9", "F", "I", "UP", "B006", "B023", "B904"]
73
+
74
+ [tool.ruff.format]
75
+ quote-style = "double"
76
+ indent-style = "space"
77
+ line-ending = "lf"
78
+ docstring-code-format = false
79
+
80
+ [tool.uv]
81
+ package = true
82
+
83
+ [build-system]
84
+ requires = ["hatchling"]
85
+ build-backend = "hatchling.build"
86
+
87
+ [tool.hatch.build.targets.wheel]
88
+ packages = ["src/darwinagent"]
89
+
90
+ [tool.hatch.build.targets.sdist]
91
+ include = [
92
+ "/src/darwinagent",
93
+ "/pyproject.toml",
94
+ "/README.pypi.md",
95
+ "/README.md",
96
+ "/README.zh-CN.md",
97
+ "/LICENSE",
98
+ ]