tunarag-python 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tunarag_python-0.2.1/.gitattributes +7 -0
- tunarag_python-0.2.1/.github/workflows/ci.yml +56 -0
- tunarag_python-0.2.1/.github/workflows/publish.yml +74 -0
- tunarag_python-0.2.1/.gitignore +26 -0
- tunarag_python-0.2.1/CHANGELOG.md +48 -0
- tunarag_python-0.2.1/CONTRIBUTING.md +23 -0
- tunarag_python-0.2.1/DECISIONS.md +88 -0
- tunarag_python-0.2.1/FLOW.md +66 -0
- tunarag_python-0.2.1/PKG-INFO +1164 -0
- tunarag_python-0.2.1/PROJECT_PLAN.md +111 -0
- tunarag_python-0.2.1/README.md +1122 -0
- tunarag_python-0.2.1/RELEASE_CHECKLIST.md +26 -0
- tunarag_python-0.2.1/SECURITY.md +24 -0
- tunarag_python-0.2.1/docs/01_API_Specification.docx +0 -0
- tunarag_python-0.2.1/docs/02_Coding_Conventions.docx +0 -0
- tunarag_python-0.2.1/docs/03_Component_Map.docx +0 -0
- tunarag_python-0.2.1/docs/04_Data_Extraction_Specification.docx +0 -0
- tunarag_python-0.2.1/docs/05_Database_Schema.docx +0 -0
- tunarag_python-0.2.1/docs/06_Environment_Config_Reference.docx +0 -0
- tunarag_python-0.2.1/docs/07_File_Folder_Structure.docx +0 -0
- tunarag_python-0.2.1/docs/08_Product_Requirement_Document.docx +0 -0
- tunarag_python-0.2.1/docs/09_System_Architecture_Document.docx +0 -0
- tunarag_python-0.2.1/docs/10_Tech_Stack.docx +0 -0
- tunarag_python-0.2.1/docs/11_Technical_Design_Document.docx +0 -0
- tunarag_python-0.2.1/docs/12_Documentation_Index.docx +0 -0
- tunarag_python-0.2.1/docs/api-specification.md +307 -0
- tunarag_python-0.2.1/docs/coding-conventions.md +42 -0
- tunarag_python-0.2.1/docs/component-map.md +27 -0
- tunarag_python-0.2.1/docs/data-extraction-specification.md +53 -0
- tunarag_python-0.2.1/docs/database-schema.md +57 -0
- tunarag_python-0.2.1/docs/decision.md +255 -0
- tunarag_python-0.2.1/docs/environment-config-reference.md +41 -0
- tunarag_python-0.2.1/docs/file-folder-structure.md +50 -0
- tunarag_python-0.2.1/docs/flow.md +148 -0
- tunarag_python-0.2.1/docs/product-requirements-document.md +55 -0
- tunarag_python-0.2.1/docs/system-architecture-document.md +59 -0
- tunarag_python-0.2.1/docs/tech-stack.md +25 -0
- tunarag_python-0.2.1/docs/technical-design-document.md +197 -0
- tunarag_python-0.2.1/docs/ui-ux-wireframes.md +37 -0
- tunarag_python-0.2.1/examples/README.md +41 -0
- tunarag_python-0.2.1/examples/__init__.py +1 -0
- tunarag_python-0.2.1/examples/quickstart.py +83 -0
- tunarag_python-0.2.1/examples/reference_rag.py +105 -0
- tunarag_python-0.2.1/pyproject.toml +87 -0
- tunarag_python-0.2.1/src/tunarag/__init__.py +166 -0
- tunarag_python-0.2.1/src/tunarag/cache.py +326 -0
- tunarag_python-0.2.1/src/tunarag/config.py +257 -0
- tunarag_python-0.2.1/src/tunarag/contracts.py +88 -0
- tunarag_python-0.2.1/src/tunarag/dataset.py +481 -0
- tunarag_python-0.2.1/src/tunarag/domain.py +83 -0
- tunarag_python-0.2.1/src/tunarag/engine.py +979 -0
- tunarag_python-0.2.1/src/tunarag/errors.py +179 -0
- tunarag_python-0.2.1/src/tunarag/evaluators.py +186 -0
- tunarag_python-0.2.1/src/tunarag/integrations/__init__.py +25 -0
- tunarag_python-0.2.1/src/tunarag/integrations/mlflow.py +155 -0
- tunarag_python-0.2.1/src/tunarag/integrations/runnables.py +270 -0
- tunarag_python-0.2.1/src/tunarag/objective.py +75 -0
- tunarag_python-0.2.1/src/tunarag/py.typed +1 -0
- tunarag_python-0.2.1/src/tunarag/result.py +309 -0
- tunarag_python-0.2.1/src/tunarag/retry.py +70 -0
- tunarag_python-0.2.1/src/tunarag/search.py +286 -0
- tunarag_python-0.2.1/src/tunarag/serialization.py +78 -0
- tunarag_python-0.2.1/src/tunarag/stopping.py +193 -0
- tunarag_python-0.2.1/src/tunarag/store.py +941 -0
- tunarag_python-0.2.1/src/tunarag/synthetic.py +257 -0
- tunarag_python-0.2.1/tests/test_cache.py +133 -0
- tunarag_python-0.2.1/tests/test_callbacks.py +173 -0
- tunarag_python-0.2.1/tests/test_concurrency.py +116 -0
- tunarag_python-0.2.1/tests/test_config.py +87 -0
- tunarag_python-0.2.1/tests/test_dataset_ingestion.py +141 -0
- tunarag_python-0.2.1/tests/test_domain.py +35 -0
- tunarag_python-0.2.1/tests/test_engine.py +300 -0
- tunarag_python-0.2.1/tests/test_errors.py +59 -0
- tunarag_python-0.2.1/tests/test_evaluators.py +164 -0
- tunarag_python-0.2.1/tests/test_framework_compatibility.py +61 -0
- tunarag_python-0.2.1/tests/test_llm_search.py +113 -0
- tunarag_python-0.2.1/tests/test_mlflow_integration.py +96 -0
- tunarag_python-0.2.1/tests/test_objective_and_stopping.py +98 -0
- tunarag_python-0.2.1/tests/test_package.py +9 -0
- tunarag_python-0.2.1/tests/test_properties.py +52 -0
- tunarag_python-0.2.1/tests/test_reference_example.py +16 -0
- tunarag_python-0.2.1/tests/test_resilience.py +362 -0
- tunarag_python-0.2.1/tests/test_result_reporting.py +106 -0
- tunarag_python-0.2.1/tests/test_resume.py +301 -0
- tunarag_python-0.2.1/tests/test_retry.py +47 -0
- tunarag_python-0.2.1/tests/test_runnable_integrations.py +150 -0
- tunarag_python-0.2.1/tests/test_search.py +58 -0
- tunarag_python-0.2.1/tests/test_store.py +276 -0
- tunarag_python-0.2.1/tests/test_synthetic.py +91 -0
- tunarag_python-0.2.1/tests/test_usage_documentation.py +531 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
pull_request:
|
|
5
|
+
push:
|
|
6
|
+
branches: [main]
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
quality:
|
|
13
|
+
strategy:
|
|
14
|
+
fail-fast: false
|
|
15
|
+
matrix:
|
|
16
|
+
os: [ubuntu-latest, windows-latest]
|
|
17
|
+
python-version: ["3.10", "3.13"]
|
|
18
|
+
|
|
19
|
+
runs-on: ${{ matrix.os }}
|
|
20
|
+
|
|
21
|
+
steps:
|
|
22
|
+
- uses: actions/checkout@v4
|
|
23
|
+
- uses: actions/setup-python@v5
|
|
24
|
+
with:
|
|
25
|
+
python-version: ${{ matrix.python-version }}
|
|
26
|
+
cache: pip
|
|
27
|
+
- name: Install package and development tools
|
|
28
|
+
run: python -m pip install --upgrade pip && python -m pip install -e ".[dev]"
|
|
29
|
+
- name: Check formatting
|
|
30
|
+
run: python -m ruff format --check .
|
|
31
|
+
- name: Lint
|
|
32
|
+
run: python -m ruff check .
|
|
33
|
+
- name: Type check
|
|
34
|
+
run: python -m mypy
|
|
35
|
+
- name: Test
|
|
36
|
+
run: python -m pytest
|
|
37
|
+
- name: Build distributions
|
|
38
|
+
run: python -m build
|
|
39
|
+
- name: Install built wheel
|
|
40
|
+
run: python -m pip install --force-reinstall --no-deps dist/tunarag_python-0.2.1-py3-none-any.whl
|
|
41
|
+
- name: Import built wheel
|
|
42
|
+
run: python -c "import tunarag; assert tunarag.__version__ == '0.2.1'"
|
|
43
|
+
|
|
44
|
+
framework-compatibility:
|
|
45
|
+
runs-on: ubuntu-latest
|
|
46
|
+
|
|
47
|
+
steps:
|
|
48
|
+
- uses: actions/checkout@v4
|
|
49
|
+
- uses: actions/setup-python@v5
|
|
50
|
+
with:
|
|
51
|
+
python-version: "3.13"
|
|
52
|
+
cache: pip
|
|
53
|
+
- name: Install framework integrations
|
|
54
|
+
run: python -m pip install --upgrade pip && python -m pip install -e ".[dev,frameworks]"
|
|
55
|
+
- name: Test real framework compatibility
|
|
56
|
+
run: python -m pytest tests/test_framework_compatibility.py
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
name: Publish Python package
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
workflow_dispatch:
|
|
5
|
+
inputs:
|
|
6
|
+
target:
|
|
7
|
+
description: Package index to publish to
|
|
8
|
+
required: true
|
|
9
|
+
type: choice
|
|
10
|
+
options:
|
|
11
|
+
- testpypi
|
|
12
|
+
- pypi
|
|
13
|
+
|
|
14
|
+
permissions:
|
|
15
|
+
contents: read
|
|
16
|
+
|
|
17
|
+
jobs:
|
|
18
|
+
build:
|
|
19
|
+
runs-on: ubuntu-latest
|
|
20
|
+
|
|
21
|
+
steps:
|
|
22
|
+
- uses: actions/checkout@v4
|
|
23
|
+
- uses: actions/setup-python@v5
|
|
24
|
+
with:
|
|
25
|
+
python-version: "3.13"
|
|
26
|
+
cache: pip
|
|
27
|
+
- name: Install build tools
|
|
28
|
+
run: python -m pip install --upgrade pip build twine
|
|
29
|
+
- name: Build distributions
|
|
30
|
+
run: python -m build
|
|
31
|
+
- name: Validate distributions
|
|
32
|
+
run: python -m twine check dist/*
|
|
33
|
+
- uses: actions/upload-artifact@v4
|
|
34
|
+
with:
|
|
35
|
+
name: python-package-distributions
|
|
36
|
+
path: dist/
|
|
37
|
+
if-no-files-found: error
|
|
38
|
+
retention-days: 7
|
|
39
|
+
|
|
40
|
+
publish-testpypi:
|
|
41
|
+
if: inputs.target == 'testpypi'
|
|
42
|
+
needs: build
|
|
43
|
+
runs-on: ubuntu-latest
|
|
44
|
+
environment:
|
|
45
|
+
name: testpypi
|
|
46
|
+
url: https://test.pypi.org/p/tunarag-python
|
|
47
|
+
permissions:
|
|
48
|
+
id-token: write
|
|
49
|
+
|
|
50
|
+
steps:
|
|
51
|
+
- uses: actions/download-artifact@v4
|
|
52
|
+
with:
|
|
53
|
+
name: python-package-distributions
|
|
54
|
+
path: dist/
|
|
55
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
56
|
+
with:
|
|
57
|
+
repository-url: https://test.pypi.org/legacy/
|
|
58
|
+
|
|
59
|
+
publish-pypi:
|
|
60
|
+
if: inputs.target == 'pypi'
|
|
61
|
+
needs: build
|
|
62
|
+
runs-on: ubuntu-latest
|
|
63
|
+
environment:
|
|
64
|
+
name: pypi
|
|
65
|
+
url: https://pypi.org/p/tunarag-python
|
|
66
|
+
permissions:
|
|
67
|
+
id-token: write
|
|
68
|
+
|
|
69
|
+
steps:
|
|
70
|
+
- uses: actions/download-artifact@v4
|
|
71
|
+
with:
|
|
72
|
+
name: python-package-distributions
|
|
73
|
+
path: dist/
|
|
74
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Python bytecode and test/tool caches
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
.pytest_cache/
|
|
5
|
+
.mypy_cache/
|
|
6
|
+
.ruff_cache/
|
|
7
|
+
.hypothesis/
|
|
8
|
+
.tmp/
|
|
9
|
+
.coverage
|
|
10
|
+
htmlcov/
|
|
11
|
+
|
|
12
|
+
# Local environments
|
|
13
|
+
.venv/
|
|
14
|
+
venv/
|
|
15
|
+
env/
|
|
16
|
+
|
|
17
|
+
# Build and distribution output
|
|
18
|
+
build/
|
|
19
|
+
dist/
|
|
20
|
+
*.egg-info/
|
|
21
|
+
|
|
22
|
+
# Local configuration and editor files
|
|
23
|
+
.env
|
|
24
|
+
.env.*
|
|
25
|
+
.vscode/
|
|
26
|
+
.idea/
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to TunaRAG follow semantic versioning.
|
|
4
|
+
|
|
5
|
+
## 0.2.1 - 2026-10-04
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Comprehensive product-first README with executable examples for every public workflow.
|
|
10
|
+
- CI coverage that extracts and runs all 36 Python examples from the README.
|
|
11
|
+
- Manual trusted-publishing workflow for separate TestPyPI and PyPI environments.
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- Corrected dataset, LangGraph state, cache, resumability, optimizer-construction, and report-path
|
|
16
|
+
examples found through runtime documentation testing.
|
|
17
|
+
- Replaced corrupted version-range punctuation in package metadata.
|
|
18
|
+
|
|
19
|
+
## 0.2.0 - 2026-10-04
|
|
20
|
+
|
|
21
|
+
### Added
|
|
22
|
+
|
|
23
|
+
- Official `LangChainAdapter` for any asynchronous LangChain runnable.
|
|
24
|
+
- Official `LangGraphAdapter` for compiled LangGraph graphs.
|
|
25
|
+
- Default mapping from evaluation queries to `question` input and candidate parameters to the
|
|
26
|
+
framework `configurable` namespace.
|
|
27
|
+
- Custom input, configuration, output, usage, and per-candidate runnable-factory hooks.
|
|
28
|
+
- Automatic runnable latency and standard message token-usage capture.
|
|
29
|
+
- Optional `langchain`, `langgraph`, and combined `frameworks` installation extras.
|
|
30
|
+
- Real-framework compatibility tests against supported LangChain 1.x and LangGraph 1.x releases.
|
|
31
|
+
|
|
32
|
+
## 0.1.0 - 2026-10-04
|
|
33
|
+
|
|
34
|
+
### Added
|
|
35
|
+
|
|
36
|
+
- Typed `RAGAdapter`, dataset, evaluator, search, objective, usage, callback, and result contracts.
|
|
37
|
+
- Random and provider-neutral LLM search over arbitrary typed search spaces.
|
|
38
|
+
- Async optimization with a safe synchronous wrapper, bounded concurrency, retry policies,
|
|
39
|
+
timeouts, caching, failure isolation, stopping rules, and resumable studies.
|
|
40
|
+
- SQLite-first experiment storage with lifecycle events, recovery leases, and durable usage.
|
|
41
|
+
- Canonical JSON/JSONL/CSV dataset ingestion, deterministic splitting, and synthetic QA support.
|
|
42
|
+
- Optional lazy RAGAS evaluation and best-effort MLflow event mirroring.
|
|
43
|
+
- Redacted JSON/CSV reports and a deterministic provider-free reference RAG quickstart.
|
|
44
|
+
|
|
45
|
+
### Deferred
|
|
46
|
+
|
|
47
|
+
- Formal Pareto-front reporting, Bayesian and grid search, distributed execution, a CLI, and
|
|
48
|
+
framework-specific LangChain/LlamaIndex adapters remain post-0.1 work.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
## Setup
|
|
4
|
+
|
|
5
|
+
```powershell
|
|
6
|
+
python -m venv .venv
|
|
7
|
+
.venv\Scripts\python.exe -m pip install -e ".[dev]"
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
On macOS or Linux, activate `.venv` and use `python` in place of the Windows executable.
|
|
11
|
+
|
|
12
|
+
## Change workflow
|
|
13
|
+
|
|
14
|
+
1. Create a focused feature, fix, documentation, or chore branch from `main`.
|
|
15
|
+
2. Update implementation, tests, and the living documents under `docs/` together.
|
|
16
|
+
3. Run `python -m ruff format .`, `python -m ruff check .`, `python -m mypy`,
|
|
17
|
+
`python -m pytest`, and `python -m build`.
|
|
18
|
+
4. Open a pull request with a concise summary and verification evidence.
|
|
19
|
+
5. Merge only after the Linux and Windows CI matrix passes.
|
|
20
|
+
|
|
21
|
+
Keep core dependencies provider-neutral. User code, provider calls, callbacks, and sleeps must
|
|
22
|
+
not run inside SQLite transactions. New public behavior needs tests and API documentation;
|
|
23
|
+
optional integrations must remain importable only when requested.
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
# Decision Log
|
|
2
|
+
|
|
3
|
+
The canonical living decision log is [`docs/decision.md`](docs/decision.md). This root file is
|
|
4
|
+
retained for compatibility with existing links. New decisions belong in the canonical file.
|
|
5
|
+
|
|
6
|
+
## Planning baseline
|
|
7
|
+
|
|
8
|
+
### Use a Python-first library with async-first internals
|
|
9
|
+
|
|
10
|
+
Decision: implement the core orchestration asynchronously and provide a synchronous convenience wrapper.
|
|
11
|
+
|
|
12
|
+
Reasoning: RAG calls, evaluators, and provider calls are commonly I/O-bound; async orchestration supports timeouts, cancellation, and future concurrency while the wrapper keeps simple scripts approachable.
|
|
13
|
+
|
|
14
|
+
### Keep the RAG application outside the package
|
|
15
|
+
|
|
16
|
+
Decision: require a user-supplied `RAGAdapter` contract rather than prescribing a framework or owning the application pipeline.
|
|
17
|
+
|
|
18
|
+
Reasoning: the package optimizes existing systems and must work across frameworks. This keeps the core portable and prevents the reference Auto-RAG-Research implementation from becoming a hidden dependency.
|
|
19
|
+
|
|
20
|
+
### Make integrations optional
|
|
21
|
+
|
|
22
|
+
Decision: keep RAGAS, MLflow, and provider-specific integrations behind optional dependency groups and import boundaries.
|
|
23
|
+
|
|
24
|
+
Reasoning: the core evaluation, storage, and orchestration contracts should remain installable and testable without external services or heavyweight packages.
|
|
25
|
+
|
|
26
|
+
### Use SQLite as the default source of durable state
|
|
27
|
+
|
|
28
|
+
Decision: store studies, trials, attempts, metrics, usage, events, and recovery state in SQLite; treat MLflow as an optional mirror.
|
|
29
|
+
|
|
30
|
+
Reasoning: SQLite is local, transactional, auditable, and sufficient for the V0.1 single-process scope. An external tracking service should not be required for correctness or resume.
|
|
31
|
+
|
|
32
|
+
### Record exactness explicitly
|
|
33
|
+
|
|
34
|
+
Decision: usage and aggregate reports distinguish exact, estimated, mixed, and unavailable values.
|
|
35
|
+
|
|
36
|
+
Reasoning: unknown cost or latency must not silently become zero, and derived pricing must remain auditable. Budget policies can then state how estimates and unknowns affect scheduling.
|
|
37
|
+
|
|
38
|
+
### Preserve secret safety as a cross-cutting invariant
|
|
39
|
+
|
|
40
|
+
Decision: use redacted secret types, centralized canonical serialization, and explicit exclusions for secrets from hashes, logs, events, exports, cache metadata, and MLflow parameters.
|
|
41
|
+
|
|
42
|
+
Reasoning: configuration and provider credentials cross many boundaries; local fixes at individual call sites are too easy to miss.
|
|
43
|
+
|
|
44
|
+
### Begin with contracts and deterministic primitives
|
|
45
|
+
|
|
46
|
+
Decision: the first feature branch delivers scaffolding, public models/contracts, canonical serialization, and tests before orchestration or integrations.
|
|
47
|
+
|
|
48
|
+
Reasoning: the API specification and later persistence/search work depend on stable types and reproducible identity. Establishing those boundaries first reduces rework and makes each later branch reviewable.
|
|
49
|
+
|
|
50
|
+
### Use `tunarag` as the import package name
|
|
51
|
+
|
|
52
|
+
Decision: publish the distribution as `tunarag-python` and expose it through the `tunarag` Python import package.
|
|
53
|
+
|
|
54
|
+
Reasoning: the repository/distribution name follows the requested project name while the shorter import name is idiomatic and leaves room for related tooling without a hyphenated module.
|
|
55
|
+
|
|
56
|
+
### Use frozen dataclasses for core values
|
|
57
|
+
|
|
58
|
+
Decision: represent the first domain values with frozen, slotted dataclasses and keep extension points as typing protocols.
|
|
59
|
+
|
|
60
|
+
Reasoning: immutable values make trial identity and persistence safer, while protocols let callers implement adapters without adopting a package-specific model framework.
|
|
61
|
+
|
|
62
|
+
### Exclude secret fields from canonical identity
|
|
63
|
+
|
|
64
|
+
Decision: canonical mappings omit `Secret` values entirely and represent standalone secrets only with a redacted marker.
|
|
65
|
+
|
|
66
|
+
Reasoning: credentials must never influence cache identity or appear in serialized bytes; nonsecret configuration remains stable and hashable.
|
|
67
|
+
|
|
68
|
+
### Use package-owned random streams for V0.1 random search
|
|
69
|
+
|
|
70
|
+
Decision: implement `RandomSearch` with an isolated `random.Random` instance owned by the strategy.
|
|
71
|
+
|
|
72
|
+
Reasoning: a private seeded stream makes candidate sequences reproducible without mutating application-global random state or coupling search-space sampling to external frameworks.
|
|
73
|
+
|
|
74
|
+
### Preserve value confidence during objective aggregation
|
|
75
|
+
|
|
76
|
+
Decision: objective results carry `exact`, `estimated`, or `unavailable` status instead of converting missing or estimated values to zero.
|
|
77
|
+
|
|
78
|
+
Reasoning: budget and ranking decisions must remain auditable; silently treating unknown values as exact would make optimization results misleading.
|
|
79
|
+
|
|
80
|
+
### Use explicit SQLite transactions for lifecycle writes
|
|
81
|
+
|
|
82
|
+
Decision: keep the first persistence layer on the standard-library `sqlite3` driver with explicit schema initialization and transaction scopes.
|
|
83
|
+
|
|
84
|
+
Reasoning: V0.1 is single-process and local-first; direct transactions preserve lifecycle atomicity without adding an ORM before a second store backend exists.
|
|
85
|
+
|
|
86
|
+
## Update rule
|
|
87
|
+
|
|
88
|
+
Every important implementation choice records what was chosen, why it fits the design documents, and what alternative was rejected. Entries are added as decisions are made; this file is not a generic project diary.
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Code Flow
|
|
2
|
+
|
|
3
|
+
The canonical living code-flow document is [`docs/flow.md`](docs/flow.md). This root file is
|
|
4
|
+
retained for compatibility with existing links. Update both the implementation and canonical
|
|
5
|
+
flow document in the same feature branch.
|
|
6
|
+
|
|
7
|
+
## Status
|
|
8
|
+
|
|
9
|
+
V0.1 is implementation-complete. The canonical flow covers typed contracts, datasets and
|
|
10
|
+
synthetic generation, RandomSearch and LLMSearch, objectives and stopping, SQLite storage and
|
|
11
|
+
caching, retries and timeouts, bounded concurrency, callbacks, resumability, optional RAGAS and
|
|
12
|
+
MLflow integrations, deterministic reporting, and the reference example.
|
|
13
|
+
|
|
14
|
+
## Intended entry points
|
|
15
|
+
|
|
16
|
+
- Public async study runner: accepts a study configuration, adapter, dataset, strategy, evaluator, objective, store, cache, and lifecycle policies.
|
|
17
|
+
- Public synchronous convenience wrapper: runs the async entry point for ordinary scripts.
|
|
18
|
+
- Public result/reporting APIs: load a completed or resumed study and expose ranked candidates, usage, budgets, and diagnostics.
|
|
19
|
+
- Optional CLI or example entry points: demonstrate the library without becoming part of the core runtime contract.
|
|
20
|
+
|
|
21
|
+
## Intended execution flow
|
|
22
|
+
|
|
23
|
+
1. Resolve and validate configuration without persisting a wholesale environment snapshot.
|
|
24
|
+
2. Open or create a SQLite-backed study and initialize schema/version metadata.
|
|
25
|
+
3. Recover expired running attempts, apply retry policy, reconstruct budgets, and replay terminal observations when resuming.
|
|
26
|
+
4. Ask the selected strategy for a typed candidate from the search space.
|
|
27
|
+
5. Check the versioned cache using canonical candidate, dataset, evaluator, objective, and adapter identity parts.
|
|
28
|
+
6. Execute the user-owned RAG adapter against the evaluation split with timeout and cancellation boundaries.
|
|
29
|
+
7. Collect append-only usage records and exact/estimated/unavailable accounting.
|
|
30
|
+
8. Run evaluation plugins and resolve objective metrics, normalization, direction, weights, and coverage requirements.
|
|
31
|
+
9. Persist trial, metrics, usage, events, and cache data transactionally; emit callbacks/events without exposing secrets.
|
|
32
|
+
10. Update strategy state, evaluate stop conditions and budgets, then continue or finalize.
|
|
33
|
+
11. Return an actionable result with best candidates, tradeoffs, summaries, and recovery diagnostics.
|
|
34
|
+
|
|
35
|
+
## Dependency flow
|
|
36
|
+
|
|
37
|
+
Public models/contracts -> search/evaluation/objective/usage -> storage/cache -> orchestration -> optional integrations/examples.
|
|
38
|
+
|
|
39
|
+
## Persistence flow
|
|
40
|
+
|
|
41
|
+
Studies own ordered trials. Trials own attempts, candidate configurations, metrics, usage, and terminal events. Sequence numbers are monotonic and are never reused after recovery. Cache entries are independently versioned, checksummed, privacy-classified, and atomically written.
|
|
42
|
+
|
|
43
|
+
## Recovery flow
|
|
44
|
+
|
|
45
|
+
Load study and terminal observations, mark expired running attempts abandoned, apply retry policy, reconstruct exact and estimated budgets, initialize strategy state, replay observations, rebuild candidate hashes, continue sequence numbers, and emit a resume event with counts.
|
|
46
|
+
|
|
47
|
+
## Current implemented flow
|
|
48
|
+
|
|
49
|
+
The current optimization path resolves a new study, reserves candidates, checks the compatible
|
|
50
|
+
evaluation cache, starts leased attempts, iterates the repeatable dataset, calls the adapter and
|
|
51
|
+
evaluators under separate sample/trial deadlines, aggregates metrics, computes the objective,
|
|
52
|
+
and atomically commits outcomes. Retryable failures persist attempt usage before deterministic
|
|
53
|
+
backoff and a fresh attempt; exhausted failures remain isolated to their trial. Successful
|
|
54
|
+
uncached metrics are cached only after the durable trial commit, while cache failures become
|
|
55
|
+
events rather than trial failures. Resume validates the study configuration hash, rejects live
|
|
56
|
+
leases, abandons only expired attempts in the selected study, rebuilds terminal summaries and
|
|
57
|
+
usage, replays successful strategy observations, continues pending/recoverable trials, and
|
|
58
|
+
persists the final stop reason. The sync wrappers reject an active event loop. Callback dispatch
|
|
59
|
+
delivers newly committed events in sequence order, including retry scheduling before backoff;
|
|
60
|
+
callback failures append redacted diagnostics without failing optimization. Bounded concurrency
|
|
61
|
+
now schedules deterministic trial batches and independently bounded sample workers. Results and
|
|
62
|
+
strategy observations are applied by durable trial sequence after each batch.
|
|
63
|
+
|
|
64
|
+
## Update rule
|
|
65
|
+
|
|
66
|
+
When an entry point, dependency, state transition, or persistence boundary is implemented, update this file in the same feature branch and commit as part of that feature.
|