reprollm 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reprollm-0.1.1/.gitattributes +1 -0
- reprollm-0.1.1/.github/ISSUE_TEMPLATE/bug_report.yml +61 -0
- reprollm-0.1.1/.github/ISSUE_TEMPLATE/feature_request.yml +31 -0
- reprollm-0.1.1/.github/ISSUE_TEMPLATE/new_rule.yml +66 -0
- reprollm-0.1.1/.github/ISSUE_TEMPLATE/spec_change.yml +47 -0
- reprollm-0.1.1/.github/PULL_REQUEST_TEMPLATE.md +34 -0
- reprollm-0.1.1/.github/workflows/ci.yml +80 -0
- reprollm-0.1.1/.github/workflows/release.yml +86 -0
- reprollm-0.1.1/.gitignore +30 -0
- reprollm-0.1.1/AGENTS.md +146 -0
- reprollm-0.1.1/CHANGELOG.md +185 -0
- reprollm-0.1.1/CODE_OF_CONDUCT.md +81 -0
- reprollm-0.1.1/CONTRIBUTING.md +77 -0
- reprollm-0.1.1/LICENSE +201 -0
- reprollm-0.1.1/PKG-INFO +155 -0
- reprollm-0.1.1/README.md +121 -0
- reprollm-0.1.1/SECURITY.md +41 -0
- reprollm-0.1.1/pyproject.toml +102 -0
- reprollm-0.1.1/schemas/audit_report.schema.json +386 -0
- reprollm-0.1.1/schemas/config.schema.json +172 -0
- reprollm-0.1.1/schemas/diff_report.schema.json +21 -0
- reprollm-0.1.1/schemas/discover_candidates.schema.json +21 -0
- reprollm-0.1.1/schemas/lock.schema.json +1402 -0
- reprollm-0.1.1/schemas/manifest.schema.json +1755 -0
- reprollm-0.1.1/schemas/profile.schema.json +117 -0
- reprollm-0.1.1/schemas/project_rules.schema.json +181 -0
- reprollm-0.1.1/schemas/run_record.schema.json +675 -0
- reprollm-0.1.1/scripts/__init__.py +1 -0
- reprollm-0.1.1/scripts/gen_profiles_doc.py +76 -0
- reprollm-0.1.1/scripts/gen_rules_doc.py +64 -0
- reprollm-0.1.1/src/reprollm/__init__.py +3 -0
- reprollm-0.1.1/src/reprollm/cli/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/cli/audit.py +140 -0
- reprollm-0.1.1/src/reprollm/cli/doctor.py +47 -0
- reprollm-0.1.1/src/reprollm/cli/init.py +89 -0
- reprollm-0.1.1/src/reprollm/cli/main.py +86 -0
- reprollm-0.1.1/src/reprollm/cli/profiles.py +88 -0
- reprollm-0.1.1/src/reprollm/cli/schema.py +61 -0
- reprollm-0.1.1/src/reprollm/cli/templates/config.yaml.j2 +22 -0
- reprollm-0.1.1/src/reprollm/cli/templates/manifest.yaml.j2 +31 -0
- reprollm-0.1.1/src/reprollm/core/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/core/_toml.py +12 -0
- reprollm-0.1.1/src/reprollm/core/config.py +67 -0
- reprollm-0.1.1/src/reprollm/core/context.py +100 -0
- reprollm-0.1.1/src/reprollm/core/deps.py +533 -0
- reprollm-0.1.1/src/reprollm/core/diagnostics.py +138 -0
- reprollm-0.1.1/src/reprollm/core/engine.py +242 -0
- reprollm-0.1.1/src/reprollm/core/envinfo.py +80 -0
- reprollm-0.1.1/src/reprollm/core/errors.py +24 -0
- reprollm-0.1.1/src/reprollm/core/git.py +136 -0
- reprollm-0.1.1/src/reprollm/core/hashing.py +41 -0
- reprollm-0.1.1/src/reprollm/core/levels.py +23 -0
- reprollm-0.1.1/src/reprollm/core/manifest_scaffold.py +404 -0
- reprollm-0.1.1/src/reprollm/core/paths.py +79 -0
- reprollm-0.1.1/src/reprollm/core/precedence.py +1 -0
- reprollm-0.1.1/src/reprollm/core/proc.py +54 -0
- reprollm-0.1.1/src/reprollm/core/pyscan.py +118 -0
- reprollm-0.1.1/src/reprollm/core/redaction.py +44 -0
- reprollm-0.1.1/src/reprollm/core/registry.py +113 -0
- reprollm-0.1.1/src/reprollm/core/scanner.py +207 -0
- reprollm-0.1.1/src/reprollm/core/yaml_io.py +69 -0
- reprollm-0.1.1/src/reprollm/diff/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/diff/differ.py +1 -0
- reprollm-0.1.1/src/reprollm/diff/severity.py +1 -0
- reprollm-0.1.1/src/reprollm/diff/state.py +1 -0
- reprollm-0.1.1/src/reprollm/discover/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/discover/candidates.py +1 -0
- reprollm-0.1.1/src/reprollm/discover/client.py +1 -0
- reprollm-0.1.1/src/reprollm/discover/collector.py +1 -0
- reprollm-0.1.1/src/reprollm/export/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/export/exporter.py +1 -0
- reprollm-0.1.1/src/reprollm/export/templates/.gitkeep +0 -0
- reprollm-0.1.1/src/reprollm/integrations/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/integrations/base.py +1 -0
- reprollm-0.1.1/src/reprollm/integrations/huggingface.py +1 -0
- reprollm-0.1.1/src/reprollm/integrations/openai_.py +1 -0
- reprollm-0.1.1/src/reprollm/integrations/peft.py +1 -0
- reprollm-0.1.1/src/reprollm/integrations/transformers_.py +1 -0
- reprollm-0.1.1/src/reprollm/integrations/vllm.py +1 -0
- reprollm-0.1.1/src/reprollm/lock/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/lock/api_resolver.py +1 -0
- reprollm-0.1.1/src/reprollm/lock/hf_resolver.py +1 -0
- reprollm-0.1.1/src/reprollm/lock/local_resolver.py +1 -0
- reprollm-0.1.1/src/reprollm/lock/resolver.py +1 -0
- reprollm-0.1.1/src/reprollm/profiles/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/profiles/core.yaml +46 -0
- reprollm-0.1.1/src/reprollm/profiles/detect.py +349 -0
- reprollm-0.1.1/src/reprollm/profiles/evaluation.yaml +30 -0
- reprollm-0.1.1/src/reprollm/profiles/finetuning.yaml +27 -0
- reprollm-0.1.1/src/reprollm/profiles/inference.yaml +29 -0
- reprollm-0.1.1/src/reprollm/profiles/llm_judge.yaml +24 -0
- reprollm-0.1.1/src/reprollm/profiles/loader.py +193 -0
- reprollm-0.1.1/src/reprollm/profiles/privacy.yaml +20 -0
- reprollm-0.1.1/src/reprollm/profiles/safety.yaml +19 -0
- reprollm-0.1.1/src/reprollm/reporters/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/reporters/json_.py +12 -0
- reprollm-0.1.1/src/reprollm/reporters/text.py +125 -0
- reprollm-0.1.1/src/reprollm/rules/__init__.py +20 -0
- reprollm-0.1.1/src/reprollm/rules/_presence.py +68 -0
- reprollm-0.1.1/src/reprollm/rules/_stubs.py +17 -0
- reprollm-0.1.1/src/reprollm/rules/code.py +212 -0
- reprollm-0.1.1/src/reprollm/rules/consistency.py +64 -0
- reprollm-0.1.1/src/reprollm/rules/dataset.py +121 -0
- reprollm-0.1.1/src/reprollm/rules/env.py +329 -0
- reprollm-0.1.1/src/reprollm/rules/eval_.py +188 -0
- reprollm-0.1.1/src/reprollm/rules/exec_.py +184 -0
- reprollm-0.1.1/src/reprollm/rules/gen.py +123 -0
- reprollm-0.1.1/src/reprollm/rules/judge.py +113 -0
- reprollm-0.1.1/src/reprollm/rules/model.py +200 -0
- reprollm-0.1.1/src/reprollm/rules/privacy.py +98 -0
- reprollm-0.1.1/src/reprollm/rules/project.py +1 -0
- reprollm-0.1.1/src/reprollm/rules/prompt.py +93 -0
- reprollm-0.1.1/src/reprollm/rules/train.py +146 -0
- reprollm-0.1.1/src/reprollm/run/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/run/capture.py +1 -0
- reprollm-0.1.1/src/reprollm/run/hardware.py +1 -0
- reprollm-0.1.1/src/reprollm/run/slurm.py +1 -0
- reprollm-0.1.1/src/reprollm/run/wrapper.py +1 -0
- reprollm-0.1.1/src/reprollm/schemas/__init__.py +1 -0
- reprollm-0.1.1/src/reprollm/schemas/config.py +67 -0
- reprollm-0.1.1/src/reprollm/schemas/diff_report.py +18 -0
- reprollm-0.1.1/src/reprollm/schemas/discover_candidates.py +18 -0
- reprollm-0.1.1/src/reprollm/schemas/finding.py +164 -0
- reprollm-0.1.1/src/reprollm/schemas/lock.py +240 -0
- reprollm-0.1.1/src/reprollm/schemas/manifest.py +509 -0
- reprollm-0.1.1/src/reprollm/schemas/profile.py +51 -0
- reprollm-0.1.1/src/reprollm/schemas/project_rules.py +81 -0
- reprollm-0.1.1/src/reprollm/schemas/run_record.py +167 -0
- reprollm-0.1.1/uv.lock +1419 -0
- reprollm-0.1.1/val.md +226 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
* text=auto eol=lf
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
name: Bug report
|
|
2
|
+
description: Something behaved incorrectly or crashed.
|
|
3
|
+
labels: ["bug"]
|
|
4
|
+
body:
|
|
5
|
+
- type: textarea
|
|
6
|
+
id: what-happened
|
|
7
|
+
attributes:
|
|
8
|
+
label: What happened?
|
|
9
|
+
description: A clear description of the incorrect behavior. If ReproLLM crashed, paste the traceback (run with -v for the internal traceback).
|
|
10
|
+
validations:
|
|
11
|
+
required: true
|
|
12
|
+
- type: textarea
|
|
13
|
+
id: expected
|
|
14
|
+
attributes:
|
|
15
|
+
label: What did you expect to happen?
|
|
16
|
+
validations:
|
|
17
|
+
required: true
|
|
18
|
+
- type: input
|
|
19
|
+
id: version
|
|
20
|
+
attributes:
|
|
21
|
+
label: reprollm version
|
|
22
|
+
description: Output of `reprollm --version`
|
|
23
|
+
placeholder: "reprollm 0.1.1"
|
|
24
|
+
validations:
|
|
25
|
+
required: true
|
|
26
|
+
- type: dropdown
|
|
27
|
+
id: command
|
|
28
|
+
attributes:
|
|
29
|
+
label: Which command?
|
|
30
|
+
options:
|
|
31
|
+
- reprollm audit
|
|
32
|
+
- reprollm init
|
|
33
|
+
- reprollm lock
|
|
34
|
+
- reprollm run / runs
|
|
35
|
+
- reprollm diff
|
|
36
|
+
- reprollm export
|
|
37
|
+
- reprollm discover
|
|
38
|
+
- reprollm rules
|
|
39
|
+
- reprollm doctor
|
|
40
|
+
- reprollm schema export
|
|
41
|
+
- other / not command-specific
|
|
42
|
+
validations:
|
|
43
|
+
required: true
|
|
44
|
+
- type: input
|
|
45
|
+
id: rule
|
|
46
|
+
attributes:
|
|
47
|
+
label: Rule ID (if an audit finding is wrong)
|
|
48
|
+
description: e.g. `model.revision_pinned` — leave empty otherwise
|
|
49
|
+
placeholder: "model.revision_pinned"
|
|
50
|
+
- type: textarea
|
|
51
|
+
id: repro
|
|
52
|
+
attributes:
|
|
53
|
+
label: Minimal reproduction
|
|
54
|
+
description: Commands, manifest fragment, or fixture tree that triggers the bug. Redact secrets.
|
|
55
|
+
validations:
|
|
56
|
+
required: true
|
|
57
|
+
- type: input
|
|
58
|
+
id: os
|
|
59
|
+
attributes:
|
|
60
|
+
label: OS / platform
|
|
61
|
+
placeholder: "Ubuntu 24.04, Python 3.12"
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
name: Feature request
|
|
2
|
+
description: Propose a new capability that does not change the specification.
|
|
3
|
+
labels: ["enhancement"]
|
|
4
|
+
body:
|
|
5
|
+
- type: textarea
|
|
6
|
+
id: problem
|
|
7
|
+
attributes:
|
|
8
|
+
label: What problem does this solve?
|
|
9
|
+
description: The reproducibility pain you hit, with a concrete example.
|
|
10
|
+
validations:
|
|
11
|
+
required: true
|
|
12
|
+
- type: textarea
|
|
13
|
+
id: proposal
|
|
14
|
+
attributes:
|
|
15
|
+
label: Proposed behavior
|
|
16
|
+
description: What should ReproLLM do instead? Which command, manifest field, or rule output changes?
|
|
17
|
+
validations:
|
|
18
|
+
required: true
|
|
19
|
+
- type: checkboxes
|
|
20
|
+
id: non-goals
|
|
21
|
+
attributes:
|
|
22
|
+
label: Scope check
|
|
23
|
+
options:
|
|
24
|
+
- label: This does not require violating a frozen decision in docs/plan/00 (D-01 … D-42)
|
|
25
|
+
required: true
|
|
26
|
+
- label: This does not put an LLM in the audit path (only `discover` may call a model)
|
|
27
|
+
required: true
|
|
28
|
+
- type: textarea
|
|
29
|
+
id: alternatives
|
|
30
|
+
attributes:
|
|
31
|
+
label: Alternatives you considered
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
name: New audit rule proposal
|
|
2
|
+
description: Propose a new deterministic rule for the catalog (spec §12).
|
|
3
|
+
labels: ["rule", "enhancement"]
|
|
4
|
+
body:
|
|
5
|
+
- type: input
|
|
6
|
+
id: rule-id
|
|
7
|
+
attributes:
|
|
8
|
+
label: Proposed rule ID
|
|
9
|
+
description: "`<category>.<snake_name>` per D-08, e.g. `model.revision_pinned`"
|
|
10
|
+
placeholder: "model.revision_pinned"
|
|
11
|
+
validations:
|
|
12
|
+
required: true
|
|
13
|
+
- type: dropdown
|
|
14
|
+
id: category
|
|
15
|
+
attributes:
|
|
16
|
+
label: Category
|
|
17
|
+
options:
|
|
18
|
+
- code
|
|
19
|
+
- env
|
|
20
|
+
- exec
|
|
21
|
+
- model
|
|
22
|
+
- dataset
|
|
23
|
+
- gen
|
|
24
|
+
- prompt
|
|
25
|
+
- eval
|
|
26
|
+
- judge
|
|
27
|
+
- train
|
|
28
|
+
- privacy
|
|
29
|
+
- consistency
|
|
30
|
+
- project
|
|
31
|
+
validations:
|
|
32
|
+
required: true
|
|
33
|
+
- type: dropdown
|
|
34
|
+
id: severity
|
|
35
|
+
attributes:
|
|
36
|
+
label: Default severity (D-09)
|
|
37
|
+
options:
|
|
38
|
+
- CRITICAL — identity of the experiment becomes ambiguous
|
|
39
|
+
- WARNING — reproduction is materially harder but identity is determinable
|
|
40
|
+
- INFO — advisory
|
|
41
|
+
validations:
|
|
42
|
+
required: true
|
|
43
|
+
- type: textarea
|
|
44
|
+
id: fail-condition
|
|
45
|
+
attributes:
|
|
46
|
+
label: FAIL condition
|
|
47
|
+
description: Precisely when does the rule fail? Which inputs does it read (manifest field paths, lock fields, run observations)?
|
|
48
|
+
validations:
|
|
49
|
+
required: true
|
|
50
|
+
- type: textarea
|
|
51
|
+
id: fix-hint
|
|
52
|
+
attributes:
|
|
53
|
+
label: fix_hint text
|
|
54
|
+
description: One imperative sentence naming the field path or file to change.
|
|
55
|
+
validations:
|
|
56
|
+
required: true
|
|
57
|
+
- type: input
|
|
58
|
+
id: profile
|
|
59
|
+
attributes:
|
|
60
|
+
label: Which profile(s) should select it?
|
|
61
|
+
description: "core / inference / evaluation / llm_judge / finetuning / safety / privacy"
|
|
62
|
+
- type: textarea
|
|
63
|
+
id: evidence
|
|
64
|
+
attributes:
|
|
65
|
+
label: Real-world evidence
|
|
66
|
+
description: A paper, repo, or incident where the missing information broke reproducibility.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
name: Specification change
|
|
2
|
+
description: Propose a change to the normative Beta specification (docs/plan/01) or a frozen decision (docs/plan/00).
|
|
3
|
+
labels: ["spec"]
|
|
4
|
+
body:
|
|
5
|
+
- type: dropdown
|
|
6
|
+
id: document
|
|
7
|
+
attributes:
|
|
8
|
+
label: Which document?
|
|
9
|
+
options:
|
|
10
|
+
- docs/plan/01_specification.md (CLI contract, schemas, rules, redaction, diff)
|
|
11
|
+
- docs/plan/00_architecture_and_decisions.md (frozen decisions D-nn)
|
|
12
|
+
validations:
|
|
13
|
+
required: true
|
|
14
|
+
- type: input
|
|
15
|
+
id: section
|
|
16
|
+
attributes:
|
|
17
|
+
label: Section / decision ID
|
|
18
|
+
description: e.g. "§4.3 resolution rules" or "D-21"
|
|
19
|
+
validations:
|
|
20
|
+
required: true
|
|
21
|
+
- type: textarea
|
|
22
|
+
id: current
|
|
23
|
+
attributes:
|
|
24
|
+
label: Current text / behavior
|
|
25
|
+
validations:
|
|
26
|
+
required: true
|
|
27
|
+
- type: textarea
|
|
28
|
+
id: proposed
|
|
29
|
+
attributes:
|
|
30
|
+
label: Proposed change
|
|
31
|
+
description: Exact replacement text or new field/behavior. Include schemas and exported JSON Schema impact.
|
|
32
|
+
validations:
|
|
33
|
+
required: true
|
|
34
|
+
- type: textarea
|
|
35
|
+
id: rationale
|
|
36
|
+
attributes:
|
|
37
|
+
label: Rationale
|
|
38
|
+
description: Why the spec is wrong or incomplete. Dogfooding evidence welcome.
|
|
39
|
+
validations:
|
|
40
|
+
required: true
|
|
41
|
+
- type: checkboxes
|
|
42
|
+
id: compat
|
|
43
|
+
attributes:
|
|
44
|
+
label: Compatibility
|
|
45
|
+
options:
|
|
46
|
+
- label: I checked whether this is a breaking schema change (requires migration note + version bump policy D-39)
|
|
47
|
+
required: true
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
<!--
|
|
2
|
+
PR checklist (see AGENTS.md §7). Fill in every section; CI enforces the technical items.
|
|
3
|
+
-->
|
|
4
|
+
|
|
5
|
+
Closes #
|
|
6
|
+
|
|
7
|
+
## Spec sections implemented
|
|
8
|
+
|
|
9
|
+
<!-- e.g. "spec §12.4 model.*, §6.1 profile wiring" — cite docs/plan/01_specification.md -->
|
|
10
|
+
|
|
11
|
+
## Test summary
|
|
12
|
+
|
|
13
|
+
<!-- which acceptance criteria are covered by tests; paste the FAIL output of any new
|
|
14
|
+
rule's message + fix_hint here (required for rule PRs) -->
|
|
15
|
+
|
|
16
|
+
## Fixtures / snapshots
|
|
17
|
+
|
|
18
|
+
<!-- were tests/fixtures/**/expected/*.json or other snapshots updated? why is the new
|
|
19
|
+
output correct? (REPROLLM_UPDATE_SNAPSHOTS=1 output diff must be reviewed) -->
|
|
20
|
+
|
|
21
|
+
## Checklist
|
|
22
|
+
|
|
23
|
+
- [ ] Tests written for the acceptance criteria and passing
|
|
24
|
+
- [ ] `pytest -q` passes with no network access
|
|
25
|
+
- [ ] `ruff check . && ruff format --check .` pass
|
|
26
|
+
- [ ] `mypy src/` passes (`--strict`); any new `type: ignore` has an inline reason
|
|
27
|
+
- [ ] No new heavy dependencies (torch / transformers / vllm / huggingface_hub / numpy
|
|
28
|
+
or anything that pulls them)
|
|
29
|
+
- [ ] Persisted artifacts: sorted keys, relative POSIX paths, no hostnames/usernames/
|
|
30
|
+
absolute paths/secrets; `schema_version` present on new documents
|
|
31
|
+
- [ ] `schemas/` re-exported (`reprollm schema export --out schemas/`) and committed,
|
|
32
|
+
if any schema changed
|
|
33
|
+
- [ ] CHANGELOG entry added under `Unreleased`
|
|
34
|
+
- [ ] No files modified outside the issue's scope
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
concurrency:
|
|
12
|
+
group: ${{ github.workflow }}-${{ github.ref }}
|
|
13
|
+
cancel-in-progress: true
|
|
14
|
+
|
|
15
|
+
jobs:
|
|
16
|
+
quality:
|
|
17
|
+
name: ${{ matrix.os }} / py${{ matrix.python }}
|
|
18
|
+
runs-on: ${{ matrix.os }}
|
|
19
|
+
defaults:
|
|
20
|
+
run:
|
|
21
|
+
shell: bash
|
|
22
|
+
strategy:
|
|
23
|
+
fail-fast: false
|
|
24
|
+
matrix:
|
|
25
|
+
include:
|
|
26
|
+
- os: ubuntu-latest
|
|
27
|
+
python: "3.10"
|
|
28
|
+
- os: ubuntu-latest
|
|
29
|
+
python: "3.11"
|
|
30
|
+
- os: ubuntu-latest
|
|
31
|
+
python: "3.12"
|
|
32
|
+
schema-check: true
|
|
33
|
+
- os: macos-latest
|
|
34
|
+
python: "3.12"
|
|
35
|
+
- os: windows-latest
|
|
36
|
+
python: "3.12"
|
|
37
|
+
pytest-marker-filter: not linux_only
|
|
38
|
+
steps:
|
|
39
|
+
- uses: actions/checkout@v4
|
|
40
|
+
|
|
41
|
+
- uses: astral-sh/setup-uv@v6
|
|
42
|
+
with:
|
|
43
|
+
python-version: ${{ matrix.python }}
|
|
44
|
+
enable-cache: true
|
|
45
|
+
|
|
46
|
+
- name: Install dependencies
|
|
47
|
+
run: uv sync --dev
|
|
48
|
+
|
|
49
|
+
- name: Lint
|
|
50
|
+
run: uv run ruff check .
|
|
51
|
+
|
|
52
|
+
- name: Format check
|
|
53
|
+
run: uv run ruff format --check .
|
|
54
|
+
|
|
55
|
+
- name: Type check
|
|
56
|
+
run: uv run mypy src/
|
|
57
|
+
|
|
58
|
+
- name: Test
|
|
59
|
+
run: >
|
|
60
|
+
uv run pytest -q --cov=reprollm --cov-report=xml --cov-fail-under=85
|
|
61
|
+
${{ matrix.pytest-marker-filter && format('-m "{0}"', matrix.pytest-marker-filter) || '' }}
|
|
62
|
+
|
|
63
|
+
- name: Schema freshness
|
|
64
|
+
if: matrix.schema-check
|
|
65
|
+
run: |
|
|
66
|
+
uv run reprollm schema export --out /tmp/schemas
|
|
67
|
+
diff -r schemas /tmp/schemas
|
|
68
|
+
|
|
69
|
+
- name: Generated documentation freshness
|
|
70
|
+
if: matrix.schema-check
|
|
71
|
+
run: |
|
|
72
|
+
uv run python scripts/gen_rules_doc.py --check
|
|
73
|
+
uv run python scripts/gen_profiles_doc.py --check
|
|
74
|
+
|
|
75
|
+
- name: Upload coverage artifact
|
|
76
|
+
uses: actions/upload-artifact@v4
|
|
77
|
+
with:
|
|
78
|
+
name: coverage-${{ matrix.os }}-py${{ matrix.python }}
|
|
79
|
+
path: coverage.xml
|
|
80
|
+
if-no-files-found: ignore
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
permissions: {}
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
build:
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
|
|
16
|
+
- uses: astral-sh/setup-uv@v6
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
|
|
20
|
+
- name: Install dependencies
|
|
21
|
+
run: uv sync --dev
|
|
22
|
+
|
|
23
|
+
- name: Verify tag matches package version
|
|
24
|
+
run: |
|
|
25
|
+
TAG_VERSION="${GITHUB_REF_NAME#v}"
|
|
26
|
+
PKG_VERSION="$(uv run python -c 'from reprollm import __version__; print(__version__)')"
|
|
27
|
+
if [ "$TAG_VERSION" != "$PKG_VERSION" ]; then
|
|
28
|
+
echo "tag $GITHUB_REF_NAME does not match package version $PKG_VERSION" >&2
|
|
29
|
+
exit 1
|
|
30
|
+
fi
|
|
31
|
+
|
|
32
|
+
- name: Build distributions
|
|
33
|
+
run: uv build
|
|
34
|
+
|
|
35
|
+
- uses: actions/upload-artifact@v4
|
|
36
|
+
with:
|
|
37
|
+
name: dist
|
|
38
|
+
path: dist/
|
|
39
|
+
|
|
40
|
+
publish:
|
|
41
|
+
needs: build
|
|
42
|
+
runs-on: ubuntu-latest
|
|
43
|
+
environment: pypi
|
|
44
|
+
permissions:
|
|
45
|
+
# Trusted publishing (PyPI OIDC); no API token is used or stored.
|
|
46
|
+
id-token: write
|
|
47
|
+
steps:
|
|
48
|
+
- uses: actions/download-artifact@v4
|
|
49
|
+
with:
|
|
50
|
+
name: dist
|
|
51
|
+
path: dist/
|
|
52
|
+
|
|
53
|
+
- name: Publish to PyPI
|
|
54
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
55
|
+
|
|
56
|
+
github-release:
|
|
57
|
+
needs: publish
|
|
58
|
+
runs-on: ubuntu-latest
|
|
59
|
+
permissions:
|
|
60
|
+
contents: write
|
|
61
|
+
steps:
|
|
62
|
+
- uses: actions/checkout@v4
|
|
63
|
+
|
|
64
|
+
- uses: actions/download-artifact@v4
|
|
65
|
+
with:
|
|
66
|
+
name: dist
|
|
67
|
+
path: dist/
|
|
68
|
+
|
|
69
|
+
- name: Extract release notes from CHANGELOG
|
|
70
|
+
run: |
|
|
71
|
+
VERSION="${GITHUB_REF_NAME#v}"
|
|
72
|
+
awk -v ver="$VERSION" '
|
|
73
|
+
index($0, "## [" ver "]") == 1 { in_section = 1; next }
|
|
74
|
+
/^## \[/ && in_section { exit }
|
|
75
|
+
in_section && !/^\[[^]]*\]:/ { print }
|
|
76
|
+
' CHANGELOG.md > release_notes.md
|
|
77
|
+
if [ ! -s release_notes.md ]; then
|
|
78
|
+
echo "no CHANGELOG section found for $VERSION" >&2
|
|
79
|
+
exit 1
|
|
80
|
+
fi
|
|
81
|
+
|
|
82
|
+
- name: Create GitHub Release
|
|
83
|
+
uses: softprops/action-gh-release@v2
|
|
84
|
+
with:
|
|
85
|
+
files: dist/*
|
|
86
|
+
body_path: release_notes.md
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Byte-compiled / cache
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
.mypy_cache/
|
|
5
|
+
.ruff_cache/
|
|
6
|
+
.pytest_cache/
|
|
7
|
+
|
|
8
|
+
# Distribution / packaging
|
|
9
|
+
build/
|
|
10
|
+
dist/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
.eggs/
|
|
13
|
+
|
|
14
|
+
# Environments
|
|
15
|
+
.venv/
|
|
16
|
+
venv/
|
|
17
|
+
|
|
18
|
+
# Coverage
|
|
19
|
+
.coverage
|
|
20
|
+
.coverage.*
|
|
21
|
+
coverage.xml
|
|
22
|
+
htmlcov/
|
|
23
|
+
|
|
24
|
+
# Editors / OS
|
|
25
|
+
.idea/
|
|
26
|
+
.vscode/
|
|
27
|
+
.DS_Store
|
|
28
|
+
|
|
29
|
+
# NOTE: `.reprollm/` is intentionally NOT ignored. Users are expected to commit
|
|
30
|
+
# their manifests, lockfiles, and project rules (see docs/plan/00 §7).
|
reprollm-0.1.1/AGENTS.md
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# AGENTS.md — Working conventions for coding agents on ReproLLM
|
|
2
|
+
|
|
3
|
+
This file is read by coding agents (Codex CLI, Cursor, and others) working in this repository. It is short on purpose; the authoritative documents are linked.
|
|
4
|
+
|
|
5
|
+
## 1. What this project is
|
|
6
|
+
|
|
7
|
+
ReproLLM is a CLI-first, local-first reproducibility toolkit for LLM research experiments: a deterministic audit (`reprollm audit`), a manifest (`reprollm.yaml`), a lockfile (`reprollm.lock`), runtime capture (`reprollm run`), semantic drift detection (`reprollm diff`), export (`reprollm export`), and an opt-in LLM-assisted discovery step (`reprollm discover`).
|
|
8
|
+
|
|
9
|
+
Principle: **LLM discovers. Rules decide. Runtime verifies.**
|
|
10
|
+
|
|
11
|
+
## 2. Read before you write
|
|
12
|
+
|
|
13
|
+
**Every development session starts from the plan documents — no session invents its own scope.**
|
|
14
|
+
|
|
15
|
+
1. `docs/plan/00_architecture_and_decisions.md` — frozen decisions (`D-nn`). Never violate one; if a task seems to require it, stop and report.
|
|
16
|
+
2. `docs/plan/01_specification.md` — normative CLI contract, schemas, rule catalog, redaction policy, diff semantics, test requirements. Field names and behaviors come from here, not from memory.
|
|
17
|
+
3. The current sprint document (`docs/plan/M<N>_*.md`) — the task list (T-numbers), each task's acceptance criteria, and this sprint's forbidden zone (禁区). Tasks come from this document, in order; acceptance criteria are turned into tests *before* implementation; the forbidden zone is binding.
|
|
18
|
+
|
|
19
|
+
If the task description and the specification disagree, the specification wins; say so in the session report or PR.
|
|
20
|
+
|
|
21
|
+
## 3. Non-negotiables
|
|
22
|
+
|
|
23
|
+
- **No LLM in the audit path.** Only `reprollm discover` may call a model endpoint, and only behind `--experimental`/config opt-in.
|
|
24
|
+
- **No heavy dependencies.** Never add `torch`, `transformers`, `vllm`, `huggingface_hub`, `numpy`, or anything that pulls them. Call the HF Hub HTTP API with `httpx`. Read installed versions with `importlib.metadata`, never `import` the package.
|
|
25
|
+
- **No network in tests.** All `httpx` calls are mocked with `respx`; an autouse fixture fails unmocked requests. `git` and `nvidia-smi` go through `reprollm.core.proc.run_cmd` so tests can stub them.
|
|
26
|
+
- **Rule IDs are frozen once released.** Renaming requires an `aliases` entry. Never reuse an ID for different semantics.
|
|
27
|
+
- **Never write user files** except the artifacts each command owns (`init` → manifest/.reprollm; `lock` → reprollm.lock; `run` → run directory; `rules accept/ignore/add` → project-rules.yaml). Never modify user code, configs, or prompts.
|
|
28
|
+
- **Redaction is a security boundary.** Any change to `src/reprollm/core/redaction.py` must keep 100 % branch coverage and add cases to `tests/fixtures/secrets/`. A redaction bypass is a security bug, not a normal bug.
|
|
29
|
+
- **Persisted artifacts contain no absolute paths, hostnames, usernames, or secrets.**
|
|
30
|
+
- **Deterministic output.** Same inputs → byte-identical JSON (modulo timestamp fields listed in spec §22 T-02). Sort everything you emit.
|
|
31
|
+
- **`schema_version` on every persisted document.** Breaking a schema requires a CHANGELOG migration note and a spec update in the same PR.
|
|
32
|
+
|
|
33
|
+
## 4. Repository map
|
|
34
|
+
|
|
35
|
+
```text
|
|
36
|
+
src/reprollm/
|
|
37
|
+
cli/ typer commands; thin, no business logic
|
|
38
|
+
schemas/ pydantic v2 models (manifest, lock, run_record, profile, project_rules, config, finding, discover, state)
|
|
39
|
+
core/ paths, hashing, git, envinfo, proc, redaction, yaml_io, registry, engine, context, levels, precedence, errors
|
|
40
|
+
rules/ one module per category; each rule is a @register_rule class
|
|
41
|
+
profiles/ built-in *.yaml, loader.py, detect.py
|
|
42
|
+
integrations/ huggingface, vllm, openai_, peft, transformers_
|
|
43
|
+
lock/ run/ diff/ export/ discover/ reporters/
|
|
44
|
+
tests/
|
|
45
|
+
unit/ mirrors src layout
|
|
46
|
+
fixtures/repos/ golden repositories (materialized as git repos at test time)
|
|
47
|
+
fixtures/secrets/ redaction corpus
|
|
48
|
+
fixtures/hf_api/ recorded Hub responses
|
|
49
|
+
fixtures/nvidia_smi/ canned outputs
|
|
50
|
+
schemas/ exported JSON Schema (generated; CI checks freshness)
|
|
51
|
+
docs/ markdown docs; docs/plan/ holds the planning documents
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## 5. Environment and commands
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
uv sync --dev # create/refresh the environment (no optional extras exist in Beta)
|
|
58
|
+
uv run reprollm --help
|
|
59
|
+
uv run pytest -q # full suite, no network
|
|
60
|
+
uv run pytest -q tests/unit/rules # subset
|
|
61
|
+
uv run ruff check . && uv run ruff format --check .
|
|
62
|
+
uv run mypy src/
|
|
63
|
+
uv run reprollm schema export --out schemas/ # after changing any schema; commit the result
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Python ≥ 3.10. Do not use features newer than 3.10 in `src/`.
|
|
67
|
+
|
|
68
|
+
## 6. How to add things
|
|
69
|
+
|
|
70
|
+
**A rule** (`src/reprollm/rules/<category>.py`):
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
@register_rule
|
|
74
|
+
class ModelRevisionPinned(Rule):
|
|
75
|
+
id = "model.revision_pinned"
|
|
76
|
+
category = "model"
|
|
77
|
+
default_severity = Severity.CRITICAL
|
|
78
|
+
min_level = 2
|
|
79
|
+
description = "Every model has an exact resolved revision or an explicit pinnability record."
|
|
80
|
+
fix_hint = (
|
|
81
|
+
"Run `reprollm lock` with network access, or set models.<role>.revision to a commit sha."
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
def applies(self, ctx: AuditContext) -> bool: ...
|
|
85
|
+
def check(self, ctx: AuditContext) -> list[Finding]: ...
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Then: add it to the right profile YAML(s) per spec §6.1; add a PASS and a FAIL test in `tests/unit/rules/test_<category>.py`; update fixture `expected/*.json` snapshots if affected (`REPROLLM_UPDATE_SNAPSHOTS=1 uv run pytest tests/...` only after confirming the new output is correct by reading the diff); add a CHANGELOG line.
|
|
89
|
+
|
|
90
|
+
**A profile**: `src/reprollm/profiles/<name>.yaml` per spec §6; a loader test asserting inheritance closure; `reprollm profiles show <name>` output test.
|
|
91
|
+
|
|
92
|
+
**An integration**: `src/reprollm/integrations/<name>.py` implementing the `Integration` protocol (spec §14); it must not import the target library.
|
|
93
|
+
|
|
94
|
+
**A schema field**: update the pydantic model, spec §3–§8 (in the same PR), exported `schemas/`, fixtures, and CHANGELOG.
|
|
95
|
+
|
|
96
|
+
## 7. Task workflow
|
|
97
|
+
|
|
98
|
+
Every development session follows the plan documents in `docs/plan/`:
|
|
99
|
+
|
|
100
|
+
1. **Scope comes from the current sprint doc** (`docs/plan/M<N>_*.md`): implement its tasks (T-numbers) in order, honoring each task's acceptance criteria and the sprint's forbidden zone (§2). A session never invents, reorders into unrelated territory, or "improves" beyond the sprint list; anything found along the way that belongs to a later sprint goes to a backlog note, not into the diff.
|
|
101
|
+
2. **Sessions per sprint**: by maintainer decision a sprint is delivered in two sessions (typically first half / second half of the task list); both follow this protocol, and the second session starts from the first session's report.
|
|
102
|
+
3. **Per task**: write the tests implied by the acceptance criteria first, then implement, then run the full quality gate (`pytest`, `ruff`, `mypy`, schema freshness). One task = one commit with Conventional Commits (`feat(rules): add model.revision_pinned`, `fix(redaction): handle jwt with padding`, `test(fixtures): add dirty_tree repo`, `docs: …`, `chore: …`). The commit message cites the spec sections implemented, notes fixture/snapshot changes and why, and references the sprint task (e.g. "M2-T03").
|
|
103
|
+
4. **Before pushing**: quality gate green **and** the §10 dogfooding run done; then push directly to `main` (the branch/issue/PR flow applies only when the maintainer explicitly asks for a reviewed change). Session reports list: tasks completed with commit hashes, dogfooding delta versus the last baseline, deviations from the sprint doc and why.
|
|
104
|
+
5. Do not mix unrelated tasks in one commit, reformat unrelated files, or bump dependencies without being asked.
|
|
105
|
+
6. If the spec is wrong or incomplete, open an issue labeled `spec` with a concrete proposal and stop that part of the work; implement the rest. If a sprint task seems to require violating a frozen decision (`D-nn`) or the sprint's forbidden zone, stop and report — never proceed silently.
|
|
106
|
+
|
|
107
|
+
## 8. Style
|
|
108
|
+
|
|
109
|
+
- Type hints everywhere; `mypy --strict` clean.
|
|
110
|
+
- `ruff` defaults plus `I` (isort), `UP`, `B`, `SIM`; line length 100.
|
|
111
|
+
- Pure functions in `core/`; side effects (filesystem, subprocess, HTTP) isolated behind small interfaces that tests can stub.
|
|
112
|
+
- Errors: raise `reprollm.core.errors.UserError` (exit 2) for user-fixable problems with an actionable message; anything else is an internal error (exit 3).
|
|
113
|
+
- Comments explain *why*, not *what*. No narrating comments. No emojis in code or output except the fixed symbols defined in spec §21.
|
|
114
|
+
- User-facing strings: concise, imperative fix hints, always mention the field path or file.
|
|
115
|
+
|
|
116
|
+
## 9. What "done" means
|
|
117
|
+
|
|
118
|
+
A task is done when: tests for its acceptance criteria (from the sprint doc) exist and pass; quality gate is green locally and in CI; fixture snapshots were updated deliberately (with the diff reviewed, not blind-regenerated); the CHANGELOG has an entry; the commit references its sprint task and spec sections; and — where the sprint doc requires it — the command has been run against one of the golden fixtures with the expected result pasted into the session report. A *session* is done when its tasks are done, the §10 dogfooding run is recorded, and `main` is pushed with CI green.
|
|
119
|
+
|
|
120
|
+
## 10. Standing five-project validation gate
|
|
121
|
+
|
|
122
|
+
Golden fixtures are necessary but not sufficient. **After the quality gate, every development
|
|
123
|
+
session MUST read and execute the applicable gates in [`val.md`](val.md).** That document is the
|
|
124
|
+
single source of truth for the five pinned repositories, gold answers, commands, milestone resource
|
|
125
|
+
gates, and baseline-update policy.
|
|
126
|
+
|
|
127
|
+
Per session:
|
|
128
|
+
|
|
129
|
+
1. Run Gate A against all five valid pinned checkouts; at minimum run Level 0 audit, plus every
|
|
130
|
+
non-resource command changed by the session.
|
|
131
|
+
2. When the session completes a milestone or prepares a release, also run every Gate B scenario
|
|
132
|
+
activated for that milestone. Real network resolution starts at M4; minimal inference/training,
|
|
133
|
+
including a bounded local judge case, starts at M5; paired semantic diff starts at M6; and
|
|
134
|
+
export/discover plus paid API paths start at M7. Run the complete applicable matrix at M8 and
|
|
135
|
+
every release.
|
|
136
|
+
3. Run mutating commands only in disposable clones/worktrees. Use credentials, paid APIs, restricted
|
|
137
|
+
data, and GPU resources only when they are explicitly provisioned for validation.
|
|
138
|
+
4. Compare complete rule sets, profiles, hints, paths, diagnostics, and generated manifests with the
|
|
139
|
+
gold answer; summary counts alone are insufficient. Record commands, target SHAs, exit statuses,
|
|
140
|
+
elapsed times, expected/actual results, and every delta in the session report.
|
|
141
|
+
|
|
142
|
+
A session is not complete if any target is missing, dirty, silently skipped, crashes, or has an
|
|
143
|
+
unexplained gold delta. An unavailable GPU, credential, provider, or approved budget marks the
|
|
144
|
+
affected resource gate **blocked**, never passed; complete and report every unaffected gate. Update
|
|
145
|
+
a pinned commit or gold answer only in a dedicated reviewed change based on independent evidence,
|
|
146
|
+
never by copying ReproLLM's current output.
|