quantum-reasoning-skill 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. quantum_reasoning_skill-1.0.0/.github/ISSUE_TEMPLATE/benchmark-result.yml +81 -0
  2. quantum_reasoning_skill-1.0.0/.github/ISSUE_TEMPLATE/bug-report.yml +38 -0
  3. quantum_reasoning_skill-1.0.0/.github/ISSUE_TEMPLATE/feature-request.yml +31 -0
  4. quantum_reasoning_skill-1.0.0/.github/PULL_REQUEST_TEMPLATE.md +15 -0
  5. quantum_reasoning_skill-1.0.0/.github/workflows/npm-publish.yml +32 -0
  6. quantum_reasoning_skill-1.0.0/.github/workflows/publish-release.yml +79 -0
  7. quantum_reasoning_skill-1.0.0/.github/workflows/validate-skill.yml +154 -0
  8. quantum_reasoning_skill-1.0.0/.gitignore +8 -0
  9. quantum_reasoning_skill-1.0.0/CHANGELOG.md +51 -0
  10. quantum_reasoning_skill-1.0.0/CITATION.cff +17 -0
  11. quantum_reasoning_skill-1.0.0/CONTRIBUTING.md +88 -0
  12. quantum_reasoning_skill-1.0.0/LICENSE +21 -0
  13. quantum_reasoning_skill-1.0.0/PKG-INFO +151 -0
  14. quantum_reasoning_skill-1.0.0/README.md +133 -0
  15. quantum_reasoning_skill-1.0.0/SECURITY.md +17 -0
  16. quantum_reasoning_skill-1.0.0/SKILL.md +146 -0
  17. quantum_reasoning_skill-1.0.0/VERSION +1 -0
  18. quantum_reasoning_skill-1.0.0/benchmark/README.md +134 -0
  19. quantum_reasoning_skill-1.0.0/benchmark/cases.jsonl +6 -0
  20. quantum_reasoning_skill-1.0.0/benchmark/evaluate.py +257 -0
  21. quantum_reasoning_skill-1.0.0/benchmark/schemas/case.schema.json +20 -0
  22. quantum_reasoning_skill-1.0.0/benchmark/schemas/comparison.schema.json +64 -0
  23. quantum_reasoning_skill-1.0.0/benchmark/schemas/metadata.schema.json +53 -0
  24. quantum_reasoning_skill-1.0.0/benchmark/schemas/result.schema.json +25 -0
  25. quantum_reasoning_skill-1.0.0/benchmark/validate_submission.py +114 -0
  26. quantum_reasoning_skill-1.0.0/docs/COMPATIBILITY.md +44 -0
  27. quantum_reasoning_skill-1.0.0/docs/MEASUREMENT.md +131 -0
  28. quantum_reasoning_skill-1.0.0/examples/host-integration.md +57 -0
  29. quantum_reasoning_skill-1.0.0/examples/usage.md +31 -0
  30. quantum_reasoning_skill-1.0.0/index.js +1 -0
  31. quantum_reasoning_skill-1.0.0/package.json +1 -0
  32. quantum_reasoning_skill-1.0.0/pyproject.toml +25 -0
  33. quantum_reasoning_skill-1.0.0/quantum_reasoning_skill/__init__.py +1 -0
  34. quantum_reasoning_skill-1.0.0/reference/branch_controller.py +256 -0
  35. quantum_reasoning_skill-1.0.0/tests/test_benchmark.py +120 -0
  36. quantum_reasoning_skill-1.0.0/tests/test_reference.py +128 -0
  37. quantum_reasoning_skill-1.0.0/tests/test_submission_validation.py +78 -0
@@ -0,0 +1,81 @@
1
+ name: Benchmark result
2
+ description: Submit an independently run baseline-vs-skill model evaluation
3
+ title: "[Benchmark] "
4
+ labels: []
5
+ body:
6
+ - type: markdown
7
+ attributes:
8
+ value: |
9
+ Submit reproducible measurements only. Do not include private chain-of-thought. For PR submissions, validate the full bundle with `python benchmark/validate_submission.py --root benchmark/results/community`.
10
+
11
+ - type: input
12
+ id: provider
13
+ attributes:
14
+ label: Provider
15
+ placeholder: Provider or local runtime
16
+ validations:
17
+ required: true
18
+
19
+ - type: input
20
+ id: model
21
+ attributes:
22
+ label: Exact model / version
23
+ placeholder: Exact model identifier and version if available
24
+ validations:
25
+ required: true
26
+
27
+ - type: input
28
+ id: skill_version
29
+ attributes:
30
+ label: Skill version / commit
31
+ placeholder: v0.3.1 or commit SHA
32
+ validations:
33
+ required: true
34
+
35
+ - type: input
36
+ id: run_date
37
+ attributes:
38
+ label: Run date
39
+ placeholder: YYYY-MM-DD
40
+ validations:
41
+ required: true
42
+
43
+ - type: textarea
44
+ id: controls
45
+ attributes:
46
+ label: Experimental controls
47
+ description: Describe sampling settings, context/output limits, tools, repetitions, seeds, prompt wrapper and hardware/runtime when relevant.
48
+ validations:
49
+ required: true
50
+
51
+ - type: textarea
52
+ id: artifacts
53
+ attributes:
54
+ label: Result artifacts
55
+ description: Link or attach metadata.json, cases.jsonl, baseline.jsonl, skill.jsonl, comparison.json and README.md. State whether the bundle passed validate_submission.py.
56
+ validations:
57
+ required: true
58
+
59
+ - type: textarea
60
+ id: summary
61
+ attributes:
62
+ label: Result summary
63
+ description: Report accuracy and compute/cost trade-offs. Include failures, refusals and timeouts.
64
+ validations:
65
+ required: true
66
+
67
+ - type: checkboxes
68
+ id: integrity
69
+ attributes:
70
+ label: Integrity checks
71
+ options:
72
+ - label: Baseline and skill used the same model/version, task set, tools, sampling settings and budget policy.
73
+ required: true
74
+ - label: I included the exact evaluated cases and did not remove failed runs, refusals or timeouts.
75
+ required: true
76
+ - label: comparison.json was generated from the submitted raw files and was not edited to improve the result.
77
+ required: true
78
+ - label: I did not include private chain-of-thought or fabricate unavailable telemetry.
79
+ required: true
80
+ - label: I disclosed modifications to SKILL.md, benchmark cases, wrappers or provider adapters.
81
+ required: true
@@ -0,0 +1,38 @@
1
+ name: Bug report
2
+ description: Report a reproducible defect in the skill, reference controller, benchmark tooling, or CI.
3
+ title: "[Bug]: "
4
+ labels: ["bug"]
5
+ body:
6
+ - type: textarea
7
+ id: summary
8
+ attributes:
9
+ label: What happened?
10
+ description: Describe the observed behavior and what you expected instead.
11
+ validations:
12
+ required: true
13
+ - type: textarea
14
+ id: reproduce
15
+ attributes:
16
+ label: Minimal reproduction
17
+ description: Include commands, inputs, or a small example that reproduces the problem.
18
+ validations:
19
+ required: true
20
+ - type: input
21
+ id: version
22
+ attributes:
23
+ label: Repository version or commit
24
+ placeholder: v0.3.0 or commit SHA
25
+ validations:
26
+ required: true
27
+ - type: textarea
28
+ id: environment
29
+ attributes:
30
+ label: Environment
31
+ description: Python version, host/agent integration, OS, and relevant tool versions.
32
+ - type: checkboxes
33
+ id: sensitive
34
+ attributes:
35
+ label: Safety check
36
+ options:
37
+ - label: I have not included secrets, credentials, private chain-of-thought, or sensitive benchmark data.
38
+ required: true
@@ -0,0 +1,31 @@
1
+ name: Feature request
2
+ description: Propose a change to the reasoning protocol, tooling, benchmarks, or integration contract.
3
+ title: "[Feature]: "
4
+ labels: ["enhancement"]
5
+ body:
6
+ - type: textarea
7
+ id: problem
8
+ attributes:
9
+ label: Problem or limitation
10
+ description: What concrete limitation does this proposal address?
11
+ validations:
12
+ required: true
13
+ - type: textarea
14
+ id: proposal
15
+ attributes:
16
+ label: Proposed change
17
+ description: Describe the behavior or interface you want to add or change.
18
+ validations:
19
+ required: true
20
+ - type: textarea
21
+ id: evidence
22
+ attributes:
23
+ label: Evidence or evaluation plan
24
+ description: If this affects reasoning quality, explain how it could be tested without relying on anecdotal examples alone.
25
+ - type: checkboxes
26
+ id: scope
27
+ attributes:
28
+ label: Scope check
29
+ options:
30
+ - label: This proposal does not present unverified benchmark claims as established results.
31
+ required: true
@@ -0,0 +1,15 @@
1
+ ## Summary
2
+
3
+ Describe the change and the behavior it affects.
4
+
5
+ ## Validation
6
+
7
+ - [ ] I ran `python -m unittest discover -s tests -v`.
8
+ - [ ] I ran `python benchmark/evaluate.py --help`.
9
+ - [ ] I added or updated tests when executable behavior changed.
10
+ - [ ] I updated documentation when the public contract changed.
11
+ - [ ] I did not include private chain-of-thought, secrets, credentials, or fabricated benchmark telemetry.
12
+
13
+ ## Benchmark-impacting changes
14
+
15
+ If this changes `SKILL.md`, scoring, collapse/revival behavior, benchmark schemas, or evaluation logic, explain how existing results should be interpreted and whether new community runs are needed.
@@ -0,0 +1,32 @@
1
+ name: npm Publish
2
+
3
+ on:
4
+ workflow_dispatch:
5
+ push:
6
+ branches: [main]
7
+ paths:
8
+ - "package.json"
9
+ - "index.js"
10
+
11
+ concurrency:
12
+ group: npm-publish
13
+ cancel-in-progress: false
14
+
15
+ permissions:
16
+ contents: read
17
+
18
+ jobs:
19
+ publish:
20
+ runs-on: ubuntu-latest
21
+ steps:
22
+ - uses: actions/checkout@v4
23
+
24
+ - uses: actions/setup-node@v4
25
+ with:
26
+ node-version: "20"
27
+ registry-url: "https://registry.npmjs.org"
28
+
29
+ - name: Publish to npm
30
+ run: npm publish --access public
31
+ env:
32
+ NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
@@ -0,0 +1,79 @@
1
+ name: publish-release
2
+
3
+ on:
4
+ workflow_run:
5
+ workflows: ["validate-skill"]
6
+ types: [completed]
7
+ branches: [main]
8
+
9
+ permissions:
10
+ contents: write
11
+
12
+ concurrency:
13
+ group: publish-release
14
+ cancel-in-progress: false
15
+
16
+ jobs:
17
+ publish:
18
+ if: ${{ github.event.workflow_run.conclusion == 'success' }}
19
+ runs-on: ubuntu-latest
20
+ steps:
21
+ - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
22
+ with:
23
+ ref: ${{ github.event.workflow_run.head_sha }}
24
+ fetch-depth: 0
25
+
26
+ - name: Resolve and validate version
27
+ id: version
28
+ shell: bash
29
+ run: |
30
+ set -euo pipefail
31
+ test -s VERSION
32
+ version="$(tr -d '[:space:]' < VERSION)"
33
+ version="${version#v}"
34
+ if [[ ! "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+([.-][0-9A-Za-z.-]+)?$ ]]; then
35
+ echo "Invalid release version: $version" >&2
36
+ exit 1
37
+ fi
38
+ echo "version=$version" >> "$GITHUB_OUTPUT"
39
+ echo "tag=v${version}" >> "$GITHUB_OUTPUT"
40
+
41
+ - name: Require matching changelog entry
42
+ shell: bash
43
+ env:
44
+ VERSION: ${{ steps.version.outputs.version }}
45
+ run: |
46
+ set -euo pipefail
47
+ grep -Fq "## [$VERSION]" CHANGELOG.md || {
48
+ echo "CHANGELOG.md must contain a ## [$VERSION] section before release." >&2
49
+ exit 1
50
+ }
51
+
52
+ - name: Check existing release
53
+ id: existing
54
+ shell: bash
55
+ env:
56
+ GH_TOKEN: ${{ github.token }}
57
+ TAG: ${{ steps.version.outputs.tag }}
58
+ run: |
59
+ set -euo pipefail
60
+ if gh release view "$TAG" >/dev/null 2>&1; then
61
+ echo "exists=true" >> "$GITHUB_OUTPUT"
62
+ echo "$TAG already exists; nothing to publish."
63
+ else
64
+ echo "exists=false" >> "$GITHUB_OUTPUT"
65
+ fi
66
+
67
+ - name: Create tag and GitHub Release
68
+ if: steps.existing.outputs.exists != 'true'
69
+ shell: bash
70
+ env:
71
+ GH_TOKEN: ${{ github.token }}
72
+ TAG: ${{ steps.version.outputs.tag }}
73
+ TARGET_SHA: ${{ github.event.workflow_run.head_sha }}
74
+ run: |
75
+ set -euo pipefail
76
+ gh release create "$TAG" \
77
+ --target "$TARGET_SHA" \
78
+ --title "$TAG" \
79
+ --generate-notes
@@ -0,0 +1,154 @@
1
+ name: validate-skill
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+
7
+ permissions:
8
+ contents: read
9
+
10
+ jobs:
11
+ validate:
12
+ runs-on: ubuntu-latest
13
+ steps:
14
+ - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
15
+
16
+ - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5
17
+ with:
18
+ python-version: "3.12"
19
+
20
+ - name: Validate skill repository contract
21
+ shell: bash
22
+ run: |
23
+ set -euo pipefail
24
+ for required in \
25
+ SKILL.md README.md LICENSE CHANGELOG.md VERSION CONTRIBUTING.md SECURITY.md CITATION.cff \
26
+ examples/usage.md examples/host-integration.md \
27
+ reference/branch_controller.py \
28
+ docs/MEASUREMENT.md docs/COMPATIBILITY.md \
29
+ benchmark/README.md benchmark/cases.jsonl benchmark/evaluate.py benchmark/validate_submission.py \
30
+ benchmark/schemas/case.schema.json benchmark/schemas/result.schema.json \
31
+ benchmark/schemas/metadata.schema.json benchmark/schemas/comparison.schema.json \
32
+ tests/test_reference.py tests/test_benchmark.py tests/test_submission_validation.py \
33
+ .github/ISSUE_TEMPLATE/benchmark-result.yml .github/ISSUE_TEMPLATE/bug-report.yml \
34
+ .github/ISSUE_TEMPLATE/feature-request.yml .github/PULL_REQUEST_TEMPLATE.md; do
35
+ test -s "$required"
36
+ done
37
+
38
+ python - <<'PY'
39
+ from pathlib import Path
40
+ import json
41
+ import re
42
+
43
+ root = Path.cwd().resolve()
44
+ skill = (root / "SKILL.md").read_text(encoding="utf-8")
45
+ lines = skill.splitlines()
46
+ if not lines or lines[0].strip() != "---":
47
+ raise SystemExit("SKILL.md must start with YAML front matter")
48
+ try:
49
+ end = next(i for i, line in enumerate(lines[1:], start=1) if line.strip() == "---")
50
+ except StopIteration as exc:
51
+ raise SystemExit("SKILL.md front matter is not closed") from exc
52
+ fields = {}
53
+ for line in lines[1:end]:
54
+ if ":" in line:
55
+ key, value = line.split(":", 1)
56
+ fields[key.strip()] = value.strip()
57
+ if fields.get("name") != "quantum-reasoning":
58
+ raise SystemExit("SKILL.md front matter must declare name: quantum-reasoning")
59
+ if not fields.get("description"):
60
+ raise SystemExit("SKILL.md front matter must contain a non-empty description")
61
+ if len(lines[end + 1:]) < 20:
62
+ raise SystemExit("SKILL.md body is unexpectedly short")
63
+
64
+ readme = (root / "README.md").read_text(encoding="utf-8")
65
+ for required in (
66
+ "SKILL.md", "examples/usage.md", "examples/host-integration.md",
67
+ "reference/branch_controller.py", "docs/MEASUREMENT.md", "docs/COMPATIBILITY.md",
68
+ "benchmark/README.md", "CONTRIBUTING.md", "SECURITY.md", "CITATION.cff",
69
+ ):
70
+ if required not in readme:
71
+ raise SystemExit(f"README.md must reference {required}")
72
+
73
+ markdown_files = [
74
+ root / "README.md", root / "SKILL.md", root / "CONTRIBUTING.md", root / "SECURITY.md",
75
+ root / "examples/usage.md", root / "examples/host-integration.md",
76
+ root / "benchmark/README.md", root / "docs/MEASUREMENT.md", root / "docs/COMPATIBILITY.md",
77
+ ]
78
+ link_pattern = re.compile(r"\[[^\]]*\]\(([^)]+)\)")
79
+ for document in markdown_files:
80
+ text = document.read_text(encoding="utf-8")
81
+ for raw_target in link_pattern.findall(text):
82
+ target = raw_target.strip().split("#", 1)[0].split("?", 1)[0]
83
+ if not target or "://" in target or target.startswith(("mailto:", "#")):
84
+ continue
85
+ resolved = (document.parent / target).resolve()
86
+ try:
87
+ resolved.relative_to(root)
88
+ except ValueError as exc:
89
+ raise SystemExit(f"{document.relative_to(root)} links outside repository: {raw_target}") from exc
90
+ if not resolved.exists():
91
+ raise SystemExit(f"Broken relative link in {document.relative_to(root)}: {raw_target}")
92
+
93
+ case_path = root / "benchmark" / "cases.jsonl"
94
+ seen_ids = set()
95
+ for line_number, raw in enumerate(case_path.read_text(encoding="utf-8").splitlines(), start=1):
96
+ if not raw.strip():
97
+ continue
98
+ try:
99
+ case = json.loads(raw)
100
+ except json.JSONDecodeError as exc:
101
+ raise SystemExit(f"benchmark/cases.jsonl:{line_number}: invalid JSON: {exc}") from exc
102
+ required = {"id", "domain", "prompt", "accepted_answers"}
103
+ missing = required - set(case)
104
+ if missing:
105
+ raise SystemExit(f"benchmark/cases.jsonl:{line_number}: missing {sorted(missing)}")
106
+ case_id = str(case["id"])
107
+ if case_id in seen_ids:
108
+ raise SystemExit(f"duplicate benchmark case id: {case_id}")
109
+ seen_ids.add(case_id)
110
+ if not isinstance(case["accepted_answers"], list) or not case["accepted_answers"]:
111
+ raise SystemExit(f"benchmark/cases.jsonl:{line_number}: accepted_answers must be non-empty")
112
+
113
+ schemas = sorted((root / "benchmark" / "schemas").glob("*.schema.json"))
114
+ if len(schemas) < 4:
115
+ raise SystemExit("expected case, result, metadata and comparison JSON Schemas")
116
+ for schema_path in schemas:
117
+ schema = json.loads(schema_path.read_text(encoding="utf-8"))
118
+ if schema.get("$schema") != "https://json-schema.org/draft/2020-12/schema":
119
+ raise SystemExit(f"{schema_path}: expected JSON Schema draft 2020-12")
120
+ if schema.get("type") != "object" or not schema.get("required"):
121
+ raise SystemExit(f"{schema_path}: must define an object with required fields")
122
+
123
+ version = (root / "VERSION").read_text(encoding="utf-8").strip().lstrip("v")
124
+ if not re.fullmatch(r"[0-9]+\.[0-9]+\.[0-9]+(?:[.-][0-9A-Za-z.-]+)?", version):
125
+ raise SystemExit("VERSION is not a valid semantic version")
126
+ cff = (root / "CITATION.cff").read_text(encoding="utf-8")
127
+ for marker in ("cff-version: 1.2.0", "title:", "authors:", "repository-code:", "license: MIT"):
128
+ if marker not in cff:
129
+ raise SystemExit(f"CITATION.cff missing {marker!r}")
130
+ if f"version: {version}" not in cff:
131
+ raise SystemExit("CITATION.cff version must match VERSION")
132
+
133
+ action_use = re.compile(r"uses:\s+(actions/[A-Za-z0-9_.-]+)@([^\s#]+)")
134
+ for workflow in (root / ".github" / "workflows").glob("*.yml"):
135
+ for action, ref in action_use.findall(workflow.read_text(encoding="utf-8")):
136
+ if not re.fullmatch(r"[0-9a-f]{40}", ref):
137
+ raise SystemExit(f"{workflow}: {action} must be pinned to an exact commit SHA")
138
+
139
+ print("Skill repository contract validated successfully.")
140
+ PY
141
+
142
+ - name: Compile Python sources
143
+ run: python -m compileall -q reference benchmark tests
144
+
145
+ - name: Run behavioral tests
146
+ run: python -m unittest discover -s tests -v
147
+
148
+ - name: Validate community benchmark bundles
149
+ run: python benchmark/validate_submission.py --root benchmark/results/community --allow-empty
150
+
151
+ - name: Check benchmark CLIs
152
+ run: |
153
+ python benchmark/evaluate.py --help > /dev/null
154
+ python benchmark/validate_submission.py --help > /dev/null
@@ -0,0 +1,8 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .pytest_cache/
5
+ .venv/
6
+ venv/
7
+ dist/
8
+ build/
@@ -0,0 +1,51 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here.
4
+
5
+ ## [Unreleased]
6
+
7
+ ## [0.3.1] - 2026-09-07
8
+
9
+ ### Added
10
+
11
+ - Community benchmark contribution policy in `CONTRIBUTING.md`.
12
+ - GitHub **Benchmark result**, **Bug report**, and **Feature request** issue forms.
13
+ - Pull-request checklist for tests, documentation, benchmark integrity, and sensitive-data hygiene.
14
+ - Draft 2020-12 JSON Schemas for benchmark cases, result rows, experiment metadata, and comparison output.
15
+ - `benchmark/validate_submission.py` for machine-validating reproducible community benchmark bundles.
16
+ - Tests that reject missing benchmark artifacts, tampered comparisons, and incomplete integrity metadata.
17
+ - Capability-based host compatibility contract in `docs/COMPATIBILITY.md`.
18
+ - Generic native-skill, persistent-instruction, and controller-assisted integration examples.
19
+ - `SECURITY.md` vulnerability-reporting policy.
20
+ - `CITATION.cff` research-software citation metadata.
21
+
22
+ ### Changed
23
+
24
+ - Benchmark documentation now explicitly assigns real-model evaluation to users and contributors rather than requiring maintainer-run access to every model/provider.
25
+ - Community benchmark PRs now include the exact `cases.jsonl` used and are re-evaluated by CI from their raw baseline/skill files.
26
+ - Third-party benchmark submissions are treated as reproducible external measurements, not automatic project endorsements or universal performance claims.
27
+ - Release publishing now runs only after `validate-skill` succeeds on `main`; a failed validation cannot publish a release.
28
+ - GitHub Actions dependencies are pinned to exact commit SHAs and CI enforces future pinning.
29
+ - CI now validates schemas, version/citation consistency, host/documentation links, and community result bundles.
30
+
31
+ ## [0.3.0] - 2026-09-07
32
+
33
+ ### Added
34
+
35
+ - Deterministic reference branch controller with explicit scoring and state transitions.
36
+ - Shared-assumption and semantic-correlation penalty.
37
+ - Quantified dormant/rejected state rules and branch revival triggers.
38
+ - Uncertainty-based recommended search width.
39
+ - Explicit collapse criteria and leader-margin checks.
40
+ - Measurement specification documenting formulas and calibration requirements.
41
+ - Reproducible baseline-vs-skill benchmark protocol.
42
+ - Seed benchmark cases and JSONL result schema.
43
+ - Benchmark evaluator for accuracy, token/tool cost, latency, branch diversity, error recovery and contradiction resolution.
44
+ - Behavioral unit tests for the branch controller and benchmark evaluator.
45
+ - CI checks for Python compilation, benchmark schema validation and behavioral tests.
46
+ - Automated GitHub tag and Release publishing driven by the `VERSION` file.
47
+
48
+ ### Changed
49
+
50
+ - `SKILL.md` now distinguishes qualitative protocol rules from unvalidated reference thresholds.
51
+ - `README.md` now documents the measurable framework and makes clear that empirical performance improvement has not yet been demonstrated.
@@ -0,0 +1,17 @@
1
+ cff-version: 1.2.0
2
+ message: "If you use this software or reasoning protocol, please cite it using the metadata below."
3
+ title: "Quantum Reasoning Skill"
4
+ type: software
5
+ authors:
6
+ - name: "Furox-Art"
7
+ repository-code: "https://github.com/Furox-Art/quantum-reasoning-skill"
8
+ url: "https://github.com/Furox-Art/quantum-reasoning-skill"
9
+ license: MIT
10
+ version: 0.3.1
11
+ date-released: 2026-09-07
12
+ keywords:
13
+ - reasoning
14
+ - language-models
15
+ - agent-systems
16
+ - test-time-compute
17
+ - benchmarking
@@ -0,0 +1,88 @@
1
+ # Contributing
2
+
3
+ Contributions are welcome for the reasoning protocol, reference controller, tests, documentation, benchmark cases, host integrations and independently run model evaluations.
4
+
5
+ Before changing host-facing behavior, read [`docs/COMPATIBILITY.md`](./docs/COMPATIBILITY.md). Executable behavior changes should include tests.
6
+
7
+ ## Community-run benchmarks
8
+
9
+ The project does **not** require the maintainer to purchase access to every model or provider. Real model evaluations are intentionally community-run.
10
+
11
+ The repository provides:
12
+
13
+ - the benchmark protocol;
14
+ - deterministic seed cases;
15
+ - machine-readable JSON Schemas;
16
+ - the result evaluator;
17
+ - a reproducible submission validator;
18
+ - CI validation and comparison tooling.
19
+
20
+ Users may run controlled baseline-vs-skill experiments on models and providers they already have access to and submit the results for review.
21
+
22
+ ### Required metadata
23
+
24
+ Every submitted model evaluation must identify at least:
25
+
26
+ - provider;
27
+ - exact model and model version when available;
28
+ - run date;
29
+ - skill version/commit when available;
30
+ - baseline and skill prompt/instruction configuration;
31
+ - temperature/sampling parameters;
32
+ - context and output limits;
33
+ - tool availability;
34
+ - number of repetitions per case;
35
+ - random seed when supported;
36
+ - local hardware/runtime details for local models;
37
+ - whether failures, refusals and timeouts were retained.
38
+
39
+ Do not omit failed runs.
40
+
41
+ ### Required artifacts for a benchmark PR
42
+
43
+ Place each submitted run under:
44
+
45
+ ```text
46
+ benchmark/results/community/<provider>-<model>-<YYYY-MM-DD>/
47
+ ```
48
+
49
+ Every bundle must include:
50
+
51
+ ```text
52
+ metadata.json
53
+ cases.jsonl
54
+ baseline.jsonl
55
+ skill.jsonl
56
+ comparison.json
57
+ README.md
58
+ ```
59
+
60
+ The exact contracts are in [`benchmark/schemas/`](./benchmark/schemas/). `cases.jsonl` must contain the exact evaluated cases, including custom cases when used.
61
+
62
+ `comparison.json` must be produced by `benchmark/evaluate.py` from the submitted raw result files rather than edited manually. Before submitting, run:
63
+
64
+ ```bash
65
+ python benchmark/validate_submission.py --root benchmark/results/community
66
+ ```
67
+
68
+ CI recomputes the comparison and rejects a bundle when its raw evidence and reported comparison disagree.
69
+
70
+ ### Result integrity
71
+
72
+ - Use the same model/version, task set, tools, sampling settings and budget policy for baseline and skill conditions.
73
+ - Do not expose or submit private chain-of-thought.
74
+ - Do not fabricate missing telemetry.
75
+ - Clearly disclose custom benchmark cases, prompt wrappers, provider-specific adapters and any modifications to `SKILL.md`.
76
+ - Third-party benchmark submissions are measurements from their submitters, not project-maintainer endorsements.
77
+ - A single positive run is not sufficient to establish a general performance claim.
78
+
79
+ ### How to submit
80
+
81
+ You can either:
82
+
83
+ 1. open a **Benchmark result** issue using the repository issue template and link to your artifacts; or
84
+ 2. open a pull request containing the reproducible result bundle described above.
85
+
86
+ For ordinary defects, use the **Bug report** form. For proposed behavior or protocol changes, use the **Feature request** form.
87
+
88
+ For code or protocol changes, include tests when the change is executable and explain which behavior is being changed. Follow the pull-request checklist in `.github/PULL_REQUEST_TEMPLATE.md`.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Furox
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.