xrefkit 0.4.7__tar.gz → 0.4.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {xrefkit-0.4.7 → xrefkit-0.4.8}/PKG-INFO +17 -1
- {xrefkit-0.4.7 → xrefkit-0.4.8}/README.md +16 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/pyproject.toml +1 -1
- xrefkit-0.4.8/tests/test_brownfield_file_editing_protocol.py +55 -0
- xrefkit-0.4.8/tests/test_human_evaluation.py +93 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_skills_sync.py +6 -6
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/__init__.py +1 -1
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/cli.py +5 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/__init__.py +3 -0
- xrefkit-0.4.8/xrefkit/models/human_evaluation.py +93 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/operations_cli.py +36 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/skillrun.py +121 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/skills_sync.py +61 -16
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit.egg-info/PKG-INFO +17 -1
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit.egg-info/SOURCES.txt +3 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/LICENSE +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/setup.cfg +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_base_sync_ownership.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_boundary_analysis.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_calibration_lint.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_check_skill_knowledge_xids.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_cli.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_collect_analyzer_sarif.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_convert_to_xrefkit_skill.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_cs_scope_probe.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_csharp_commonality.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_csharp_naming_profile.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_ctx.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_cutover_readiness.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_dashboard.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_error_policy_audit.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_error_policy_locator.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_fm_multiroot.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_gate.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_goal_desired_state.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_instruction_workflow.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_knowledge_relations_validator.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_mcp_setup.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_ownership.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_packmeta.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_project_quality_baseline.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_resource_provider.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_runtime_contracts.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_sarif_to_locator.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_skill_runtime_audit.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_skillmeta.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_structure_catalog.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_xref.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_xrefkit_instance.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_xrefkit_tools.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_xrefkit_v2_discovery.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_xrefkit_v2_models.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/tests/test_xrefkit_v2_pipeline.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/__main__.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/boundary_analysis.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/catalog_cli.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/contracts.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/ctx.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/dashboard.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/discovery.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/gate.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/goalstate.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/hashing.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/import_skill.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/instance.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/loaders.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/__init__.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/audit.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/bootstrap.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/catalog.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/cli.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/client_cache.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/context_registry.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/context_token.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/contracts.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/dist.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/ownership.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/repository.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/schemas.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/server.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/setup.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp/startup_contract_pack.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/mcp_tools.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/common.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/effective_bundle.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/local_manifest.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/package_manifest.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/run_log.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/server_config.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/models/skill_definition.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/ownership.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/packmeta.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/registry.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resolver.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resource_provider.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/contracts.json +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/current.json +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/7a682a5272907354/contracts.json +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/7a682a5272907354/model_body.md +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/9929294385ccb7b0/contracts.json +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/9929294385ccb7b0/model_body.md +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/model_body.md +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/runlog.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/skillmeta.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/structure_catalog.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/tools/__init__.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/tools/__main__.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/v2_cli.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/workspace.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/xref.py +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit.egg-info/dependency_links.txt +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit.egg-info/entry_points.txt +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit.egg-info/requires.txt +0 -0
- {xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: xrefkit
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.8
|
|
4
4
|
Summary: Portable XID, Skill, Knowledge, workflow, and MCP runtime for XRefKit
|
|
5
5
|
Author: synthaicode
|
|
6
6
|
License: MIT License
|
|
@@ -178,6 +178,22 @@ If the `xrefkit` command is not available on `PATH`, use the module form:
|
|
|
178
178
|
python -m xrefkit --help
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
+
### Markdown WBS validation and rollups
|
|
182
|
+
|
|
183
|
+
Keep one task per row in a Markdown table, then validate and regenerate the
|
|
184
|
+
feature, phase, and feature-by-phase views without editing the source table:
|
|
185
|
+
|
|
186
|
+
```powershell
|
|
187
|
+
python -m xrefkit wbs validate samples/wbs.md
|
|
188
|
+
python -m xrefkit wbs rollup samples/wbs.md
|
|
189
|
+
python -m xrefkit wbs csv samples/wbs.md .tmp/wbs.csv
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
The validator rejects missing or extra cells, duplicate IDs, non-numeric or
|
|
193
|
+
negative effort, and statuses outside `todo`, `doing`, and `done`. CSV output
|
|
194
|
+
uses UTF-8 with BOM for Excel compatibility; the Markdown WBS remains the
|
|
195
|
+
source of truth.
|
|
196
|
+
|
|
181
197
|
To use the integrated MCP server, install the optional MCP dependencies:
|
|
182
198
|
|
|
183
199
|
```powershell
|
|
@@ -139,6 +139,22 @@ If the `xrefkit` command is not available on `PATH`, use the module form:
|
|
|
139
139
|
python -m xrefkit --help
|
|
140
140
|
```
|
|
141
141
|
|
|
142
|
+
### Markdown WBS validation and rollups
|
|
143
|
+
|
|
144
|
+
Keep one task per row in a Markdown table, then validate and regenerate the
|
|
145
|
+
feature, phase, and feature-by-phase views without editing the source table:
|
|
146
|
+
|
|
147
|
+
```powershell
|
|
148
|
+
python -m xrefkit wbs validate samples/wbs.md
|
|
149
|
+
python -m xrefkit wbs rollup samples/wbs.md
|
|
150
|
+
python -m xrefkit wbs csv samples/wbs.md .tmp/wbs.csv
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
The validator rejects missing or extra cells, duplicate IDs, non-numeric or
|
|
154
|
+
negative effort, and statuses outside `todo`, `doing`, and `done`. CSV output
|
|
155
|
+
uses UTF-8 with BOM for Excel compatibility; the Markdown WBS remains the
|
|
156
|
+
source of truth.
|
|
157
|
+
|
|
142
158
|
To use the integrated MCP server, install the optional MCP dependencies:
|
|
143
159
|
|
|
144
160
|
```powershell
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_brownfield_file_editing_protocol_is_in_canonical_and_packaged_skill() -> None:
|
|
8
|
+
canonical = (ROOT / "skills" / "brownfield-workflow" / "SKILL.md").read_text(encoding="utf-8")
|
|
9
|
+
packaged = (
|
|
10
|
+
ROOT / "packages" / "xrefkit-skills-brownfield" / "src" / "xrefkit_skills_brownfield"
|
|
11
|
+
/ "skills" / "brownfield_workflow" / "entry.md"
|
|
12
|
+
).read_text(encoding="utf-8")
|
|
13
|
+
|
|
14
|
+
for keyword in (
|
|
15
|
+
"encoding",
|
|
16
|
+
"BOM",
|
|
17
|
+
"newline",
|
|
18
|
+
"Unicode",
|
|
19
|
+
"strict",
|
|
20
|
+
"after_bytes",
|
|
21
|
+
"mojibake",
|
|
22
|
+
"concurrency",
|
|
23
|
+
"revision token",
|
|
24
|
+
"compare-and-swap",
|
|
25
|
+
"abort",
|
|
26
|
+
"atomically",
|
|
27
|
+
"specification",
|
|
28
|
+
"semantic",
|
|
29
|
+
"authoritative",
|
|
30
|
+
"hypothesis",
|
|
31
|
+
"semantic_alignment",
|
|
32
|
+
"Historical conflict investigation",
|
|
33
|
+
"bounded",
|
|
34
|
+
"Git",
|
|
35
|
+
"uncommitted",
|
|
36
|
+
"newest",
|
|
37
|
+
"Uncommitted-file policy",
|
|
38
|
+
"pre_existing_human_or_unknown",
|
|
39
|
+
"ai_owned_current_work",
|
|
40
|
+
"mixed_or_overlapping",
|
|
41
|
+
"untracked",
|
|
42
|
+
"stash",
|
|
43
|
+
"New-file extension conformity",
|
|
44
|
+
"peer",
|
|
45
|
+
"companion files",
|
|
46
|
+
"extension-specific",
|
|
47
|
+
"cluster",
|
|
48
|
+
"majority",
|
|
49
|
+
"same directory",
|
|
50
|
+
"confidence",
|
|
51
|
+
"weak margin",
|
|
52
|
+
"repository-wide fallback",
|
|
53
|
+
):
|
|
54
|
+
assert keyword in canonical
|
|
55
|
+
assert keyword in packaged
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import contextlib
|
|
2
|
+
import io
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from xrefkit.__main__ import main
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _run(root: Path, *args: str) -> int:
|
|
9
|
+
with contextlib.redirect_stdout(io.StringIO()):
|
|
10
|
+
argv = list(args)
|
|
11
|
+
if argv[:2] == ["workflow", "run"]:
|
|
12
|
+
argv.extend(["--root", str(root)])
|
|
13
|
+
return main(argv)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _closed_run(tmp_path: Path) -> Path:
|
|
17
|
+
log = tmp_path / "work" / "sessions" / "preceding.md"
|
|
18
|
+
assert _run(
|
|
19
|
+
tmp_path,
|
|
20
|
+
"workflow", "run", "--task", "Produce a bounded output", "--out", str(log),
|
|
21
|
+
"--use-default-completion-conditions",
|
|
22
|
+
) == 0
|
|
23
|
+
assert _run(
|
|
24
|
+
tmp_path, "skill", "workitem", "--log", str(log), "--item", "WI-001",
|
|
25
|
+
"--text", "Produce output", "--completion-criterion", "output is recorded",
|
|
26
|
+
"--status", "done", "--role", "instruction:executor",
|
|
27
|
+
) == 0
|
|
28
|
+
for artifact_id, kind, target, role in (
|
|
29
|
+
("OUT-001", "output", "output.md", "instruction:executor"),
|
|
30
|
+
("EVD-001", "evidence", "test command passed", "instruction:checker"),
|
|
31
|
+
):
|
|
32
|
+
assert _run(
|
|
33
|
+
tmp_path, "skill", "artifact", "--log", str(log), "--artifact", artifact_id,
|
|
34
|
+
"--kind", kind, "--target", target, "--item", "WI-001", "--status", "done",
|
|
35
|
+
"--role", role,
|
|
36
|
+
) == 0
|
|
37
|
+
assert _run(tmp_path, "skill", "phase", "--log", str(log), "--phase", "execution", "--status", "done", "--role", "instruction:executor") == 0
|
|
38
|
+
assert _run(tmp_path, "skill", "phase", "--log", str(log), "--phase", "handoff", "--status", "done", "--role", "instruction:handoff_owner") == 0
|
|
39
|
+
assert _run(tmp_path, "skill", "verify", "--log", str(log)) == 0
|
|
40
|
+
assert _run(tmp_path, "skill", "close", "--log", str(log)) == 0
|
|
41
|
+
return log
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_human_evaluation_is_optional_and_scoped(tmp_path: Path) -> None:
|
|
45
|
+
log = _closed_run(tmp_path)
|
|
46
|
+
assert _run(
|
|
47
|
+
tmp_path, "skill", "evaluate", "--log", str(log),
|
|
48
|
+
"--decision", "accepted_with_conditions",
|
|
49
|
+
"--classification", "correction",
|
|
50
|
+
"--next-handling", "repair_previous_run",
|
|
51
|
+
"--purpose-fit", "The overall purpose remains valid",
|
|
52
|
+
"--verified", "WI-001 and EVD-001",
|
|
53
|
+
"--uncertainty", "target B source is not snapshotted",
|
|
54
|
+
"--scope-finding", "WI-A|accepted|Target A is acceptable",
|
|
55
|
+
"--scope-finding", "WI-B|correction|Target B needs repair",
|
|
56
|
+
"--scope-link", "WI-B|EVD-B",
|
|
57
|
+
"--context-ref", "criteria:v1",
|
|
58
|
+
"--comparability", "gap",
|
|
59
|
+
"--comparability-gap", "target B source snapshot is unavailable",
|
|
60
|
+
"--evaluated-at", "2026-08-19T01:02:03Z",
|
|
61
|
+
"--proposed-classification", "continuation",
|
|
62
|
+
) == 0
|
|
63
|
+
text = log.read_text(encoding="utf-8")
|
|
64
|
+
assert '"event":"human.evaluation"' in text
|
|
65
|
+
assert '"classification":"correction"' in text
|
|
66
|
+
assert '"preceding_run_id"' in text
|
|
67
|
+
assert '"target":"WI-B"' in text
|
|
68
|
+
assert '"linked_targets":["EVD-B"]' in text
|
|
69
|
+
assert '"comparability":"gap"' in text
|
|
70
|
+
assert '"classification_source":"human_confirmed"' in text
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_human_evaluation_does_not_accept_an_open_run(tmp_path: Path) -> None:
|
|
74
|
+
log = tmp_path / "work" / "sessions" / "open.md"
|
|
75
|
+
assert _run(
|
|
76
|
+
tmp_path, "workflow", "run", "--task", "Open work", "--out", str(log),
|
|
77
|
+
"--use-default-completion-conditions",
|
|
78
|
+
) == 0
|
|
79
|
+
assert _run(
|
|
80
|
+
tmp_path, "skill", "evaluate", "--log", str(log), "--decision", "accepted",
|
|
81
|
+
"--classification", "continuation", "--next-handling", "continue_next_step",
|
|
82
|
+
"--purpose-fit", "still fits", "--verified", "none", "--uncertainty", "none",
|
|
83
|
+
) == 1
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def test_comparability_gap_requires_a_reason(tmp_path: Path) -> None:
|
|
87
|
+
log = _closed_run(tmp_path)
|
|
88
|
+
assert _run(
|
|
89
|
+
tmp_path, "skill", "evaluate", "--log", str(log), "--decision", "accepted",
|
|
90
|
+
"--classification", "continuation", "--next-handling", "continue_next_step",
|
|
91
|
+
"--purpose-fit", "still fits", "--verified", "none", "--uncertainty", "none",
|
|
92
|
+
"--comparability", "gap",
|
|
93
|
+
) == 1
|
|
@@ -21,12 +21,12 @@ def _zip_bytes() -> bytes:
|
|
|
21
21
|
|
|
22
22
|
def test_sync_bundle_extracts_skill_and_knowledge(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None:
|
|
23
23
|
release = {
|
|
24
|
-
"tag_name": "skills-
|
|
24
|
+
"tag_name": "xrefkit-skills-demo-v1.0.0",
|
|
25
25
|
"assets": [{"name": "xrefkit-skills-demo-1.0.0.zip", "browser_download_url": "https://example.test/demo.zip"}],
|
|
26
26
|
}
|
|
27
27
|
|
|
28
|
-
def fake_json(_url: str) ->
|
|
29
|
-
return release
|
|
28
|
+
def fake_json(_url: str) -> object:
|
|
29
|
+
return [release]
|
|
30
30
|
|
|
31
31
|
def fake_download(_asset: dict) -> tuple[str, bytes]:
|
|
32
32
|
return "xrefkit-skills-demo-1.0.0.zip", _zip_bytes()
|
|
@@ -36,7 +36,7 @@ def test_sync_bundle_extracts_skill_and_knowledge(monkeypatch: pytest.MonkeyPatc
|
|
|
36
36
|
|
|
37
37
|
result = sync_bundle(repo=tmp_path, source_repository="owner/repo", bundle="demo")
|
|
38
38
|
|
|
39
|
-
assert result.release == "skills-
|
|
39
|
+
assert result.release == "xrefkit-skills-demo-v1.0.0"
|
|
40
40
|
assert (tmp_path / "skills/demo/SKILL.md").is_file()
|
|
41
41
|
assert (tmp_path / "knowledge/demo.md").is_file()
|
|
42
42
|
state = json.loads((tmp_path / ".xrefkit/skill-sync/demo.json").read_text(encoding="utf-8"))
|
|
@@ -44,8 +44,8 @@ def test_sync_bundle_extracts_skill_and_knowledge(monkeypatch: pytest.MonkeyPatc
|
|
|
44
44
|
|
|
45
45
|
|
|
46
46
|
def test_sync_bundle_refuses_unmanaged_collision(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None:
|
|
47
|
-
release = {"tag_name": "v1", "assets": [{"name": "xrefkit-skills-demo-1.0.0.zip", "browser_download_url": "https://example.test/demo.zip"}]}
|
|
48
|
-
monkeypatch.setattr("xrefkit.skills_sync._github_json", lambda _url: release)
|
|
47
|
+
release = {"tag_name": "xrefkit-skills-demo-v1.0.0", "assets": [{"name": "xrefkit-skills-demo-1.0.0.zip", "browser_download_url": "https://example.test/demo.zip"}]}
|
|
48
|
+
monkeypatch.setattr("xrefkit.skills_sync._github_json", lambda _url: [release])
|
|
49
49
|
monkeypatch.setattr("xrefkit.skills_sync._download_asset", lambda _asset: ("demo.zip", _zip_bytes()))
|
|
50
50
|
target = tmp_path / "skills/demo/SKILL.md"
|
|
51
51
|
target.parent.mkdir(parents=True)
|
|
@@ -16,6 +16,7 @@ def _print_help() -> None:
|
|
|
16
16
|
"commands:\n"
|
|
17
17
|
" init initialize or validate an XRefKit instance\n"
|
|
18
18
|
" xref manage XIDs and references\n"
|
|
19
|
+
" wbs validate and summarize a Markdown WBS\n"
|
|
19
20
|
" ctx build compact context packs\n"
|
|
20
21
|
" skill discover, validate, run, verify, and close Skills\n"
|
|
21
22
|
" workflow run the generic protocol for instructions without a Skill\n"
|
|
@@ -60,6 +61,10 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
60
61
|
from .tools import main as tools_main
|
|
61
62
|
|
|
62
63
|
return tools_main(args[1:])
|
|
64
|
+
if command == "wbs":
|
|
65
|
+
from .wbs import main as wbs_main
|
|
66
|
+
|
|
67
|
+
return wbs_main(args[1:])
|
|
63
68
|
if command == "catalog":
|
|
64
69
|
from .catalog_cli import main as catalog_main
|
|
65
70
|
|
|
@@ -11,6 +11,7 @@ from .common import (
|
|
|
11
11
|
XidRef,
|
|
12
12
|
)
|
|
13
13
|
from .effective_bundle import BundleReferences, DomainKnowledgeCatalogEntry, EffectiveSkillBundle, LoadedTexts
|
|
14
|
+
from .human_evaluation import HumanEvaluation, ScopedEvaluation
|
|
14
15
|
from .local_manifest import IncludeRef, LocalDomainSkill, LocalManifest
|
|
15
16
|
from .package_manifest import PackageManifest
|
|
16
17
|
from .run_log import RunLogAggregate, RunLogEvent
|
|
@@ -38,4 +39,6 @@ __all__ = [
|
|
|
38
39
|
"XidLoadedRef",
|
|
39
40
|
"XidRef",
|
|
40
41
|
"BundleReferences",
|
|
42
|
+
"HumanEvaluation",
|
|
43
|
+
"ScopedEvaluation",
|
|
41
44
|
]
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Schema for optional human evaluation at a completed-run boundary."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
from typing import Literal
|
|
7
|
+
|
|
8
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
EvaluationDecision = Literal[
|
|
12
|
+
"accepted",
|
|
13
|
+
"accepted_with_conditions",
|
|
14
|
+
"correction",
|
|
15
|
+
"rejected_or_returned_to_human",
|
|
16
|
+
"needs_clarification",
|
|
17
|
+
]
|
|
18
|
+
RelationshipClassification = Literal[
|
|
19
|
+
"continuation",
|
|
20
|
+
"correction",
|
|
21
|
+
"scope_change",
|
|
22
|
+
"new_work",
|
|
23
|
+
"needs_clarification",
|
|
24
|
+
]
|
|
25
|
+
NextHandling = Literal["continue_next_step", "repair_previous_run", "human_takeover"]
|
|
26
|
+
Comparability = Literal["comparable", "gap", "not_assessed"]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ScopedEvaluation(BaseModel):
|
|
30
|
+
"""Optional item/artifact-level finding within one preceding run."""
|
|
31
|
+
|
|
32
|
+
model_config = ConfigDict(extra="forbid")
|
|
33
|
+
|
|
34
|
+
target: str = Field(min_length=1)
|
|
35
|
+
decision: EvaluationDecision
|
|
36
|
+
note: str = Field(min_length=1)
|
|
37
|
+
linked_targets: list[str] = Field(default_factory=list)
|
|
38
|
+
|
|
39
|
+
@field_validator("target", "note", mode="before")
|
|
40
|
+
@classmethod
|
|
41
|
+
def _strip_scalar(cls, value: object) -> object:
|
|
42
|
+
return str(value).strip()
|
|
43
|
+
|
|
44
|
+
@field_validator("linked_targets", mode="before")
|
|
45
|
+
@classmethod
|
|
46
|
+
def _strip_links(cls, value: object) -> object:
|
|
47
|
+
if value is None:
|
|
48
|
+
return []
|
|
49
|
+
return [str(item).strip() for item in value if str(item).strip()]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class HumanEvaluation(BaseModel):
|
|
53
|
+
"""Human-confirmed disposition; private model reasoning is out of scope."""
|
|
54
|
+
|
|
55
|
+
model_config = ConfigDict(extra="forbid")
|
|
56
|
+
|
|
57
|
+
preceding_run_id: str = Field(min_length=1)
|
|
58
|
+
decision: EvaluationDecision
|
|
59
|
+
classification: RelationshipClassification
|
|
60
|
+
next_handling: NextHandling
|
|
61
|
+
purpose_fit: str = Field(min_length=1)
|
|
62
|
+
verified_basis: list[str] = Field(min_length=1)
|
|
63
|
+
remaining_uncertainty: list[str] = Field(min_length=1)
|
|
64
|
+
carry_forward: list[str] = Field(default_factory=list)
|
|
65
|
+
linked_targets: list[str] = Field(default_factory=list)
|
|
66
|
+
scoped_findings: list[ScopedEvaluation] = Field(default_factory=list)
|
|
67
|
+
proposed_classification: RelationshipClassification | None = None
|
|
68
|
+
classification_source: Literal["human_confirmed"] = "human_confirmed"
|
|
69
|
+
reviewer: str | None = None
|
|
70
|
+
evaluated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
71
|
+
context_refs: list[str] = Field(default_factory=list)
|
|
72
|
+
comparability: Comparability = "not_assessed"
|
|
73
|
+
comparability_gaps: list[str] = Field(default_factory=list)
|
|
74
|
+
|
|
75
|
+
@model_validator(mode="after")
|
|
76
|
+
def _require_comparability_reason(self) -> "HumanEvaluation":
|
|
77
|
+
if self.comparability == "gap" and not self.comparability_gaps:
|
|
78
|
+
raise ValueError("comparability=gap requires at least one comparability gap")
|
|
79
|
+
return self
|
|
80
|
+
|
|
81
|
+
@field_validator("preceding_run_id", "purpose_fit", "reviewer", mode="before")
|
|
82
|
+
@classmethod
|
|
83
|
+
def _strip_scalar(cls, value: object) -> object:
|
|
84
|
+
return None if value is None else str(value).strip()
|
|
85
|
+
|
|
86
|
+
@field_validator(
|
|
87
|
+
"verified_basis", "remaining_uncertainty", "carry_forward", "linked_targets", mode="before"
|
|
88
|
+
)
|
|
89
|
+
@classmethod
|
|
90
|
+
def _strip_list(cls, value: object) -> object:
|
|
91
|
+
if value is None:
|
|
92
|
+
return []
|
|
93
|
+
return [str(item).strip() for item in value if str(item).strip()]
|
|
@@ -471,6 +471,38 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
471
471
|
p_skill_feedback.add_argument("--note", required=True, help="Observed feedback or outcome")
|
|
472
472
|
p_skill_feedback.add_argument("--json", action="store_true", help="Emit JSON")
|
|
473
473
|
|
|
474
|
+
p_skill_evaluate = skill_sub.add_parser(
|
|
475
|
+
"evaluate",
|
|
476
|
+
help="Optionally record a human-confirmed evaluation and next-request relationship for a closed run",
|
|
477
|
+
)
|
|
478
|
+
p_skill_evaluate.add_argument("--log", required=True, help="Completed preceding run log to update")
|
|
479
|
+
p_skill_evaluate.add_argument("--decision", required=True, choices=[
|
|
480
|
+
"accepted", "accepted_with_conditions", "correction",
|
|
481
|
+
"rejected_or_returned_to_human", "needs_clarification",
|
|
482
|
+
])
|
|
483
|
+
p_skill_evaluate.add_argument("--classification", required=True, choices=[
|
|
484
|
+
"continuation", "correction", "scope_change", "new_work", "needs_clarification",
|
|
485
|
+
], help="Human-confirmed relationship of the subsequent request to the preceding run")
|
|
486
|
+
p_skill_evaluate.add_argument("--next-handling", required=True, choices=[
|
|
487
|
+
"continue_next_step", "repair_previous_run", "human_takeover",
|
|
488
|
+
])
|
|
489
|
+
p_skill_evaluate.add_argument("--purpose-fit", required=True, help="Human-stated fit to the preceding purpose/intent")
|
|
490
|
+
p_skill_evaluate.add_argument("--verified", action="append", default=[], help="Verified artifact/check basis; repeatable, or use 'none'")
|
|
491
|
+
p_skill_evaluate.add_argument("--uncertainty", action="append", default=[], help="Remaining uncertainty/risk; repeatable, or use 'none'")
|
|
492
|
+
p_skill_evaluate.add_argument("--carry-forward", action="append", default=[], help="Constraint or added context for the next run; repeatable")
|
|
493
|
+
p_skill_evaluate.add_argument("--link", action="append", default=[], help="Preceding artifact, check, or evidence link; repeatable")
|
|
494
|
+
p_skill_evaluate.add_argument("--scope-finding", action="append", default=[], help="Scoped finding as TARGET|DECISION|NOTE; repeatable")
|
|
495
|
+
p_skill_evaluate.add_argument("--scope-link", action="append", default=[], help="Scoped evidence link as TARGET|LINK; repeatable")
|
|
496
|
+
p_skill_evaluate.add_argument("--proposed-classification", choices=[
|
|
497
|
+
"continuation", "correction", "scope_change", "new_work", "needs_clarification",
|
|
498
|
+
], help="Optional AI proposal; --classification remains human-confirmed")
|
|
499
|
+
p_skill_evaluate.add_argument("--reviewer", default=None, help="Optional human reviewer identifier")
|
|
500
|
+
p_skill_evaluate.add_argument("--evaluated-at", default=None, help="Optional ISO-8601 evaluation timestamp")
|
|
501
|
+
p_skill_evaluate.add_argument("--context-ref", action="append", default=[], help="Versioned/declarative prior-run context reference; repeatable")
|
|
502
|
+
p_skill_evaluate.add_argument("--comparability", choices=["comparable", "gap", "not_assessed"], default="not_assessed")
|
|
503
|
+
p_skill_evaluate.add_argument("--comparability-gap", action="append", default=[], help="Why prior and current evidence/criteria are not comparable; repeatable")
|
|
504
|
+
p_skill_evaluate.add_argument("--json", action="store_true", help="Emit JSON")
|
|
505
|
+
|
|
474
506
|
p_skill_phase = skill_sub.add_parser("phase", help="Update a Skill run log phase state")
|
|
475
507
|
p_skill_phase.add_argument("--log", required=True, help="Skill run log to update")
|
|
476
508
|
p_skill_phase.add_argument(
|
|
@@ -672,6 +704,10 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
672
704
|
from xrefkit.skillrun import cmd_skill_feedback
|
|
673
705
|
|
|
674
706
|
return cmd_skill_feedback(args)
|
|
707
|
+
if args.skill_cmd == "evaluate":
|
|
708
|
+
from xrefkit.skillrun import cmd_skill_evaluate
|
|
709
|
+
|
|
710
|
+
return cmd_skill_evaluate(args)
|
|
675
711
|
if args.skill_cmd == "phase":
|
|
676
712
|
from xrefkit.skillrun import cmd_skill_phase
|
|
677
713
|
|
|
@@ -28,6 +28,7 @@ from xrefkit.skillmeta import (
|
|
|
28
28
|
resolve_os_contract,
|
|
29
29
|
validate_skill_meta,
|
|
30
30
|
)
|
|
31
|
+
from xrefkit.models.human_evaluation import HumanEvaluation
|
|
31
32
|
|
|
32
33
|
|
|
33
34
|
@dataclass
|
|
@@ -134,6 +135,13 @@ def _locked_log_update(func):
|
|
|
134
135
|
|
|
135
136
|
|
|
136
137
|
VALID_PHASES = {"startup", "planning", "execution", "check", "quality", "closure", "handoff"}
|
|
138
|
+
EVALUATION_DECISIONS = {
|
|
139
|
+
"accepted",
|
|
140
|
+
"accepted_with_conditions",
|
|
141
|
+
"correction",
|
|
142
|
+
"rejected_or_returned_to_human",
|
|
143
|
+
"needs_clarification",
|
|
144
|
+
}
|
|
137
145
|
VALID_PHASE_STATUSES = {"pending", "in_progress", "done", "blocked", "unknown", "escalated"}
|
|
138
146
|
VALID_WORKITEM_STATUSES = {"pending", "in_progress", "done", "blocked", "unknown", "escalated"}
|
|
139
147
|
VALID_ARTIFACT_STATUSES = {"pending", "in_progress", "done", "blocked", "unknown", "escalated"}
|
|
@@ -601,6 +609,11 @@ def _render_log(
|
|
|
601
609
|
- status: `pending`
|
|
602
610
|
- rule: record human acceptance, correction, or rejection with `xrefkit skill feedback --kind human`
|
|
603
611
|
|
|
612
|
+
## Human Evaluation
|
|
613
|
+
|
|
614
|
+
- status: `pending`
|
|
615
|
+
- rule: optional run-boundary evaluation; control returns to the human without waiting, and a subsequent request may record a human-confirmed relationship with `xrefkit skill evaluate`
|
|
616
|
+
|
|
604
617
|
## Outcome Feedback
|
|
605
618
|
|
|
606
619
|
- status: `pending`
|
|
@@ -2253,6 +2266,110 @@ def update_feedback_observation(args) -> SkillRunResult:
|
|
|
2253
2266
|
)
|
|
2254
2267
|
|
|
2255
2268
|
|
|
2269
|
+
@_locked_log_update
|
|
2270
|
+
def update_human_evaluation(args) -> SkillRunResult:
|
|
2271
|
+
"""Record optional human evaluation after control has returned."""
|
|
2272
|
+
log_path = Path(args.log).resolve()
|
|
2273
|
+
text, error = _validate_observation_log(log_path)
|
|
2274
|
+
if error is not None:
|
|
2275
|
+
return error
|
|
2276
|
+
assert text is not None
|
|
2277
|
+
closure_status = _section_status(text, "Closure Gate")
|
|
2278
|
+
if closure_status not in ACCEPTED_CLOSE_STATUSES:
|
|
2279
|
+
return SkillRunResult(
|
|
2280
|
+
ok=False,
|
|
2281
|
+
skill_id=_log_skill_id(text),
|
|
2282
|
+
skill_doc=None,
|
|
2283
|
+
run_log=str(log_path),
|
|
2284
|
+
errors=[
|
|
2285
|
+
"human evaluation requires the preceding run Closure Gate to be done or escalated; "
|
|
2286
|
+
f"current={closure_status or 'missing'}"
|
|
2287
|
+
],
|
|
2288
|
+
run_id=_log_field(text, "run_id"),
|
|
2289
|
+
)
|
|
2290
|
+
|
|
2291
|
+
verified_basis = [str(value).strip() for value in getattr(args, "verified", []) if str(value).strip()]
|
|
2292
|
+
remaining_uncertainty = [
|
|
2293
|
+
str(value).strip() for value in getattr(args, "uncertainty", []) if str(value).strip()
|
|
2294
|
+
]
|
|
2295
|
+
scoped_links: dict[str, list[str]] = {}
|
|
2296
|
+
scope_errors: list[str] = []
|
|
2297
|
+
for raw_link in getattr(args, "scope_link", []):
|
|
2298
|
+
parts = str(raw_link).split("|", 1)
|
|
2299
|
+
if len(parts) != 2 or not parts[0].strip() or not parts[1].strip():
|
|
2300
|
+
scope_errors.append("--scope-link must use TARGET|LINK")
|
|
2301
|
+
continue
|
|
2302
|
+
scoped_links.setdefault(parts[0].strip(), []).append(parts[1].strip())
|
|
2303
|
+
scoped_findings: list[dict[str, object]] = []
|
|
2304
|
+
for raw_finding in getattr(args, "scope_finding", []):
|
|
2305
|
+
parts = str(raw_finding).split("|", 2)
|
|
2306
|
+
if len(parts) != 3 or not all(part.strip() for part in parts):
|
|
2307
|
+
scope_errors.append("--scope-finding must use TARGET|DECISION|NOTE")
|
|
2308
|
+
continue
|
|
2309
|
+
target, decision, note = (part.strip() for part in parts)
|
|
2310
|
+
if decision not in EVALUATION_DECISIONS:
|
|
2311
|
+
scope_errors.append(f"invalid scoped finding decision: {decision}")
|
|
2312
|
+
continue
|
|
2313
|
+
scoped_findings.append({
|
|
2314
|
+
"target": target,
|
|
2315
|
+
"decision": decision,
|
|
2316
|
+
"note": note,
|
|
2317
|
+
"linked_targets": scoped_links.get(target, []),
|
|
2318
|
+
})
|
|
2319
|
+
if scope_errors:
|
|
2320
|
+
return SkillRunResult(
|
|
2321
|
+
ok=False,
|
|
2322
|
+
skill_id=_log_skill_id(text),
|
|
2323
|
+
skill_doc=None,
|
|
2324
|
+
run_log=str(log_path),
|
|
2325
|
+
errors=scope_errors,
|
|
2326
|
+
run_id=_log_field(text, "run_id"),
|
|
2327
|
+
)
|
|
2328
|
+
payload = {
|
|
2329
|
+
"preceding_run_id": _log_field(text, "run_id"),
|
|
2330
|
+
"decision": args.decision,
|
|
2331
|
+
"classification": args.classification,
|
|
2332
|
+
"next_handling": args.next_handling,
|
|
2333
|
+
"purpose_fit": args.purpose_fit,
|
|
2334
|
+
"verified_basis": verified_basis or ["none"],
|
|
2335
|
+
"remaining_uncertainty": remaining_uncertainty or ["none"],
|
|
2336
|
+
"carry_forward": [str(value).strip() for value in getattr(args, "carry_forward", []) if str(value).strip()],
|
|
2337
|
+
"linked_targets": [str(value).strip() for value in getattr(args, "link", []) if str(value).strip()],
|
|
2338
|
+
"scoped_findings": scoped_findings,
|
|
2339
|
+
"proposed_classification": getattr(args, "proposed_classification", None),
|
|
2340
|
+
"classification_source": "human_confirmed",
|
|
2341
|
+
"reviewer": str(getattr(args, "reviewer", None) or "").strip() or None,
|
|
2342
|
+
"evaluated_at": getattr(args, "evaluated_at", None),
|
|
2343
|
+
"context_refs": [str(value).strip() for value in getattr(args, "context_ref", []) if str(value).strip()],
|
|
2344
|
+
"comparability": getattr(args, "comparability", "not_assessed"),
|
|
2345
|
+
"comparability_gaps": [str(value).strip() for value in getattr(args, "comparability_gap", []) if str(value).strip()],
|
|
2346
|
+
}
|
|
2347
|
+
if not payload["evaluated_at"]:
|
|
2348
|
+
payload.pop("evaluated_at")
|
|
2349
|
+
try:
|
|
2350
|
+
evaluation = HumanEvaluation.model_validate(payload)
|
|
2351
|
+
except Exception as exc:
|
|
2352
|
+
return SkillRunResult(
|
|
2353
|
+
ok=False,
|
|
2354
|
+
skill_id=_log_skill_id(text),
|
|
2355
|
+
skill_doc=None,
|
|
2356
|
+
run_log=str(log_path),
|
|
2357
|
+
errors=[f"invalid human evaluation: {exc}"],
|
|
2358
|
+
run_id=_log_field(text, "run_id"),
|
|
2359
|
+
)
|
|
2360
|
+
event = {"event": "human.evaluation", **evaluation.model_dump(mode="json", exclude_none=True)}
|
|
2361
|
+
text = _append_observation_event(text, section="Human Evaluation", event=event)
|
|
2362
|
+
_atomic_write_text(log_path, text)
|
|
2363
|
+
return SkillRunResult(
|
|
2364
|
+
ok=True,
|
|
2365
|
+
skill_id=_log_skill_id(text),
|
|
2366
|
+
skill_doc=None,
|
|
2367
|
+
run_log=str(log_path),
|
|
2368
|
+
errors=[],
|
|
2369
|
+
run_id=_log_field(text, "run_id"),
|
|
2370
|
+
)
|
|
2371
|
+
|
|
2372
|
+
|
|
2256
2373
|
@_locked_log_update
|
|
2257
2374
|
def update_token_usage(args) -> SkillRunResult:
|
|
2258
2375
|
log_path = Path(args.log).resolve()
|
|
@@ -2779,3 +2896,7 @@ def cmd_skill_knowledge(args) -> int:
|
|
|
2779
2896
|
|
|
2780
2897
|
def cmd_skill_feedback(args) -> int:
|
|
2781
2898
|
return _cmd_observation(args, update_feedback_observation, "feedback")
|
|
2899
|
+
|
|
2900
|
+
|
|
2901
|
+
def cmd_skill_evaluate(args) -> int:
|
|
2902
|
+
return _cmd_observation(args, update_human_evaluation, "evaluate")
|
|
@@ -5,7 +5,6 @@ from __future__ import annotations
|
|
|
5
5
|
import argparse
|
|
6
6
|
import hashlib
|
|
7
7
|
import json
|
|
8
|
-
import re
|
|
9
8
|
import tempfile
|
|
10
9
|
import urllib.error
|
|
11
10
|
import urllib.parse
|
|
@@ -18,7 +17,6 @@ from pathlib import Path, PurePosixPath
|
|
|
18
17
|
DEFAULT_SOURCE_REPOSITORY = "synthaicode/XRefKit"
|
|
19
18
|
DEFAULT_GITHUB_API = "https://api.github.com"
|
|
20
19
|
ASSET_PREFIX = "xrefkit-skills-"
|
|
21
|
-
VERSION_SUFFIX_RE = re.compile(r"^(?P<bundle>.+)-(?P<version>\d+\.\d+(?:\.\d+)?(?:[-+][0-9A-Za-z.-]+)?)$")
|
|
22
20
|
ALLOWED_ROOTS = {"skills", "knowledge", "review_axes", "schemas"}
|
|
23
21
|
|
|
24
22
|
|
|
@@ -44,7 +42,7 @@ class SyncResult:
|
|
|
44
42
|
}
|
|
45
43
|
|
|
46
44
|
|
|
47
|
-
def _github_json(url: str) ->
|
|
45
|
+
def _github_json(url: str) -> object:
|
|
48
46
|
request = urllib.request.Request(
|
|
49
47
|
url,
|
|
50
48
|
headers={
|
|
@@ -57,8 +55,6 @@ def _github_json(url: str) -> dict[str, object]:
|
|
|
57
55
|
payload = json.loads(response.read().decode("utf-8"))
|
|
58
56
|
except (urllib.error.URLError, urllib.error.HTTPError) as exc:
|
|
59
57
|
raise RuntimeError(f"could not read GitHub release metadata: {url}") from exc
|
|
60
|
-
if not isinstance(payload, dict):
|
|
61
|
-
raise RuntimeError(f"GitHub release metadata was not an object: {url}")
|
|
62
58
|
return payload
|
|
63
59
|
|
|
64
60
|
|
|
@@ -69,7 +65,17 @@ def _release(source_repository: str, release: str) -> dict[str, object]:
|
|
|
69
65
|
else:
|
|
70
66
|
encoded_release = urllib.parse.quote(release, safe="")
|
|
71
67
|
url = f"{DEFAULT_GITHUB_API}/repos/{encoded_repo}/releases/tags/{encoded_release}"
|
|
72
|
-
|
|
68
|
+
payload = _github_json(url)
|
|
69
|
+
return payload
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _releases(source_repository: str) -> list[dict[str, object]]:
|
|
73
|
+
encoded_repo = urllib.parse.quote(source_repository, safe="/")
|
|
74
|
+
url = f"{DEFAULT_GITHUB_API}/repos/{encoded_repo}/releases?per_page=100"
|
|
75
|
+
payload = _github_json(url)
|
|
76
|
+
if not isinstance(payload, list):
|
|
77
|
+
raise RuntimeError(f"GitHub releases metadata was not a list: {url}")
|
|
78
|
+
return [item for item in payload if isinstance(item, dict)]
|
|
73
79
|
|
|
74
80
|
|
|
75
81
|
def _assets(release: dict[str, object]) -> list[dict[str, object]]:
|
|
@@ -94,6 +100,52 @@ def _asset_for_bundle(release: dict[str, object], bundle: str) -> dict[str, obje
|
|
|
94
100
|
return candidates[0]
|
|
95
101
|
|
|
96
102
|
|
|
103
|
+
def _latest_release_for_bundle(source_repository: str, bundle: str) -> dict[str, object]:
|
|
104
|
+
tag_prefix = f"{ASSET_PREFIX}{bundle}-v"
|
|
105
|
+
candidates = []
|
|
106
|
+
for release in _releases(source_repository):
|
|
107
|
+
tag = release.get("tag_name")
|
|
108
|
+
if not isinstance(tag, str) or not tag.startswith(tag_prefix) or release.get("draft"):
|
|
109
|
+
continue
|
|
110
|
+
try:
|
|
111
|
+
_asset_for_bundle(release, bundle)
|
|
112
|
+
except RuntimeError:
|
|
113
|
+
continue
|
|
114
|
+
candidates.append(release)
|
|
115
|
+
if not candidates:
|
|
116
|
+
raise RuntimeError(
|
|
117
|
+
f"could not find a published Skill bundle release for {bundle}; "
|
|
118
|
+
f"expected tag prefix {tag_prefix}"
|
|
119
|
+
)
|
|
120
|
+
candidates.sort(
|
|
121
|
+
key=lambda release: str(release.get("published_at") or release.get("created_at") or ""),
|
|
122
|
+
reverse=True,
|
|
123
|
+
)
|
|
124
|
+
return candidates[0]
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _release_for_bundle(source_repository: str, release: str, bundle: str) -> dict[str, object]:
|
|
128
|
+
if release == "latest":
|
|
129
|
+
return _latest_release_for_bundle(source_repository, bundle)
|
|
130
|
+
return _release(source_repository, release)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _bundle_names(releases: list[dict[str, object]]) -> list[str]:
|
|
134
|
+
names: list[str] = []
|
|
135
|
+
for release in releases:
|
|
136
|
+
tag = release.get("tag_name")
|
|
137
|
+
if not isinstance(tag, str) or not tag.startswith(ASSET_PREFIX) or "-v" not in tag:
|
|
138
|
+
continue
|
|
139
|
+
bundle = tag[len(ASSET_PREFIX) :].rsplit("-v", 1)[0]
|
|
140
|
+
try:
|
|
141
|
+
_asset_for_bundle(release, bundle)
|
|
142
|
+
except RuntimeError:
|
|
143
|
+
continue
|
|
144
|
+
if bundle not in names:
|
|
145
|
+
names.append(bundle)
|
|
146
|
+
return names
|
|
147
|
+
|
|
148
|
+
|
|
97
149
|
def _download_asset(asset: dict[str, object]) -> tuple[str, bytes]:
|
|
98
150
|
name = asset.get("name")
|
|
99
151
|
url = asset.get("browser_download_url")
|
|
@@ -163,7 +215,7 @@ def sync_bundle(
|
|
|
163
215
|
repo = repo.resolve()
|
|
164
216
|
if not repo.is_dir():
|
|
165
217
|
raise FileNotFoundError(f"XRefKit repository not found: {repo}")
|
|
166
|
-
release_payload =
|
|
218
|
+
release_payload = _release_for_bundle(source_repository, release, bundle)
|
|
167
219
|
asset = _asset_for_bundle(release_payload, bundle)
|
|
168
220
|
asset_name, data = _download_asset(asset)
|
|
169
221
|
digest = hashlib.sha256(data).hexdigest()
|
|
@@ -240,7 +292,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
240
292
|
sync.add_argument("--repo", default=".", help="Target XRefKit repository; defaults to the current directory")
|
|
241
293
|
sync.add_argument("--source-repository", default=DEFAULT_SOURCE_REPOSITORY, help="GitHub owner/repository containing releases")
|
|
242
294
|
sync.add_argument("--bundle", action="append", help="Bundle name, for example csharp; repeat for multiple bundles")
|
|
243
|
-
sync.add_argument("--all", action="store_true", help="Synchronize every
|
|
295
|
+
sync.add_argument("--all", action="store_true", help="Synchronize every bundle with a matching Skill release")
|
|
244
296
|
sync.add_argument("--release", default="latest", help="Release tag, or latest (default)")
|
|
245
297
|
sync.add_argument("--force", action="store_true", help="Allow overwriting files not owned by a previous sync")
|
|
246
298
|
sync.add_argument("--dry-run", action="store_true", help="Download and validate without writing the repository")
|
|
@@ -248,16 +300,9 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
248
300
|
args = parser.parse_args(argv)
|
|
249
301
|
if not args.bundle and not args.all:
|
|
250
302
|
parser.error("one of --bundle or --all is required")
|
|
251
|
-
release_payload = _release(args.source_repository, args.release)
|
|
252
303
|
bundles = list(args.bundle or [])
|
|
253
304
|
if args.all:
|
|
254
|
-
|
|
255
|
-
name = asset.get("name")
|
|
256
|
-
if not isinstance(name, str) or not name.startswith(ASSET_PREFIX) or not name.endswith(".zip"):
|
|
257
|
-
continue
|
|
258
|
-
stem = name[:-4][len(ASSET_PREFIX) :]
|
|
259
|
-
match = VERSION_SUFFIX_RE.match(stem)
|
|
260
|
-
bundles.append(match.group("bundle") if match else stem)
|
|
305
|
+
bundles.extend(_bundle_names(_releases(args.source_repository)))
|
|
261
306
|
bundles = list(dict.fromkeys(bundles))
|
|
262
307
|
results = [
|
|
263
308
|
sync_bundle(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: xrefkit
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.8
|
|
4
4
|
Summary: Portable XID, Skill, Knowledge, workflow, and MCP runtime for XRefKit
|
|
5
5
|
Author: synthaicode
|
|
6
6
|
License: MIT License
|
|
@@ -178,6 +178,22 @@ If the `xrefkit` command is not available on `PATH`, use the module form:
|
|
|
178
178
|
python -m xrefkit --help
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
+
### Markdown WBS validation and rollups
|
|
182
|
+
|
|
183
|
+
Keep one task per row in a Markdown table, then validate and regenerate the
|
|
184
|
+
feature, phase, and feature-by-phase views without editing the source table:
|
|
185
|
+
|
|
186
|
+
```powershell
|
|
187
|
+
python -m xrefkit wbs validate samples/wbs.md
|
|
188
|
+
python -m xrefkit wbs rollup samples/wbs.md
|
|
189
|
+
python -m xrefkit wbs csv samples/wbs.md .tmp/wbs.csv
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
The validator rejects missing or extra cells, duplicate IDs, non-numeric or
|
|
193
|
+
negative effort, and statuses outside `todo`, `doing`, and `done`. CSV output
|
|
194
|
+
uses UTF-8 with BOM for Excel compatibility; the Markdown WBS remains the
|
|
195
|
+
source of truth.
|
|
196
|
+
|
|
181
197
|
To use the integrated MCP server, install the optional MCP dependencies:
|
|
182
198
|
|
|
183
199
|
```powershell
|
|
@@ -3,6 +3,7 @@ README.md
|
|
|
3
3
|
pyproject.toml
|
|
4
4
|
tests/test_base_sync_ownership.py
|
|
5
5
|
tests/test_boundary_analysis.py
|
|
6
|
+
tests/test_brownfield_file_editing_protocol.py
|
|
6
7
|
tests/test_calibration_lint.py
|
|
7
8
|
tests/test_check_skill_knowledge_xids.py
|
|
8
9
|
tests/test_cli.py
|
|
@@ -19,6 +20,7 @@ tests/test_error_policy_locator.py
|
|
|
19
20
|
tests/test_fm_multiroot.py
|
|
20
21
|
tests/test_gate.py
|
|
21
22
|
tests/test_goal_desired_state.py
|
|
23
|
+
tests/test_human_evaluation.py
|
|
22
24
|
tests/test_instruction_workflow.py
|
|
23
25
|
tests/test_knowledge_relations_validator.py
|
|
24
26
|
tests/test_mcp_setup.py
|
|
@@ -93,6 +95,7 @@ xrefkit/mcp/startup_contract_pack.py
|
|
|
93
95
|
xrefkit/models/__init__.py
|
|
94
96
|
xrefkit/models/common.py
|
|
95
97
|
xrefkit/models/effective_bundle.py
|
|
98
|
+
xrefkit/models/human_evaluation.py
|
|
96
99
|
xrefkit/models/local_manifest.py
|
|
97
100
|
xrefkit/models/package_manifest.py
|
|
98
101
|
xrefkit/models/run_log.py
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/7a682a5272907354/contracts.json
RENAMED
|
File without changes
|
{xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/7a682a5272907354/model_body.md
RENAMED
|
File without changes
|
{xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/9929294385ccb7b0/contracts.json
RENAMED
|
File without changes
|
{xrefkit-0.4.7 → xrefkit-0.4.8}/xrefkit/resources/base/generations/9929294385ccb7b0/model_body.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|