xrefkit 0.3.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. {xrefkit-0.3.0 → xrefkit-0.4.1}/PKG-INFO +60 -2
  2. {xrefkit-0.3.0 → xrefkit-0.4.1}/README.md +59 -1
  3. {xrefkit-0.3.0 → xrefkit-0.4.1}/pyproject.toml +1 -1
  4. xrefkit-0.4.1/tests/test_boundary_analysis.py +157 -0
  5. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_dashboard.py +69 -0
  6. xrefkit-0.4.1/tests/test_instruction_workflow.py +100 -0
  7. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/__init__.py +1 -1
  8. xrefkit-0.4.1/xrefkit/boundary_analysis.py +653 -0
  9. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/cli.py +3 -1
  10. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/dashboard.py +319 -13
  11. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/catalog.py +4 -3
  12. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/contracts.py +5 -3
  13. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/schemas.py +6 -1
  14. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/server.py +21 -14
  15. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/operations_cli.py +76 -0
  16. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/skillrun.py +140 -8
  17. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit.egg-info/PKG-INFO +60 -2
  18. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit.egg-info/SOURCES.txt +3 -0
  19. {xrefkit-0.3.0 → xrefkit-0.4.1}/LICENSE +0 -0
  20. {xrefkit-0.3.0 → xrefkit-0.4.1}/setup.cfg +0 -0
  21. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_base_sync_ownership.py +0 -0
  22. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_calibration_lint.py +0 -0
  23. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_check_skill_knowledge_xids.py +0 -0
  24. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_cli.py +0 -0
  25. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_collect_analyzer_sarif.py +0 -0
  26. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_convert_to_xrefkit_skill.py +0 -0
  27. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_cs_scope_probe.py +0 -0
  28. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_csharp_commonality.py +0 -0
  29. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_csharp_naming_profile.py +0 -0
  30. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_ctx.py +0 -0
  31. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_cutover_readiness.py +0 -0
  32. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_error_policy_audit.py +0 -0
  33. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_error_policy_locator.py +0 -0
  34. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_fm_multiroot.py +0 -0
  35. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_gate.py +0 -0
  36. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_goal_desired_state.py +0 -0
  37. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_knowledge_relations_validator.py +0 -0
  38. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_ownership.py +0 -0
  39. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_packmeta.py +0 -0
  40. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_project_quality_baseline.py +0 -0
  41. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_resource_provider.py +0 -0
  42. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_runtime_contracts.py +0 -0
  43. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_sarif_to_locator.py +0 -0
  44. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_skill_runtime_audit.py +0 -0
  45. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_skillmeta.py +0 -0
  46. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_structure_catalog.py +0 -0
  47. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_xref.py +0 -0
  48. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_xrefkit_instance.py +0 -0
  49. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_xrefkit_tools.py +0 -0
  50. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_xrefkit_v2_discovery.py +0 -0
  51. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_xrefkit_v2_models.py +0 -0
  52. {xrefkit-0.3.0 → xrefkit-0.4.1}/tests/test_xrefkit_v2_pipeline.py +0 -0
  53. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/__main__.py +0 -0
  54. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/catalog_cli.py +0 -0
  55. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/contracts.py +0 -0
  56. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/ctx.py +0 -0
  57. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/discovery.py +0 -0
  58. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/gate.py +0 -0
  59. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/goalstate.py +0 -0
  60. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/hashing.py +0 -0
  61. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/import_skill.py +0 -0
  62. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/instance.py +0 -0
  63. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/loaders.py +0 -0
  64. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/__init__.py +0 -0
  65. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/audit.py +0 -0
  66. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/bootstrap.py +0 -0
  67. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/cli.py +0 -0
  68. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/client_cache.py +0 -0
  69. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/context_registry.py +0 -0
  70. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/dist.py +0 -0
  71. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/ownership.py +0 -0
  72. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/repository.py +0 -0
  73. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp/startup_contract_pack.py +0 -0
  74. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/mcp_tools.py +0 -0
  75. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/__init__.py +0 -0
  76. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/common.py +0 -0
  77. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/effective_bundle.py +0 -0
  78. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/local_manifest.py +0 -0
  79. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/package_manifest.py +0 -0
  80. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/run_log.py +0 -0
  81. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/server_config.py +0 -0
  82. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/models/skill_definition.py +0 -0
  83. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/ownership.py +0 -0
  84. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/packmeta.py +0 -0
  85. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/registry.py +0 -0
  86. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resolver.py +0 -0
  87. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resource_provider.py +0 -0
  88. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/contracts.json +0 -0
  89. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/current.json +0 -0
  90. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/generations/7a682a5272907354/contracts.json +0 -0
  91. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/generations/7a682a5272907354/model_body.md +0 -0
  92. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/generations/9929294385ccb7b0/contracts.json +0 -0
  93. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/generations/9929294385ccb7b0/model_body.md +0 -0
  94. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/resources/base/model_body.md +0 -0
  95. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/runlog.py +0 -0
  96. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/skillmeta.py +0 -0
  97. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/structure_catalog.py +0 -0
  98. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/tools/__init__.py +0 -0
  99. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/tools/__main__.py +0 -0
  100. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/v2_cli.py +0 -0
  101. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/workspace.py +0 -0
  102. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit/xref.py +0 -0
  103. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit.egg-info/dependency_links.txt +0 -0
  104. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit.egg-info/entry_points.txt +0 -0
  105. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit.egg-info/requires.txt +0 -0
  106. {xrefkit-0.3.0 → xrefkit-0.4.1}/xrefkit.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: xrefkit
3
- Version: 0.3.0
3
+ Version: 0.4.1
4
4
  Summary: Portable XID, Skill, Knowledge, workflow, and MCP runtime for XRefKit
5
5
  Author: synthaicode
6
6
  License: MIT License
@@ -91,7 +91,7 @@ XRefKit makes AI work explicit by separating:
91
91
 
92
92
  - Skills: executable work units, each identified by a capability/tuning/responsibility triad and carrying its execution and check contract
93
93
  - Knowledge: source-backed domain facts and local rules loaded only when needed
94
- - Workflow protocol: the generic, deterministic per-Skill control (phases, verification, closure) that wraps every Skill run
94
+ - Workflow protocol: the generic, deterministic control for Skill-backed and instruction-backed runs (phases, verification, closure)
95
95
  - Semantic routing: selecting the right Skill for a goal from user intent and the Skill catalog
96
96
  - Evidence: logs, judgments, concerns, and quality checks
97
97
  - XIDs: stable references that survive file movement and restructuring so AI can load targeted context without treating the whole repository as one prompt
@@ -109,6 +109,18 @@ and handoff records from collapsing into one opaque instruction block.
109
109
  4. Agents are routed semantically to the right Skill and load only the relevant context.
110
110
  5. Evidence and quality gates make incomplete or unsupported work visible.
111
111
 
112
+ When an instruction has no matching Skill, open an instruction-backed workflow
113
+ run explicitly. The run requires either user-supplied procedural completion
114
+ conditions or an explicit opt-in to the repository defaults:
115
+
116
+ ```powershell
117
+ xrefkit workflow run --task "Perform the requested procedure" `
118
+ --use-default-completion-conditions --json
119
+ ```
120
+
121
+ `verify` and `close` determine procedural completion only. Output quality is
122
+ recorded separately after human acceptance with the existing feedback record.
123
+
112
124
  ## Quick Start
113
125
 
114
126
  Install the package and initialize an instance:
@@ -133,6 +145,52 @@ searches and XID resolutions then share the same `run_id` as the client Skill
133
145
  Run. The client separately records actual model-context loading and judgment
134
146
  application with `xrefkit skill knowledge --action load|apply`.
135
147
 
148
+ ## Skill Run Observation Dashboard
149
+
150
+ The local dashboard lets a human inspect Skill run status, closure and quality
151
+ gates, evidence, handoffs, XID usage, missing information, and proposal-only
152
+ boundary analysis.
153
+
154
+ Start it from the repository root:
155
+
156
+ ```powershell
157
+ python -m xrefkit dashboard serve --root .
158
+ ```
159
+
160
+ Open [http://127.0.0.1:8765/](http://127.0.0.1:8765/). To open the browser
161
+ automatically, add `--open-browser`. Use `--port 8766` when the default port is
162
+ already in use, or `--sessions-dir path\to\sessions` when logs are stored
163
+ elsewhere.
164
+
165
+ The main tabs are:
166
+
167
+ - **Overview / Attention / Closure**: run status, blockers, phases, closure,
168
+ and quality-gate state.
169
+ - **Evidence / Handoff**: outputs, checks, handoffs, unknowns, risks, and
170
+ judgments needed for review and continuity.
171
+ - **XID Usage**: selected, resolved, loaded, used, available, and unused XIDs.
172
+ - **Analysis**: deterministic candidates for Knowledge correction, Skill
173
+ correction, split, merge, or usage-gap investigation. Review the evidence,
174
+ counterevidence, unknowns, and verification plan before changing canonical
175
+ files. The dashboard never applies these proposals automatically.
176
+ - **Missing Information**: absent correlation, MCP, Knowledge, or feedback
177
+ records.
178
+
179
+ Export the dashboard data and create a human-reviewable boundary report:
180
+
181
+ ```powershell
182
+ python -m xrefkit dashboard data --root . > work/reports/dashboard-observation.json
183
+ python -m xrefkit analysis boundary report `
184
+ --input work/reports/dashboard-observation.json `
185
+ --out work/reports/boundary-observation.md
186
+ ```
187
+
188
+ The running dashboard also exposes JSON at
189
+ [http://127.0.0.1:8765/api/runs](http://127.0.0.1:8765/api/runs) and health at
190
+ [http://127.0.0.1:8765/healthz](http://127.0.0.1:8765/healthz). Stop a foreground
191
+ server with `Ctrl+C`. For the complete review loop and screen guide, see the
192
+ [Skill Run Observation Dashboard Usage guide](docs/guides/086_skill_run_observation_dashboard_usage.md).
193
+
136
194
  XRefKit is designed to be driven by an AI agent. The agent first resolves the
137
195
  startup contract XID, selects a Skill or source target from a compact catalog,
138
196
  and expands only the selected body.
@@ -52,7 +52,7 @@ XRefKit makes AI work explicit by separating:
52
52
 
53
53
  - Skills: executable work units, each identified by a capability/tuning/responsibility triad and carrying its execution and check contract
54
54
  - Knowledge: source-backed domain facts and local rules loaded only when needed
55
- - Workflow protocol: the generic, deterministic per-Skill control (phases, verification, closure) that wraps every Skill run
55
+ - Workflow protocol: the generic, deterministic control for Skill-backed and instruction-backed runs (phases, verification, closure)
56
56
  - Semantic routing: selecting the right Skill for a goal from user intent and the Skill catalog
57
57
  - Evidence: logs, judgments, concerns, and quality checks
58
58
  - XIDs: stable references that survive file movement and restructuring so AI can load targeted context without treating the whole repository as one prompt
@@ -70,6 +70,18 @@ and handoff records from collapsing into one opaque instruction block.
70
70
  4. Agents are routed semantically to the right Skill and load only the relevant context.
71
71
  5. Evidence and quality gates make incomplete or unsupported work visible.
72
72
 
73
+ When an instruction has no matching Skill, open an instruction-backed workflow
74
+ run explicitly. The run requires either user-supplied procedural completion
75
+ conditions or an explicit opt-in to the repository defaults:
76
+
77
+ ```powershell
78
+ xrefkit workflow run --task "Perform the requested procedure" `
79
+ --use-default-completion-conditions --json
80
+ ```
81
+
82
+ `verify` and `close` determine procedural completion only. Output quality is
83
+ recorded separately after human acceptance with the existing feedback record.
84
+
73
85
  ## Quick Start
74
86
 
75
87
  Install the package and initialize an instance:
@@ -94,6 +106,52 @@ searches and XID resolutions then share the same `run_id` as the client Skill
94
106
  Run. The client separately records actual model-context loading and judgment
95
107
  application with `xrefkit skill knowledge --action load|apply`.
96
108
 
109
+ ## Skill Run Observation Dashboard
110
+
111
+ The local dashboard lets a human inspect Skill run status, closure and quality
112
+ gates, evidence, handoffs, XID usage, missing information, and proposal-only
113
+ boundary analysis.
114
+
115
+ Start it from the repository root:
116
+
117
+ ```powershell
118
+ python -m xrefkit dashboard serve --root .
119
+ ```
120
+
121
+ Open [http://127.0.0.1:8765/](http://127.0.0.1:8765/). To open the browser
122
+ automatically, add `--open-browser`. Use `--port 8766` when the default port is
123
+ already in use, or `--sessions-dir path\to\sessions` when logs are stored
124
+ elsewhere.
125
+
126
+ The main tabs are:
127
+
128
+ - **Overview / Attention / Closure**: run status, blockers, phases, closure,
129
+ and quality-gate state.
130
+ - **Evidence / Handoff**: outputs, checks, handoffs, unknowns, risks, and
131
+ judgments needed for review and continuity.
132
+ - **XID Usage**: selected, resolved, loaded, used, available, and unused XIDs.
133
+ - **Analysis**: deterministic candidates for Knowledge correction, Skill
134
+ correction, split, merge, or usage-gap investigation. Review the evidence,
135
+ counterevidence, unknowns, and verification plan before changing canonical
136
+ files. The dashboard never applies these proposals automatically.
137
+ - **Missing Information**: absent correlation, MCP, Knowledge, or feedback
138
+ records.
139
+
140
+ Export the dashboard data and create a human-reviewable boundary report:
141
+
142
+ ```powershell
143
+ python -m xrefkit dashboard data --root . > work/reports/dashboard-observation.json
144
+ python -m xrefkit analysis boundary report `
145
+ --input work/reports/dashboard-observation.json `
146
+ --out work/reports/boundary-observation.md
147
+ ```
148
+
149
+ The running dashboard also exposes JSON at
150
+ [http://127.0.0.1:8765/api/runs](http://127.0.0.1:8765/api/runs) and health at
151
+ [http://127.0.0.1:8765/healthz](http://127.0.0.1:8765/healthz). Stop a foreground
152
+ server with `Ctrl+C`. For the complete review loop and screen guide, see the
153
+ [Skill Run Observation Dashboard Usage guide](docs/guides/086_skill_run_observation_dashboard_usage.md).
154
+
97
155
  XRefKit is designed to be driven by an AI agent. The agent first resolves the
98
156
  startup contract XID, selects a Skill or source target from a compact catalog,
99
157
  and expands only the selected body.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "xrefkit"
7
- version = "0.3.0"
7
+ version = "0.4.1"
8
8
  description = "Portable XID, Skill, Knowledge, workflow, and MCP runtime for XRefKit"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -0,0 +1,157 @@
1
+ import contextlib
2
+ import io
3
+ import json
4
+ import tempfile
5
+ import unittest
6
+ from pathlib import Path
7
+
8
+ from xrefkit.__main__ import main
9
+ from xrefkit.boundary_analysis import analyze_dashboard_payload, render_markdown
10
+
11
+
12
+ class BoundaryAnalysisTests(unittest.TestCase):
13
+ def _run(
14
+ self,
15
+ skill_id: str,
16
+ path: str,
17
+ xids: list[str],
18
+ *,
19
+ feedback: list[dict[str, str]] | None = None,
20
+ ) -> dict[str, object]:
21
+ return {
22
+ "path": path,
23
+ "name": Path(path).name,
24
+ "skill_id": skill_id,
25
+ "run_id": f"run-{path}",
26
+ "mcp_session_id": f"mcp-{path}",
27
+ "repository_fingerprint": "repo-001",
28
+ "status": "closed",
29
+ "closure_status": "done",
30
+ "quality_required": True,
31
+ "quality_status": "done",
32
+ "selected_xids": xids,
33
+ "queried_xids": xids,
34
+ "loaded_xids": xids,
35
+ "used_xids": xids,
36
+ "unused_xids": [],
37
+ "available_xids": xids,
38
+ "queried_not_loaded_xids": [],
39
+ "loaded_not_applied_xids": [],
40
+ "missing_information": [],
41
+ "observation_events": feedback or [],
42
+ "mcp_events": [],
43
+ }
44
+
45
+ def _payload(self) -> dict[str, object]:
46
+ xid_a = "XID-A-001"
47
+ xid_b = "XID-B-001"
48
+ return {
49
+ "schema": "xrefkit.dashboard/v1",
50
+ "audit_errors": [],
51
+ "runs": [
52
+ self._run(
53
+ "alpha",
54
+ "work/sessions/alpha-a-1.md",
55
+ [xid_a],
56
+ feedback=[
57
+ {"event": "human.feedback", "status": "corrected", "target": xid_a, "note": "wrong fact"},
58
+ {"event": "human.feedback", "status": "corrected", "target": "OUT-001", "note": "wrong procedure"},
59
+ ],
60
+ ),
61
+ self._run(
62
+ "alpha",
63
+ "work/sessions/alpha-a-2.md",
64
+ [xid_a],
65
+ feedback=[
66
+ {"event": "human.feedback", "status": "rejected", "target": xid_a, "note": "stale fact"},
67
+ {"event": "human.feedback", "status": "rejected", "target": "OUT-002", "note": "wrong decision"},
68
+ ],
69
+ ),
70
+ self._run("alpha", "work/sessions/alpha-b-1.md", [xid_b]),
71
+ self._run("alpha", "work/sessions/alpha-b-2.md", [xid_b]),
72
+ self._run("beta", "work/sessions/beta-1.md", [xid_a, xid_b]),
73
+ self._run("beta", "work/sessions/beta-2.md", [xid_a, xid_b]),
74
+ ],
75
+ }
76
+
77
+ def test_analysis_emits_conservative_boundary_candidates(self) -> None:
78
+ report = analyze_dashboard_payload(self._payload(), source_hash="source-001", min_samples=2)
79
+
80
+ self.assertEqual("xrefkit.boundary_observation/v1", report["schema"])
81
+ self.assertEqual("proposal_only", report["status"])
82
+ self.assertEqual(6, report["sample_count"])
83
+ self.assertEqual(6, report["correlation"]["exact"])
84
+ categories = {item["category"] for item in report["proposals"]}
85
+ self.assertIn("split", categories)
86
+ self.assertIn("merge", categories)
87
+ self.assertIn("knowledge_correction", categories)
88
+ self.assertIn("skill_correction", categories)
89
+ self.assertTrue(all(item["decision"]["status"] == "pending" for item in report["proposals"]))
90
+
91
+ xid_row = next(item for item in report["xid_usage"] if item["xid"] == "XID-A-001")
92
+ self.assertEqual(4, xid_row["run_count"])
93
+ self.assertEqual(4, xid_row["used_count"])
94
+
95
+ def test_markdown_explains_proposal_only_boundary(self) -> None:
96
+ markdown = render_markdown(analyze_dashboard_payload(self._payload(), source_hash="source-001"))
97
+
98
+ self.assertIn("Proposal-only output", markdown)
99
+ self.assertIn("## Proposals", markdown)
100
+ self.assertIn("Counterevidence", markdown)
101
+ self.assertIn("## Decision", markdown)
102
+ self.assertIn("XID-A-001", markdown)
103
+
104
+ def test_cli_writes_markdown_and_can_emit_json(self) -> None:
105
+ with tempfile.TemporaryDirectory() as tmp:
106
+ root = Path(tmp)
107
+ source = root / "dashboard.json"
108
+ output = root / "reports" / "boundary.md"
109
+ source.write_text(json.dumps(self._payload()), encoding="utf-8")
110
+ stdout = io.StringIO()
111
+ with contextlib.redirect_stdout(stdout):
112
+ result = main(
113
+ [
114
+ "analysis",
115
+ "boundary",
116
+ "report",
117
+ "--input",
118
+ str(source),
119
+ "--out",
120
+ str(output),
121
+ "--json",
122
+ ]
123
+ )
124
+
125
+ self.assertEqual(0, result)
126
+ self.assertTrue(output.exists())
127
+ self.assertIn("Proposal-only output", output.read_text(encoding="utf-8"))
128
+ response = json.loads(stdout.getvalue())
129
+ self.assertEqual("proposal_only", response["status"])
130
+ self.assertEqual(4, response["summary"]["proposals"])
131
+
132
+ def test_invalid_dashboard_input_fails_without_writing(self) -> None:
133
+ with tempfile.TemporaryDirectory() as tmp:
134
+ source = Path(tmp) / "invalid.json"
135
+ output = Path(tmp) / "boundary.md"
136
+ source.write_text(json.dumps({"runs": "not-a-list"}), encoding="utf-8")
137
+ stdout = io.StringIO()
138
+ with contextlib.redirect_stdout(stdout):
139
+ result = main(
140
+ [
141
+ "analysis",
142
+ "boundary",
143
+ "report",
144
+ "--input",
145
+ str(source),
146
+ "--out",
147
+ str(output),
148
+ ]
149
+ )
150
+
151
+ self.assertEqual(1, result)
152
+ self.assertFalse(output.exists())
153
+ self.assertIn("runs array", stdout.getvalue())
154
+
155
+
156
+ if __name__ == "__main__":
157
+ unittest.main()
@@ -379,6 +379,20 @@ class DashboardTests(unittest.TestCase):
379
379
  self.assertEqual({}, by_run)
380
380
  self.assertIn("cannot read audit log", errors[0])
381
381
 
382
+ def test_dashboard_payload_includes_proposal_only_boundary_analysis(self) -> None:
383
+ with tempfile.TemporaryDirectory() as tmp:
384
+ root = Path(tmp)
385
+ self._write_closed_run(root)
386
+
387
+ payload = build_payload(root, root / "work" / "sessions")
388
+ analysis = payload["boundary_analysis"]
389
+
390
+ self.assertIsInstance(analysis, dict)
391
+ self.assertEqual("xrefkit.boundary_observation/v1", analysis["schema"])
392
+ self.assertEqual("proposal_only", analysis["status"])
393
+ self.assertEqual(1, analysis["sample_count"])
394
+ self.assertEqual(0, analysis["summary"]["proposals"])
395
+
382
396
  def test_dashboard_html_splits_categories(self) -> None:
383
397
  with tempfile.TemporaryDirectory() as tmp:
384
398
  root = Path(tmp)
@@ -392,6 +406,7 @@ class DashboardTests(unittest.TestCase):
392
406
  self.assertIn('data-panel="evidence"', html)
393
407
  self.assertIn('data-panel="handoff"', html)
394
408
  self.assertIn('data-panel="xids"', html)
409
+ self.assertIn('data-panel="analysis"', html)
395
410
  self.assertIn('data-panel="missing-information"', html)
396
411
  self.assertIn('id="overview"', html)
397
412
  self.assertIn('id="attention"', html)
@@ -399,8 +414,62 @@ class DashboardTests(unittest.TestCase):
399
414
  self.assertIn('id="evidence"', html)
400
415
  self.assertIn('id="handoff"', html)
401
416
  self.assertIn('id="xids"', html)
417
+ self.assertIn('id="analysis"', html)
402
418
  self.assertIn('id="missing-information"', html)
419
+ self.assertIn("Proposal-only analysis", html)
420
+ self.assertIn("No boundary proposals reached", html)
403
421
  self.assertIn("Missing Information Ranking", html)
404
422
  self.assertIn("Available Knowledge XIDs (base/local)", html)
405
423
  self.assertIn("local-service-map-001", html)
406
424
  self.assertIn("LOCAL-KNOWLEDGE-UNUSED-001", html)
425
+ self.assertIn('id="run-search"', html)
426
+ self.assertIn('data-status="blocked"', html)
427
+ self.assertIn('id="refresh-runs"', html)
428
+ self.assertIn("async function refreshDashboard()", html)
429
+ self.assertIn("selectRun(run.dataset.runPath)", html)
430
+ self.assertIn("data-run-path=", html)
431
+
432
+ def test_dashboard_html_search_index_contains_run_correlation_fields(self) -> None:
433
+ with tempfile.TemporaryDirectory() as tmp:
434
+ root = Path(tmp)
435
+ self._write_closed_run(root)
436
+
437
+ payload = build_payload(root, root / "work" / "sessions")
438
+ run_id = payload["runs"][0]["run_id"]
439
+ html = _html_page(payload)
440
+
441
+ self.assertIsNotNone(run_id)
442
+ self.assertIn(str(run_id).lower(), html)
443
+ self.assertIn("sample_skill", html)
444
+
445
+ def test_dashboard_html_escapes_boundary_proposal_fields(self) -> None:
446
+ with tempfile.TemporaryDirectory() as tmp:
447
+ root = Path(tmp)
448
+ self._write_closed_run(root)
449
+
450
+ payload = build_payload(root, root / "work" / "sessions")
451
+ payload["boundary_analysis"]["summary"]["proposals"] = 1
452
+ payload["boundary_analysis"]["proposals"] = [
453
+ {
454
+ "proposal_id": "bo-test",
455
+ "proposal": "investigate",
456
+ "category": "skill_correction",
457
+ "skill_ids": ["sample_skill"],
458
+ "subject_xids": ["xid-test"],
459
+ "support": 2,
460
+ "evidence_refs": ["<script>alert(1)</script>"],
461
+ "rationale": "<script>alert(2)</script>",
462
+ "counterevidence": ["counter"],
463
+ "unknowns": ["unknown"],
464
+ "verification_plan": ["verify"],
465
+ "decision": {"status": "pending", "owner": None},
466
+ }
467
+ ]
468
+
469
+ html = _html_page(payload)
470
+
471
+ self.assertIn("bo-test", html)
472
+ self.assertIn("Counterevidence", html)
473
+ self.assertIn("&lt;script&gt;alert(1)&lt;/script&gt;", html)
474
+ self.assertIn("&lt;script&gt;alert(2)&lt;/script&gt;", html)
475
+ self.assertNotIn("<script>alert(1)", html)
@@ -0,0 +1,100 @@
1
+ import contextlib
2
+ import io
3
+ import json
4
+ from pathlib import Path
5
+
6
+ from xrefkit.__main__ import main
7
+
8
+
9
+ def _run(root: Path, *args: str) -> int:
10
+ with contextlib.redirect_stdout(io.StringIO()):
11
+ argv = list(args)
12
+ if argv[:2] == ["workflow", "run"]:
13
+ argv.extend(["--root", str(root)])
14
+ return main(argv)
15
+
16
+
17
+ def test_instruction_workflow_requires_completion_conditions(tmp_path: Path) -> None:
18
+ out = tmp_path / "work" / "sessions" / "run.md"
19
+ assert _run(tmp_path, "workflow", "run", "--task", "Do work", "--out", str(out)) == 1
20
+ assert not out.exists()
21
+
22
+
23
+ def test_instruction_workflow_uses_default_conditions_and_shared_protocol(tmp_path: Path) -> None:
24
+ out = tmp_path / "work" / "sessions" / "run.md"
25
+ assert _run(
26
+ tmp_path,
27
+ "workflow",
28
+ "run",
29
+ "--task",
30
+ "Do work",
31
+ "--out",
32
+ str(out),
33
+ "--use-default-completion-conditions",
34
+ ) == 0
35
+ text = out.read_text(encoding="utf-8")
36
+ assert "# Workflow Run Log" in text
37
+ assert "## Run Load Gate" in text
38
+ assert "- basis: `default`" in text
39
+ assert "- quality_policy: `human_acceptance`" in text
40
+
41
+ assert _run(
42
+ tmp_path,
43
+ "skill",
44
+ "workitem",
45
+ "--log",
46
+ str(out),
47
+ "--item",
48
+ "WI-001",
49
+ "--text",
50
+ "Perform the instruction",
51
+ "--status",
52
+ "done",
53
+ "--role",
54
+ "instruction:executor",
55
+ ) == 0
56
+ for artifact_id, kind, target, role in (
57
+ ("OUT-001", "output", "output.md", "instruction:executor"),
58
+ ("EVD-001", "evidence", "test command", "instruction:checker"),
59
+ ):
60
+ assert _run(
61
+ tmp_path,
62
+ "skill",
63
+ "artifact",
64
+ "--log",
65
+ str(out),
66
+ "--artifact",
67
+ artifact_id,
68
+ "--kind",
69
+ kind,
70
+ "--target",
71
+ target,
72
+ "--item",
73
+ "WI-001",
74
+ "--status",
75
+ "done",
76
+ "--role",
77
+ role,
78
+ ) == 0
79
+ assert _run(tmp_path, "skill", "phase", "--log", str(out), "--phase", "execution", "--status", "done", "--role", "instruction:executor") == 0
80
+ assert _run(tmp_path, "skill", "phase", "--log", str(out), "--phase", "handoff", "--status", "done", "--role", "instruction:handoff_owner") == 0
81
+ assert _run(tmp_path, "skill", "verify", "--log", str(out)) == 0
82
+
83
+ # Quality is a human decision and is recorded separately from progression.
84
+ assert _run(
85
+ tmp_path,
86
+ "skill",
87
+ "feedback",
88
+ "--log",
89
+ str(out),
90
+ "--kind",
91
+ "human",
92
+ "--status",
93
+ "accepted",
94
+ "--target",
95
+ "OUT-001",
96
+ "--note",
97
+ "human accepted output quality",
98
+ ) == 0
99
+ assert _run(tmp_path, "skill", "close", "--log", str(out)) == 0
100
+ assert "## Completion Conditions" in out.read_text(encoding="utf-8")
@@ -2,4 +2,4 @@
2
2
 
3
3
  __all__ = ["__version__"]
4
4
 
5
- __version__ = "0.3.0"
5
+ __version__ = "0.4.1"