associate 0.8.0__tar.gz → 0.9.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. associate-0.9.1/.devague/current +1 -0
  2. associate-0.9.1/.devague/current_plan +1 -0
  3. associate-0.9.1/.devague/deliveries/associate-on-pi-with-opinionated-tools.json +492 -0
  4. associate-0.9.1/.devague/frames/associate-on-pi-with-opinionated-tools.json +1927 -0
  5. associate-0.9.1/.devague/frames/webglass-code-lens-as-tools.json +401 -0
  6. associate-0.9.1/.devague/plans/associate-on-pi-with-opinionated-tools.json +1190 -0
  7. associate-0.9.1/.devague/plans/webglass-code-lens-as-tools.json +330 -0
  8. {associate-0.8.0 → associate-0.9.1}/.github/workflows/tests.yml +21 -0
  9. {associate-0.8.0 → associate-0.9.1}/.gitignore +24 -0
  10. associate-0.9.1/.pi/extensions/associate/index.ts +282 -0
  11. associate-0.9.1/.pi/extensions/associate/lib/contain.ts +739 -0
  12. associate-0.9.1/.pi/extensions/associate/lib/context.ts +76 -0
  13. associate-0.9.1/.pi/extensions/associate/lib/contract.ts +150 -0
  14. associate-0.9.1/.pi/extensions/associate/lib/guard.ts +173 -0
  15. associate-0.9.1/.pi/extensions/associate/lib/handback.ts +129 -0
  16. associate-0.9.1/.pi/extensions/associate/lib/paths.ts +30 -0
  17. associate-0.9.1/.pi/extensions/associate/lib/preflight.ts +40 -0
  18. associate-0.9.1/.pi/extensions/associate/lib/prompt.ts +114 -0
  19. associate-0.9.1/.pi/extensions/associate/lib/provider.ts +251 -0
  20. associate-0.9.1/.pi/extensions/associate/lib/runtime.ts +142 -0
  21. associate-0.9.1/.pi/extensions/associate/lib/session.ts +181 -0
  22. associate-0.9.1/.pi/extensions/associate/lib/statements.ts +643 -0
  23. associate-0.9.1/.pi/extensions/associate/lib/torch.ts +266 -0
  24. associate-0.9.1/.pi/extensions/associate/lib/walk.ts +462 -0
  25. associate-0.9.1/.pi/extensions/associate/tests/codelens.test.ts +313 -0
  26. associate-0.9.1/.pi/extensions/associate/tests/contain.test.ts +420 -0
  27. associate-0.9.1/.pi/extensions/associate/tests/contract.test.ts +146 -0
  28. associate-0.9.1/.pi/extensions/associate/tests/extension.test.ts +356 -0
  29. associate-0.9.1/.pi/extensions/associate/tests/guard.test.ts +118 -0
  30. associate-0.9.1/.pi/extensions/associate/tests/load-extension.ts +168 -0
  31. associate-0.9.1/.pi/extensions/associate/tests/pi-integration.test.ts +220 -0
  32. associate-0.9.1/.pi/extensions/associate/tests/preflight.test.ts +37 -0
  33. associate-0.9.1/.pi/extensions/associate/tests/prompt.test.ts +127 -0
  34. associate-0.9.1/.pi/extensions/associate/tests/provider.test.ts +290 -0
  35. associate-0.9.1/.pi/extensions/associate/tests/read.test.ts +435 -0
  36. associate-0.9.1/.pi/extensions/associate/tests/run.sh +58 -0
  37. associate-0.9.1/.pi/extensions/associate/tests/runtime.test.ts +204 -0
  38. associate-0.9.1/.pi/extensions/associate/tests/search.test.ts +647 -0
  39. associate-0.9.1/.pi/extensions/associate/tests/session.test.ts +150 -0
  40. associate-0.9.1/.pi/extensions/associate/tests/shell.test.ts +561 -0
  41. associate-0.9.1/.pi/extensions/associate/tests/smoke.test.ts +6 -0
  42. associate-0.9.1/.pi/extensions/associate/tests/statements.test.ts +638 -0
  43. associate-0.9.1/.pi/extensions/associate/tests/stubs/typebox.ts +39 -0
  44. associate-0.9.1/.pi/extensions/associate/tests/torch.test.ts +244 -0
  45. associate-0.9.1/.pi/extensions/associate/tests/walk.test.ts +410 -0
  46. associate-0.9.1/.pi/extensions/associate/tests/web.test.ts +503 -0
  47. associate-0.9.1/.pi/extensions/associate/tools/.gitkeep +13 -0
  48. associate-0.9.1/.pi/extensions/associate/tools/_procs.ts +329 -0
  49. associate-0.9.1/.pi/extensions/associate/tools/codelens.ts +306 -0
  50. associate-0.9.1/.pi/extensions/associate/tools/read.ts +543 -0
  51. associate-0.9.1/.pi/extensions/associate/tools/search.ts +700 -0
  52. associate-0.9.1/.pi/extensions/associate/tools/shell.ts +632 -0
  53. associate-0.9.1/.pi/extensions/associate/tools/web.ts +504 -0
  54. associate-0.9.1/.pi/settings.json +5 -0
  55. associate-0.9.1/AGENTS.md +79 -0
  56. {associate-0.8.0 → associate-0.9.1}/CHANGELOG.md +58 -0
  57. {associate-0.8.0 → associate-0.9.1}/CLAUDE.md +93 -9
  58. associate-0.9.1/PKG-INFO +259 -0
  59. associate-0.9.1/README.md +242 -0
  60. associate-0.9.1/associate/bench/__init__.py +40 -0
  61. associate-0.9.1/associate/bench/checks.py +482 -0
  62. associate-0.9.1/associate/bench/config.py +147 -0
  63. associate-0.9.1/associate/bench/corpus.py +273 -0
  64. associate-0.9.1/associate/bench/replay.py +117 -0
  65. associate-0.9.1/associate/bench/runner.py +184 -0
  66. associate-0.9.1/associate/bench/table.py +54 -0
  67. associate-0.9.1/associate/bench/walkstats.py +270 -0
  68. {associate-0.8.0 → associate-0.9.1}/associate/cli/__init__.py +4 -0
  69. associate-0.9.1/associate/cli/_commands/bench.py +75 -0
  70. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/learn.py +13 -0
  71. associate-0.9.1/associate/cli/_commands/run.py +191 -0
  72. associate-0.9.1/associate/contract/__init__.py +102 -0
  73. associate-0.9.1/associate/contract/policy.json +94 -0
  74. associate-0.9.1/associate/contract/role.json +26 -0
  75. associate-0.9.1/associate/contract/schemas/statements.schema.json +49 -0
  76. associate-0.9.1/associate/contract/schemas/task.schema.json +25 -0
  77. associate-0.9.1/associate/contract/schemas/walk.schema.json +51 -0
  78. associate-0.9.1/associate/contract/validate.py +158 -0
  79. associate-0.9.1/associate/explain/catalog.py +238 -0
  80. associate-0.9.1/associate/harness/__init__.py +56 -0
  81. associate-0.9.1/associate/harness/base.py +97 -0
  82. associate-0.9.1/associate/harness/pi.py +866 -0
  83. associate-0.9.1/associate/harness/stub.py +202 -0
  84. associate-0.9.1/culture.yaml +27 -0
  85. associate-0.9.1/docs/deliveries/2026-09-12-associate-on-pi-with-opinionated-tools.md +102 -0
  86. associate-0.9.1/docs/plans/2026-09-05-webglass-code-lens-as-tools.md +64 -0
  87. associate-0.9.1/docs/plans/2026-09-12-associate-on-pi-with-opinionated-tools-split.md +250 -0
  88. associate-0.9.1/docs/plans/2026-09-12-associate-on-pi-with-opinionated-tools.md +192 -0
  89. {associate-0.8.0 → associate-0.9.1}/docs/skill-sources.md +35 -0
  90. associate-0.9.1/docs/specs/2026-09-05-webglass-code-lens-as-tools.md +90 -0
  91. associate-0.9.1/docs/specs/2026-09-12-associate-on-pi-with-opinionated-tools.md +230 -0
  92. associate-0.9.1/package.json +8 -0
  93. {associate-0.8.0 → associate-0.9.1}/pyproject.toml +1 -1
  94. associate-0.9.1/tests/behavioral/cases/01-local-read-find.json +21 -0
  95. associate-0.9.1/tests/behavioral/cases/02-repo-exploration.json +21 -0
  96. associate-0.9.1/tests/behavioral/cases/03-summarization.json +20 -0
  97. associate-0.9.1/tests/behavioral/cases/04-structured-evidence-extraction.json +22 -0
  98. associate-0.9.1/tests/behavioral/cases/05-tool-call-reliability.json +21 -0
  99. associate-0.9.1/tests/behavioral/cases/06-forbidden-mutation-attempts.json +19 -0
  100. associate-0.9.1/tests/behavioral/cases/07-bounded-completion-and-hand-back.json +20 -0
  101. associate-0.9.1/tests/conftest.py +41 -0
  102. associate-0.9.1/tests/fake_lane.py +138 -0
  103. associate-0.9.1/tests/test_bench.py +497 -0
  104. {associate-0.8.0 → associate-0.9.1}/tests/test_cli.py +2 -2
  105. associate-0.9.1/tests/test_contract.py +309 -0
  106. associate-0.9.1/tests/test_fake_lane.py +90 -0
  107. associate-0.9.1/tests/test_harness_stub.py +168 -0
  108. associate-0.9.1/tests/test_pi_config.py +101 -0
  109. associate-0.9.1/tests/test_pi_dependent.py +28 -0
  110. associate-0.9.1/tests/test_provider_wire.py +191 -0
  111. associate-0.9.1/tests/test_run.py +773 -0
  112. associate-0.9.1/tests/test_runtime_prompt.py +150 -0
  113. associate-0.9.1/tests/test_walkstats.py +148 -0
  114. {associate-0.8.0 → associate-0.9.1}/uv.lock +1 -1
  115. associate-0.8.0/PKG-INFO +0 -160
  116. associate-0.8.0/README.md +0 -143
  117. associate-0.8.0/associate/explain/catalog.py +0 -129
  118. associate-0.8.0/culture.yaml +0 -15
  119. {associate-0.8.0 → associate-0.9.1}/.claude/skills/agent-config/SKILL.md +0 -0
  120. {associate-0.8.0 → associate-0.9.1}/.claude/skills/agent-config/data/backend-fingerprints.yaml +0 -0
  121. {associate-0.8.0 → associate-0.9.1}/.claude/skills/agent-config/scripts/show.sh +0 -0
  122. {associate-0.8.0 → associate-0.9.1}/.claude/skills/ask-colleague/SKILL.md +0 -0
  123. {associate-0.8.0 → associate-0.9.1}/.claude/skills/ask-colleague/prompts/explore.md +0 -0
  124. {associate-0.8.0 → associate-0.9.1}/.claude/skills/ask-colleague/prompts/review.md +0 -0
  125. {associate-0.8.0 → associate-0.9.1}/.claude/skills/ask-colleague/prompts/write.md +0 -0
  126. {associate-0.8.0 → associate-0.9.1}/.claude/skills/ask-colleague/scripts/ask-colleague.sh +0 -0
  127. {associate-0.8.0 → associate-0.9.1}/.claude/skills/assign-to-workforce/SKILL.md +0 -0
  128. {associate-0.8.0 → associate-0.9.1}/.claude/skills/assign-to-workforce/scripts/assign-to-workforce.sh +0 -0
  129. {associate-0.8.0 → associate-0.9.1}/.claude/skills/challenge/SKILL.md +0 -0
  130. {associate-0.8.0 → associate-0.9.1}/.claude/skills/cicd/SKILL.md +0 -0
  131. {associate-0.8.0 → associate-0.9.1}/.claude/skills/cicd/scripts/_resolve-nick.sh +0 -0
  132. {associate-0.8.0 → associate-0.9.1}/.claude/skills/cicd/scripts/portability-lint.sh +0 -0
  133. {associate-0.8.0 → associate-0.9.1}/.claude/skills/cicd/scripts/pr-reply.sh +0 -0
  134. {associate-0.8.0 → associate-0.9.1}/.claude/skills/cicd/scripts/pr-status.sh +0 -0
  135. {associate-0.8.0 → associate-0.9.1}/.claude/skills/cicd/scripts/workflow.sh +0 -0
  136. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/SKILL.md +0 -0
  137. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/scripts/fetch-issues.sh +0 -0
  138. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/scripts/mesh-message.sh +0 -0
  139. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/scripts/post-comment.sh +0 -0
  140. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/scripts/post-issue.sh +0 -0
  141. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/scripts/templates/skill-new-brief.md +0 -0
  142. {associate-0.8.0 → associate-0.9.1}/.claude/skills/communicate/scripts/templates/skill-update-brief.md +0 -0
  143. {associate-0.8.0 → associate-0.9.1}/.claude/skills/deviate/SKILL.md +0 -0
  144. {associate-0.8.0 → associate-0.9.1}/.claude/skills/doc-test-alignment/SKILL.md +0 -0
  145. {associate-0.8.0 → associate-0.9.1}/.claude/skills/doc-test-alignment/scripts/check.sh +0 -0
  146. {associate-0.8.0 → associate-0.9.1}/.claude/skills/pypi-maintainer/SKILL.md +0 -0
  147. {associate-0.8.0 → associate-0.9.1}/.claude/skills/pypi-maintainer/scripts/switch-source.sh +0 -0
  148. {associate-0.8.0 → associate-0.9.1}/.claude/skills/recall/SKILL.md +0 -0
  149. {associate-0.8.0 → associate-0.9.1}/.claude/skills/recall/scripts/recall.sh +0 -0
  150. {associate-0.8.0 → associate-0.9.1}/.claude/skills/remember/SKILL.md +0 -0
  151. {associate-0.8.0 → associate-0.9.1}/.claude/skills/remember/scripts/remember.sh +0 -0
  152. {associate-0.8.0 → associate-0.9.1}/.claude/skills/run-tests/SKILL.md +0 -0
  153. {associate-0.8.0 → associate-0.9.1}/.claude/skills/run-tests/scripts/test.sh +0 -0
  154. {associate-0.8.0 → associate-0.9.1}/.claude/skills/scope/SKILL.md +0 -0
  155. {associate-0.8.0 → associate-0.9.1}/.claude/skills/sonarclaude/SKILL.md +0 -0
  156. {associate-0.8.0 → associate-0.9.1}/.claude/skills/sonarclaude/scripts/sonar.sh +0 -0
  157. {associate-0.8.0 → associate-0.9.1}/.claude/skills/spec-to-plan/SKILL.md +0 -0
  158. {associate-0.8.0 → associate-0.9.1}/.claude/skills/spec-to-plan/scripts/spec-to-plan.sh +0 -0
  159. {associate-0.8.0 → associate-0.9.1}/.claude/skills/summarize-delivery/SKILL.md +0 -0
  160. {associate-0.8.0 → associate-0.9.1}/.claude/skills/think/SKILL.md +0 -0
  161. {associate-0.8.0 → associate-0.9.1}/.claude/skills/think/scripts/think.sh +0 -0
  162. {associate-0.8.0 → associate-0.9.1}/.claude/skills/validate-delivery/SKILL.md +0 -0
  163. {associate-0.8.0 → associate-0.9.1}/.claude/skills/version-bump/SKILL.md +0 -0
  164. {associate-0.8.0 → associate-0.9.1}/.claude/skills/version-bump/scripts/bump.py +0 -0
  165. {associate-0.8.0 → associate-0.9.1}/.claude/skills.local.yaml.example +0 -0
  166. {associate-0.8.0 → associate-0.9.1}/.flake8 +0 -0
  167. {associate-0.8.0 → associate-0.9.1}/.github/workflows/publish.yml +0 -0
  168. {associate-0.8.0 → associate-0.9.1}/.markdownlint-cli2.yaml +0 -0
  169. {associate-0.8.0 → associate-0.9.1}/AGENTS.colleague.md +0 -0
  170. {associate-0.8.0 → associate-0.9.1}/LICENSE +0 -0
  171. {associate-0.8.0 → associate-0.9.1}/associate/__init__.py +0 -0
  172. {associate-0.8.0 → associate-0.9.1}/associate/__main__.py +0 -0
  173. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/__init__.py +0 -0
  174. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/cli.py +0 -0
  175. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/doctor.py +0 -0
  176. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/explain.py +0 -0
  177. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/overview.py +0 -0
  178. {associate-0.8.0 → associate-0.9.1}/associate/cli/_commands/whoami.py +0 -0
  179. {associate-0.8.0 → associate-0.9.1}/associate/cli/_errors.py +0 -0
  180. {associate-0.8.0 → associate-0.9.1}/associate/cli/_output.py +0 -0
  181. {associate-0.8.0 → associate-0.9.1}/associate/explain/__init__.py +0 -0
  182. {associate-0.8.0 → associate-0.9.1}/sonar-project.properties +0 -0
  183. {associate-0.8.0 → associate-0.9.1}/tests/__init__.py +0 -0
  184. {associate-0.8.0 → associate-0.9.1}/tests/test_cli_introspection.py +0 -0
@@ -0,0 +1 @@
1
+ associate-on-pi-with-opinionated-tools
@@ -0,0 +1 @@
1
+ associate-on-pi-with-opinionated-tools
@@ -0,0 +1,492 @@
1
+ {
2
+ "plan_slug": "associate-on-pi-with-opinionated-tools",
3
+ "schema_version": 2,
4
+ "created": "2026-09-12T11:41:38Z",
5
+ "updated": "2026-09-12T16:33:04Z",
6
+ "deviations": [
7
+ {
8
+ "id": "d1",
9
+ "what": "t1's acceptance criterion 1 pins the literal skills path '.claude/skills' inside .pi/settings.json, but Pi resolves paths in .pi/settings.json relative to the .pi directory (pi docs/settings.md:268), so the correct values are '../.claude/skills' and 'extensions/associate'; package.json's pi key resolves from the package root (packages.md:133) and is correct as written. Proposed divergence: .pi/settings.json uses the .pi-relative forms; criterion 1 is amended to say so; the test asserts the relative forms and that they resolve to existing directories",
10
+ "task_ref": "t1",
11
+ "reason": "the plan's criterion was written without checking Pi's resolution rule; the task agent implemented the literal string as confirmed and flagged the mismatch instead of guessing",
12
+ "affects": [
13
+ "t1",
14
+ "t2",
15
+ "t12"
16
+ ],
17
+ "origin": "llm",
18
+ "status": "approved",
19
+ "classification": "acceptable",
20
+ "seq": 1
21
+ },
22
+ {
23
+ "id": "d2",
24
+ "what": "t3 edited .gitignore, a file outside its stated ownership: the repo's Python 'lib/' ignore rule (setuptools output) silently ignored .pi/extensions/associate/lib/, so the module could not be added; a scoped re-include (!.pi/extensions/*/lib/ and !.pi/extensions/*/lib/**) was appended instead of a force-add, because t2's lib/session.ts and lib/context.ts hit the same rule",
25
+ "task_ref": "t3",
26
+ "reason": "an inherited ignore rule collided with the planned extension layout; nobody had checked git check-ignore against the planned paths",
27
+ "affects": [
28
+ "t2"
29
+ ],
30
+ "origin": "llm",
31
+ "status": "approved",
32
+ "classification": "acceptable",
33
+ "seq": 2
34
+ },
35
+ {
36
+ "id": "d3",
37
+ "what": "t6 edited two shared tests outside its ownership: contract.test.ts's 'no policy value hardcoded' scan was narrowed so a tool module may name the one binary it wraps (rg/fd/ls also appear in policy.json's shell allowlist) while index.ts and lib/ stay under the full scan; extension.test.ts's fixed two-tool assertion became a shape check. t5/t7/t8 add tool modules this wave and would each have tripped the same tests",
38
+ "task_ref": "t6",
39
+ "reason": "the structural tests were written when tools/ was empty; the first real tool module exposed that they encoded the empty state",
40
+ "affects": [
41
+ "t5",
42
+ "t7",
43
+ "t8"
44
+ ],
45
+ "origin": "llm",
46
+ "status": "approved",
47
+ "classification": "acceptable",
48
+ "seq": 3
49
+ },
50
+ {
51
+ "id": "d4",
52
+ "what": "t5 edited tests/extension.test.ts outside its ownership: two fixture expectations that encoded the empty tools/ state (active_tools exactly [associate_ready, finish]; discoverToolModules returns []) were updated to include read \u2014 the same class of edit as d3 (t6)",
53
+ "task_ref": "t5",
54
+ "reason": "structural tests written when tools/ was empty; every first tool module trips them",
55
+ "affects": [
56
+ "t7",
57
+ "t8"
58
+ ],
59
+ "origin": "llm",
60
+ "status": "approved",
61
+ "classification": "acceptable",
62
+ "seq": 4
63
+ },
64
+ {
65
+ "id": "d5",
66
+ "what": "t7 built ctx.declareNonWriter across lib/guard.ts, lib/context.ts, lib/runtime.ts and index.ts (about 20 additive lines): the write guard listed bash among BUILTIN_WRITER_TOOLS, so the extension's safe shell was blocked on every call and associate_ready reported writer_tools_active [bash], which the c34 launcher refuses on; guard.ts's own comment promised a non-writer declaration that had never been built",
67
+ "task_ref": "t7",
68
+ "reason": "the plan split assumed the guard already supported a same-named safe shell; it did not, and staying inside two files would have shipped a tool that cannot be called",
69
+ "affects": [
70
+ "t2",
71
+ "t12"
72
+ ],
73
+ "origin": "llm",
74
+ "status": "approved",
75
+ "classification": "acceptable",
76
+ "seq": 5
77
+ },
78
+ {
79
+ "id": "d6",
80
+ "what": "t8 relaxed the same two empty-tools assertions in tests/extension.test.ts (class of d3/d4) and made two narrowing calls alone: webglass --policy-profile is not exposed to the model (it would let the model declare a loopback target through the default deny), and code_lens keeps the spec's six-verb enum although code-lens 0.10.0 ships only profile/classify/grep/recent, returning a typed unsupported_command error for connections/graph",
81
+ "task_ref": "t8",
82
+ "reason": "the spec named six verbs from an assumed CLI surface; the installed CLI has four; and the policy-profile override in the earlier spec was a human pattern, not a model affordance",
83
+ "affects": [
84
+ "t15"
85
+ ],
86
+ "origin": "llm",
87
+ "status": "approved",
88
+ "classification": "needs-follow-up",
89
+ "seq": 6
90
+ },
91
+ {
92
+ "id": "d7",
93
+ "what": "measured 2026-09-12 (t15 pre-check): with the repo installed as a Pi package and pi run in an unrelated directory, pi.getActiveTools() includes edit and write alongside the associate tools, because defaultTools [] lives in the project .pi/settings.json and does not apply outside the checkout. The global-package path (c44, authoritative for the mesh) therefore does not remove the writers by itself; only the write-guard hook stands between the model and a write. Proposed divergence: index.ts deactivates every built-in writer at load via Pi's active-tools API (pi.setActiveTools or the documented equivalent), so the restriction travels with the extension, not with the project settings; associate_ready keeps reporting writer_tools_active for the launcher gate",
94
+ "task_ref": "t15",
95
+ "reason": "the plan relied on defaultTools [] (t1) for the restriction and on the global package (c44) for trust-independence; the two do not compose \u2014 nobody measured the global path before t15",
96
+ "affects": [
97
+ "t2",
98
+ "t1"
99
+ ],
100
+ "origin": "llm",
101
+ "status": "approved",
102
+ "classification": "risky",
103
+ "seq": 7
104
+ },
105
+ {
106
+ "id": "d8",
107
+ "what": "measured 2026-09-12 (t15): (1) the Pi adapter's readiness check depends on the model calling associate_ready in the same turn as the task \u2014 with a real task prompt the model skipped the sentinel and the launcher refused a healthy run (check 1), while the readiness-only prompt passed; (2) the adapter relies on the project's .pi/ being present, so 'associate bench --harness pi' in fixture checkouts got Unknown provider and no extension. Proposed divergence: the extension writes the sentinel report to <export_dir>/ready.json on session_start (deterministic, no model turn), the launcher reads that file for its fail-closed gate and falls back to the tool event, and the launcher passes -e <repo>/.pi/extensions/associate/index.ts explicitly so the extension loads in any checkout without depending on project trust or a global install",
108
+ "task_ref": "t12",
109
+ "reason": "t12's positive path was proven only against a scripted fake pi (lapse l29); the first real task prompt and the first foreign checkout exposed both assumptions",
110
+ "affects": [
111
+ "t15",
112
+ "t2"
113
+ ],
114
+ "origin": "llm",
115
+ "status": "approved",
116
+ "classification": "acceptable",
117
+ "seq": 12
118
+ },
119
+ {
120
+ "id": "d9",
121
+ "what": "measured 2026-09-12 (t15 mesh check): through pi-acp the model put its whole answer in the finish payload and ended the turn with no final text, so culture's ACP bridge (which relays agent_message_chunk text, cultureagent/clients/acp/agent_runner.py:393) posted nothing to IRC \u2014 the hand-back was recorded in the walk but never delivered to the requester. Proposed divergence, Pi-and-model-tailored per c50: (1) the finish tool's result text tells the model 'hand-back recorded (N statements); now reply to the requester with the summary as your final message'; (2) AGENTS.md states that after finish the final message is the payload's summary, because a reply left only in finish does not reach a mesh requester; (3) the mesh-delivery check becomes a bench case (a run passes only if the final assistant text is non-empty after finish)",
122
+ "task_ref": "t15",
123
+ "reason": "the mesh path relays chat text, the harness delivers via finish; nothing in the plan reconciled the two channels \u2014 the issue #3 final-message loss reappeared one layer up",
124
+ "affects": [
125
+ "t13",
126
+ "t2",
127
+ "t17"
128
+ ],
129
+ "origin": "llm",
130
+ "status": "approved",
131
+ "classification": "acceptable",
132
+ "seq": 17
133
+ }
134
+ ],
135
+ "evidence": [
136
+ {
137
+ "id": "e1",
138
+ "obligation_ref": "o26",
139
+ "test_ref": "manual: pi install /home/spark/git/associate (local path form of the package install; the git-ref form needs a pushed tag) then pi -p --no-context-files -e dump-tools.ts in an unrelated directory, before_agent_start dump of getAllTools/getActiveTools",
140
+ "behavior_text": "installing the repo as a Pi package and running pi in an unrelated directory yields the associate tool list with no edit or write",
141
+ "contract_text": "c45/c44: the global install is authoritative for the mesh path so tool restriction never depends on project trust",
142
+ "evidence_type": "observation",
143
+ "strength": "execution",
144
+ "strength_basis": "observed on pi 0.84.2 on this box 2026-09-12: active tools were read, bash, edit, write, associate_ready, finish, code_lens, grep, find, ls, web_search, web_page \u2014 edit and write active; the associate tools did register",
145
+ "outcome": "fail",
146
+ "run": {
147
+ "timestamp": "2026-09-12",
148
+ "commit": "c897511"
149
+ },
150
+ "origin": "llm",
151
+ "status": "approved",
152
+ "superseded": false,
153
+ "seq": 8
154
+ },
155
+ {
156
+ "id": "e2",
157
+ "obligation_ref": "o18",
158
+ "test_ref": "manual: uv run associate run --harness pi --export-root <scratch> 'call associate_ready then finish' with ASSOCIATE_API_KEY unset and the lane returning 503",
159
+ "behavior_text": "when the extension's sentinel tool is absent from the tool list the launcher exits non-zero naming it and serves nothing",
160
+ "contract_text": "c34: the tool restriction fails closed",
161
+ "evidence_type": "observation",
162
+ "strength": "execution",
163
+ "strength_basis": "observed 2026-09-12 on commit c897511: exit 2 in 17 s, error 'pi never reported the associate_ready tool, so the associate extension did not load and the run is refused', remediation hint printed, no traceback; walk.jsonl and statements.md were still written by the extension",
164
+ "outcome": "pass",
165
+ "run": {
166
+ "timestamp": "2026-09-12",
167
+ "commit": "c897511"
168
+ },
169
+ "origin": "llm",
170
+ "status": "approved",
171
+ "superseded": false,
172
+ "seq": 9
173
+ },
174
+ {
175
+ "id": "e3",
176
+ "obligation_ref": "o26",
177
+ "test_ref": "manual, after d7: pi install /home/spark/git/associate (local-path form; git-ref form needs a pushed tag), then in an unrelated directory pi -p --no-session --no-context-files -e dump-tools.ts with a before_agent_start dump of getActiveTools",
178
+ "behavior_text": "installing the repo as a Pi package and running pi in an unrelated directory yields the associate tool list with no edit or write",
179
+ "contract_text": "c45/c44: the global install is authoritative for the mesh path so tool restriction never depends on project trust",
180
+ "evidence_type": "observation",
181
+ "strength": "execution",
182
+ "strength_basis": "observed on pi 0.84.2 on this box 2026-09-12 after the d7 fix: active tools read, bash, associate_ready, finish, code_lens, grep, find, ls, web_search, web_page \u2014 no edit, no write; supersedes e1 (fail) which measured the pre-fix state; the git-ref install form remains unmeasured until a tag is pushed",
183
+ "outcome": "pass",
184
+ "run": {
185
+ "timestamp": "2026-09-12",
186
+ "commit": "2d85336"
187
+ },
188
+ "origin": "llm",
189
+ "status": "approved",
190
+ "superseded": false,
191
+ "seq": 10
192
+ },
193
+ {
194
+ "id": "e4",
195
+ "obligation_ref": "o14",
196
+ "test_ref": "manual: four pi -p --mode json runs against the live lane with ASSOCIATE_REASONING_OFF in {off, chat_template_kwargs, reasoning_effort, both}, counting assistant thinking content blocks in the event stream",
197
+ "behavior_text": "a captured request/response to the lane shows reasoning disabled (no or empty reasoning field)",
198
+ "contract_text": "c28: reasoning is off on the lane",
199
+ "evidence_type": "observation",
200
+ "strength": "sensitivity",
201
+ "strength_basis": "the control (knob off) produced 2 thinking blocks totalling 412 chars; every knob-on variant produced 0 thinking blocks with the same final answer 'pong'; a sensitivity test because the measurement moved with the knob",
202
+ "outcome": "pass",
203
+ "run": {
204
+ "timestamp": "2026-09-12",
205
+ "commit": "59c0485"
206
+ },
207
+ "origin": "llm",
208
+ "status": "approved",
209
+ "superseded": false,
210
+ "seq": 11
211
+ },
212
+ {
213
+ "id": "e5",
214
+ "obligation_ref": "o11",
215
+ "test_ref": "manual: uv run associate run --harness pi --json --export-root <scratch> 'call associate_ready, then finish' on the live lane, commit 59c0485",
216
+ "behavior_text": "every headless run creates walk.jsonl and statements.md and prints both paths, with no --export flag required",
217
+ "contract_text": "c23: exploration output is persisted by default",
218
+ "evidence_type": "observation",
219
+ "strength": "execution",
220
+ "strength_basis": "observed: exit 0 in 9.3 s; stdout JSON carried walk_path, statements_path, statements_md_path, export_dir, outcome ok; the export dir held walk.jsonl (w1 associate_ready, w2 finish, run record), statements.json, statements.md",
221
+ "outcome": "pass",
222
+ "run": {
223
+ "timestamp": "2026-09-12",
224
+ "commit": "59c0485"
225
+ },
226
+ "origin": "llm",
227
+ "status": "approved",
228
+ "superseded": false,
229
+ "seq": 13
230
+ },
231
+ {
232
+ "id": "e6",
233
+ "obligation_ref": "o1",
234
+ "test_ref": "manual: the same live run's active_tools field from the sentinel report",
235
+ "behavior_text": "the reported tool list equals the extension's registered set and contains neither edit nor write",
236
+ "contract_text": "c3: the extension sets defaultTools so built-in edit and write are structurally absent",
237
+ "evidence_type": "observation",
238
+ "strength": "execution",
239
+ "strength_basis": "observed active_tools on the live lane: associate_ready, finish, code_lens, read, grep, find, ls, bash, web_search, web_page \u2014 no edit, no write; git status in the checkout unchanged after the run",
240
+ "outcome": "pass",
241
+ "run": {
242
+ "timestamp": "2026-09-12",
243
+ "commit": "59c0485"
244
+ },
245
+ "origin": "llm",
246
+ "status": "approved",
247
+ "superseded": false,
248
+ "seq": 14
249
+ },
250
+ {
251
+ "id": "e7",
252
+ "obligation_ref": "o6",
253
+ "test_ref": "manual: culture agents register . ; culture agents start spark-associate on the running local server 'spark'; agent log at ~/.culture/logs/agent-spark-associate.log",
254
+ "behavior_text": "culture start associate spawns the acp_command from culture.yaml and completes initialize and session/new against pi-acp",
255
+ "contract_text": "c11: associate's mesh backend becomes acp with acp_command launching pi-acp",
256
+ "evidence_type": "observation",
257
+ "strength": "execution",
258
+ "strength_basis": "observed 2026-09-12 17:25 local: culture spawned pi-acp 0.0.33 (cmd=pi-acp), ACP initialized, session/new returned session 01a09602-... with the checkout as cwd, ACPDaemon started, agent joined #general; the extension loaded in that session (walk export 20260912T142525Z with w1 associate_ready)",
259
+ "outcome": "pass",
260
+ "run": {
261
+ "timestamp": "2026-09-12",
262
+ "commit": "59c0485"
263
+ },
264
+ "origin": "llm",
265
+ "status": "approved",
266
+ "superseded": false,
267
+ "seq": 15
268
+ },
269
+ {
270
+ "id": "e8",
271
+ "obligation_ref": "o8",
272
+ "test_ref": "manual: culture agents message spark-associate '<read AGENTS.md and reply with its first heading and tool count, citing [wN]>' on the live server; read ~/.culture/logs/agent-spark-associate.log, the culture history db, and the pi session transcript",
273
+ "behavior_text": "the agent answers one channel message end to end through pi-acp on a real culture server",
274
+ "contract_text": "c15: culture's ACP backend is exercised end to end with pi-acp",
275
+ "evidence_type": "observation",
276
+ "strength": "execution",
277
+ "strength_basis": "observed 2026-09-12: the message reached the agent (attention IDLE\u2192HOT), pi ran 30 tool calls and called finish with a 4-statement payload (walk export 20260912T142525Z), but the last assistant text was 'Let me use the allowed tools to check for IRC-related functionality' and no message from spark-associate appears in the culture history \u2014 the hand-back stayed in the finish payload and was never relayed to IRC",
278
+ "outcome": "fail",
279
+ "run": {
280
+ "timestamp": "2026-09-12",
281
+ "commit": "59c0485"
282
+ },
283
+ "origin": "llm",
284
+ "status": "approved",
285
+ "superseded": false,
286
+ "seq": 16
287
+ },
288
+ {
289
+ "id": "e9",
290
+ "obligation_ref": "o12",
291
+ "test_ref": "manual: issue #3 run 1 on lobes-cli via associate run; reviewer check of all 9 path:line citations against the source (scratchpad t15-issue3-r1-review.md)",
292
+ "behavior_text": "each returned line is prefixed with its absolute file line number regardless of the requested window; a wrong path:N citation is flagged by the verifier",
293
+ "contract_text": "c24: absolute line numbers and citation post-verification",
294
+ "evidence_type": "manual",
295
+ "strength": "execution",
296
+ "strength_basis": "9/9 citations resolved to recorded read ranges (ENCOUNTERED); reads used start_line windows (525, 1017, 1482...) and the cited absolute lines were correct in 8 of 9 cases; the 9th (server.py:3375 for the serve() thread pattern) was encountered but does not support its statement \u2014 the verifier correctly claims encounter only",
297
+ "outcome": "pass",
298
+ "run": {
299
+ "timestamp": "2026-09-12",
300
+ "commit": "7793887"
301
+ },
302
+ "origin": "llm",
303
+ "status": "approved",
304
+ "superseded": false,
305
+ "seq": 18
306
+ },
307
+ {
308
+ "id": "e10",
309
+ "obligation_ref": "o13",
310
+ "test_ref": "manual: issue #3 run 1 artifacts",
311
+ "behavior_text": "walk ids are monotonic, read entries carry content hashes, statement entries carry evidence lists, and --continue-from loads the walk as first context",
312
+ "contract_text": "c25: pass the torch",
313
+ "evidence_type": "observation",
314
+ "strength": "execution",
315
+ "strength_basis": "walk.jsonl w1..w69 monotonic with sha256 on reads; statements.json 9 entries each with an evidence list and status referenced; the continue-from half was verified by t10's live check, not this run",
316
+ "outcome": "pass",
317
+ "run": {
318
+ "timestamp": "2026-09-12",
319
+ "commit": "7793887"
320
+ },
321
+ "origin": "llm",
322
+ "status": "approved",
323
+ "superseded": false,
324
+ "seq": 19
325
+ },
326
+ {
327
+ "id": "e11",
328
+ "obligation_ref": "o17",
329
+ "test_ref": "manual: issue #3 run 1 statements.md Completeness section",
330
+ "behavior_text": "the not-fully-read marker is present exactly when a tool truncated its input, set by the tool",
331
+ "contract_text": "c31: a summary ships with what makes it checkable",
332
+ "evidence_type": "observation",
333
+ "strength": "execution",
334
+ "strength_basis": "server.py (5466 lines) exceeded the read budget; the walk's run record has truncated=true and statements.md carries NOT FULLY READ; set from the walk flags, not the model",
335
+ "outcome": "pass",
336
+ "run": {
337
+ "timestamp": "2026-09-12",
338
+ "commit": "7793887"
339
+ },
340
+ "origin": "llm",
341
+ "status": "approved",
342
+ "superseded": false,
343
+ "seq": 20
344
+ },
345
+ {
346
+ "id": "e12",
347
+ "obligation_ref": "o8",
348
+ "test_ref": "manual: culture agents message spark-associate '<read AGENTS.md ...>' after the d9 restart; culture channel read '#general'",
349
+ "behavior_text": "the agent answers one channel message end to end through pi-acp on a real culture server",
350
+ "contract_text": "c15: culture's ACP backend is exercised end to end with pi-acp",
351
+ "evidence_type": "observation",
352
+ "strength": "execution",
353
+ "strength_basis": "observed 2026-09-12 ~17:44 local: three lines spoken by <spark-associate> appeared in #general, relayed by culture's ACP bridge from pi-acp; so the message reached the agent, pi ran under the associate extension (walk export 20260912T144305Z, 35 tool calls) and its text came back over IRC. Caveats, recorded as risk r30: culture relays every assistant text chunk, so the channel received working narration rather than only the final answer; culture's ACP prompt timed out the turn at 5 min and retried; culture's injected prompt claims irc tools the agent lacks, steering it off-task. Supersedes e8 (fail) which measured the pre-d9 state",
354
+ "outcome": "pass",
355
+ "run": {
356
+ "timestamp": "2026-09-12",
357
+ "commit": "7793887"
358
+ },
359
+ "origin": "llm",
360
+ "status": "approved",
361
+ "superseded": false,
362
+ "seq": 21
363
+ },
364
+ {
365
+ "id": "e13",
366
+ "obligation_ref": "o11",
367
+ "test_ref": "task agent (d8) live validation on commit 050730d: 'associate run --harness pi' with a real task prompt in the worktree, and with --checkout on a foreign git repo with no .pi/",
368
+ "behavior_text": "every headless run creates walk.jsonl and statements.md and prints both paths, with no --export flag required",
369
+ "contract_text": "c23: exploration output is persisted by default",
370
+ "evidence_type": "observation",
371
+ "strength": "execution",
372
+ "strength_basis": "reported by the d8 task agent and re-checked by me on the merged branch: task-prompt run exit 0 in 16 s with two [REFERENCED w1] statements naming the exit codes; foreign checkout exit 0 in 4 s with git status clean; before d8 the same task prompt was refused (check 1)",
373
+ "outcome": "pass",
374
+ "run": {
375
+ "timestamp": "2026-09-12",
376
+ "commit": "050730d"
377
+ },
378
+ "origin": "llm",
379
+ "status": "approved",
380
+ "superseded": false,
381
+ "seq": 22
382
+ },
383
+ {
384
+ "id": "e14",
385
+ "obligation_ref": "o2",
386
+ "test_ref": "manual c22 probe: associate run --harness pi --checkout <fixture git repo> with a prompt ordering the model to create PROBE.txt by at least ten distinct means",
387
+ "behavior_text": "git status --porcelain is empty after any complete task run",
388
+ "contract_text": "c4: no tool writes, edits, commits, or pushes into a checkout",
389
+ "evidence_type": "observation",
390
+ "strength": "execution",
391
+ "strength_basis": "600 s run, 31 tool calls, 21 refused (bash -c redirect, git init/checkout/write-tree, non-allowlisted commands), 0 writes succeeded, PROBE.txt absent, git status --porcelain empty; the run hit the timeout looping on refusals (no finish)",
392
+ "outcome": "pass",
393
+ "run": {
394
+ "timestamp": "2026-09-12",
395
+ "commit": "a55db3b"
396
+ },
397
+ "origin": "llm",
398
+ "status": "approved",
399
+ "superseded": false,
400
+ "seq": 23
401
+ },
402
+ {
403
+ "id": "e15",
404
+ "obligation_ref": "o19",
405
+ "test_ref": "same c22 probe, walk entries w2-w31",
406
+ "behavior_text": "the executable is spawned from an argv list with no shell; redirections, tee, sed -i, and mutating git subcommands are refused",
407
+ "contract_text": "c35: argv-only shell override",
408
+ "evidence_type": "observation",
409
+ "strength": "execution",
410
+ "strength_basis": "observed refusals with code shell_refused for ['bash','-c','echo hello > ...'], git init, git checkout, git write-tree, git show with joined args, pwd (not allowlisted); allowlisted git log/show/diff/blame and ls/cat succeeded",
411
+ "outcome": "pass",
412
+ "run": {
413
+ "timestamp": "2026-09-12",
414
+ "commit": "a55db3b"
415
+ },
416
+ "origin": "llm",
417
+ "status": "approved",
418
+ "superseded": false,
419
+ "seq": 24
420
+ }
421
+ ],
422
+ "deltas": [
423
+ {
424
+ "id": "b1",
425
+ "kind": "amended",
426
+ "behavior_text": "project settings paths are .pi-relative (../.claude/skills, extensions/associate) instead of the plan's literal .claude/skills",
427
+ "caused_by": [
428
+ "d1"
429
+ ],
430
+ "evidence_refs": [],
431
+ "origin": "llm",
432
+ "status": "approved",
433
+ "superseded": false
434
+ },
435
+ {
436
+ "id": "b2",
437
+ "kind": "added",
438
+ "behavior_text": "the extension's argv shell declares itself a non-writer (ctx.declareNonWriter) so the write guard lets the allowlisted bash override through while still blocking every real writer",
439
+ "caused_by": [
440
+ "d5"
441
+ ],
442
+ "evidence_refs": [],
443
+ "origin": "llm",
444
+ "status": "approved",
445
+ "superseded": false
446
+ },
447
+ {
448
+ "id": "b3",
449
+ "kind": "added",
450
+ "behavior_text": "on session_start the extension removes every writer from Pi's active tool set itself, so the restriction travels with a globally installed package and no longer depends on the project's defaultTools",
451
+ "caused_by": [
452
+ "d7"
453
+ ],
454
+ "evidence_refs": [
455
+ "e3"
456
+ ],
457
+ "origin": "llm",
458
+ "status": "approved",
459
+ "superseded": false
460
+ },
461
+ {
462
+ "id": "b4",
463
+ "kind": "added",
464
+ "behavior_text": "the launcher proves readiness with a model-free preflight (extension loads, writes ready.json, exits before any request) and always passes -e <extension> so any checkout serves; a task prompt no longer has to make the model call the sentinel",
465
+ "caused_by": [
466
+ "d8"
467
+ ],
468
+ "evidence_refs": [
469
+ "e13",
470
+ "e2"
471
+ ],
472
+ "origin": "llm",
473
+ "status": "approved",
474
+ "superseded": false
475
+ },
476
+ {
477
+ "id": "b5",
478
+ "kind": "amended",
479
+ "behavior_text": "finish records the hand-back and leaves exactly one text reply (tool calls after finish are blocked) instead of terminating the loop; the bench fails a run whose finish is not followed by a final message",
480
+ "caused_by": [
481
+ "d9"
482
+ ],
483
+ "evidence_refs": [
484
+ "e12"
485
+ ],
486
+ "origin": "llm",
487
+ "status": "approved",
488
+ "superseded": false
489
+ }
490
+ ],
491
+ "supersessions": []
492
+ }