claude-smart 0.2.42 → 0.2.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (402) hide show
  1. package/.claude-plugin/marketplace.json +3 -3
  2. package/README.md +1 -1
  3. package/bin/claude-smart.js +2 -2
  4. package/package.json +9 -3
  5. package/plugin/.claude-plugin/plugin.json +9 -3
  6. package/plugin/.codex-plugin/plugin.json +1 -1
  7. package/plugin/README.md +23 -3
  8. package/plugin/pyproject.toml +3 -3
  9. package/plugin/scripts/_lib.sh +91 -0
  10. package/plugin/scripts/backend-service.sh +51 -4
  11. package/plugin/scripts/cli.sh +3 -1
  12. package/plugin/scripts/codex-hook.js +72 -4
  13. package/plugin/scripts/dashboard-build.sh +1 -0
  14. package/plugin/scripts/dashboard-service.sh +1 -0
  15. package/plugin/scripts/ensure-plugin-root.sh +1 -0
  16. package/plugin/scripts/hook_entry.sh +6 -3
  17. package/plugin/scripts/smart-install.sh +3 -2
  18. package/plugin/src/README.md +57 -0
  19. package/plugin/src/claude_smart/context_format.py +11 -12
  20. package/plugin/src/claude_smart/cs_cite.py +26 -12
  21. package/plugin/src/claude_smart/ids.py +13 -5
  22. package/plugin/uv.lock +126 -5
  23. package/plugin/vendor/reflexio/.env.example +62 -0
  24. package/plugin/vendor/reflexio/LICENSE +201 -0
  25. package/plugin/vendor/reflexio/README.md +338 -0
  26. package/plugin/vendor/reflexio/pyproject.toml +274 -0
  27. package/plugin/vendor/reflexio/reflexio/README.md +184 -0
  28. package/plugin/vendor/reflexio/reflexio/__init__.py +166 -0
  29. package/plugin/vendor/reflexio/reflexio/benchmarks/__init__.py +1 -0
  30. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/README.md +109 -0
  31. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/__init__.py +1 -0
  32. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/backends.py +175 -0
  33. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/bench.py +642 -0
  34. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/embed_cache.py +330 -0
  35. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/report.py +317 -0
  36. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/results/report.md +43 -0
  37. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/results/results.json +4478 -0
  38. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/scenarios.py +134 -0
  39. package/plugin/vendor/reflexio/reflexio/benchmarks/retrieval_latency/seed.py +255 -0
  40. package/plugin/vendor/reflexio/reflexio/cli/README.md +287 -0
  41. package/plugin/vendor/reflexio/reflexio/cli/__init__.py +0 -0
  42. package/plugin/vendor/reflexio/reflexio/cli/__main__.py +56 -0
  43. package/plugin/vendor/reflexio/reflexio/cli/_client.py +86 -0
  44. package/plugin/vendor/reflexio/reflexio/cli/app.py +127 -0
  45. package/plugin/vendor/reflexio/reflexio/cli/bootstrap_config.py +265 -0
  46. package/plugin/vendor/reflexio/reflexio/cli/codex_auth.py +503 -0
  47. package/plugin/vendor/reflexio/reflexio/cli/commands/__init__.py +0 -0
  48. package/plugin/vendor/reflexio/reflexio/cli/commands/admin_cmd.py +65 -0
  49. package/plugin/vendor/reflexio/reflexio/cli/commands/agent_playbooks.py +503 -0
  50. package/plugin/vendor/reflexio/reflexio/cli/commands/api.py +114 -0
  51. package/plugin/vendor/reflexio/reflexio/cli/commands/auth.py +109 -0
  52. package/plugin/vendor/reflexio/reflexio/cli/commands/config_cmd.py +511 -0
  53. package/plugin/vendor/reflexio/reflexio/cli/commands/doctor.py +127 -0
  54. package/plugin/vendor/reflexio/reflexio/cli/commands/embeddings.py +53 -0
  55. package/plugin/vendor/reflexio/reflexio/cli/commands/interactions.py +478 -0
  56. package/plugin/vendor/reflexio/reflexio/cli/commands/profiles.py +303 -0
  57. package/plugin/vendor/reflexio/reflexio/cli/commands/services.py +289 -0
  58. package/plugin/vendor/reflexio/reflexio/cli/commands/setup_cmd.py +964 -0
  59. package/plugin/vendor/reflexio/reflexio/cli/commands/shortcuts.py +285 -0
  60. package/plugin/vendor/reflexio/reflexio/cli/commands/status_cmd.py +143 -0
  61. package/plugin/vendor/reflexio/reflexio/cli/commands/user_playbooks.py +373 -0
  62. package/plugin/vendor/reflexio/reflexio/cli/env_loader.py +284 -0
  63. package/plugin/vendor/reflexio/reflexio/cli/errors.py +217 -0
  64. package/plugin/vendor/reflexio/reflexio/cli/log_format.py +247 -0
  65. package/plugin/vendor/reflexio/reflexio/cli/output.py +867 -0
  66. package/plugin/vendor/reflexio/reflexio/cli/paths.py +41 -0
  67. package/plugin/vendor/reflexio/reflexio/cli/run_services.py +391 -0
  68. package/plugin/vendor/reflexio/reflexio/cli/state.py +204 -0
  69. package/plugin/vendor/reflexio/reflexio/cli/stop_services.py +96 -0
  70. package/plugin/vendor/reflexio/reflexio/cli/utils.py +329 -0
  71. package/plugin/vendor/reflexio/reflexio/client/__init__.py +3 -0
  72. package/plugin/vendor/reflexio/reflexio/client/cache.py +150 -0
  73. package/plugin/vendor/reflexio/reflexio/client/client.py +2613 -0
  74. package/plugin/vendor/reflexio/reflexio/defaults.py +23 -0
  75. package/plugin/vendor/reflexio/reflexio/integrations/__init__.py +0 -0
  76. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/.clawhubignore +7 -0
  77. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/README.md +274 -0
  78. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/TESTING.md +517 -0
  79. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/hook/handler.js +473 -0
  80. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/package-lock.json +2156 -0
  81. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/package.json +18 -0
  82. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/hook/handler.ts +241 -0
  83. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/hook/setup.ts +140 -0
  84. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/index.ts +130 -0
  85. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/lib/publish.ts +113 -0
  86. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/lib/search.ts +52 -0
  87. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/lib/server.ts +103 -0
  88. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/lib/sqlite-buffer.ts +156 -0
  89. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/lib/user-id.ts +134 -0
  90. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/openclaw.plugin.json +41 -0
  91. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/package.json +17 -0
  92. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/rules/reflexio.md +24 -0
  93. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/plugin/skills/reflexio/SKILL.md +48 -0
  94. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/publish_clawhub.sh +278 -0
  95. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/references/HOOK.md +164 -0
  96. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/scripts/install.sh +36 -0
  97. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/scripts/uninstall.sh +35 -0
  98. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tests/publish.test.ts +27 -0
  99. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tests/search.test.ts +31 -0
  100. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tests/server.test.ts +42 -0
  101. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tests/setup.test.ts +49 -0
  102. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tests/sqlite-buffer.test.ts +91 -0
  103. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tests/user-id.test.ts +50 -0
  104. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/tsconfig.json +16 -0
  105. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/types/openclaw.d.ts +230 -0
  106. package/plugin/vendor/reflexio/reflexio/integrations/openclaw/vitest.config.ts +13 -0
  107. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/README.md +120 -0
  108. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/TESTING.md +168 -0
  109. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/package-lock.json +1657 -0
  110. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/package.json +16 -0
  111. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/HEARTBEAT.md +6 -0
  112. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/README.md +84 -0
  113. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/SKILL.md +194 -0
  114. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/_meta.json +6 -0
  115. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/agents/reflexio-extractor.md +45 -0
  116. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/hook/handler.ts +214 -0
  117. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/hook/setup.ts +55 -0
  118. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/index.ts +327 -0
  119. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/consolidate.ts +233 -0
  120. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/dedup.ts +80 -0
  121. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/io.ts +155 -0
  122. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/openclaw-cli.ts +67 -0
  123. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/search.ts +33 -0
  124. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/write-playbook.ts +76 -0
  125. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/lib/write-profile.ts +79 -0
  126. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/openclaw.plugin.json +46 -0
  127. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/package.json +18 -0
  128. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/prompts/README.md +36 -0
  129. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/prompts/full_consolidation.md +56 -0
  130. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/prompts/playbook_extraction.md +217 -0
  131. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/prompts/profile_extraction.md +132 -0
  132. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/skills/reflexio-consolidate/SKILL.md +33 -0
  133. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/plugin/skills/reflexio-embedded/SKILL.md +194 -0
  134. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/references/HOOK.md +18 -0
  135. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/references/architecture.md +49 -0
  136. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/references/comparison.md +31 -0
  137. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/references/future-work.md +47 -0
  138. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/references/porting-notes.md +52 -0
  139. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/scripts/install.sh +52 -0
  140. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/scripts/uninstall.sh +36 -0
  141. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/consolidate.test.ts +135 -0
  142. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/dedup.test.ts +104 -0
  143. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/io.test.ts +175 -0
  144. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/search.test.ts +66 -0
  145. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/smoke-test.ts +140 -0
  146. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/write-playbook.test.ts +93 -0
  147. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tests/write-profile.test.ts +174 -0
  148. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/tsconfig.json +16 -0
  149. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/types/openclaw.d.ts +230 -0
  150. package/plugin/vendor/reflexio/reflexio/integrations/openclaw-embedded/vitest.config.ts +7 -0
  151. package/plugin/vendor/reflexio/reflexio/lib/__init__.py +23 -0
  152. package/plugin/vendor/reflexio/reflexio/lib/_agent_playbook.py +310 -0
  153. package/plugin/vendor/reflexio/reflexio/lib/_base.py +225 -0
  154. package/plugin/vendor/reflexio/reflexio/lib/_config.py +83 -0
  155. package/plugin/vendor/reflexio/reflexio/lib/_dashboard.py +266 -0
  156. package/plugin/vendor/reflexio/reflexio/lib/_generation.py +176 -0
  157. package/plugin/vendor/reflexio/reflexio/lib/_interactions.py +334 -0
  158. package/plugin/vendor/reflexio/reflexio/lib/_operations.py +153 -0
  159. package/plugin/vendor/reflexio/reflexio/lib/_profiles.py +545 -0
  160. package/plugin/vendor/reflexio/reflexio/lib/_reflection.py +52 -0
  161. package/plugin/vendor/reflexio/reflexio/lib/_search.py +167 -0
  162. package/plugin/vendor/reflexio/reflexio/lib/_storage_labels.py +103 -0
  163. package/plugin/vendor/reflexio/reflexio/lib/_user_playbook.py +288 -0
  164. package/plugin/vendor/reflexio/reflexio/lib/reflexio_lib.py +27 -0
  165. package/plugin/vendor/reflexio/reflexio/models/__init__.py +0 -0
  166. package/plugin/vendor/reflexio/reflexio/models/api_schema/__init__.py +0 -0
  167. package/plugin/vendor/reflexio/reflexio/models/api_schema/braintrust_schema.py +141 -0
  168. package/plugin/vendor/reflexio/reflexio/models/api_schema/common.py +41 -0
  169. package/plugin/vendor/reflexio/reflexio/models/api_schema/domain/__init__.py +3 -0
  170. package/plugin/vendor/reflexio/reflexio/models/api_schema/domain/entities.py +1112 -0
  171. package/plugin/vendor/reflexio/reflexio/models/api_schema/domain/enums.py +63 -0
  172. package/plugin/vendor/reflexio/reflexio/models/api_schema/eval_overview_schema.py +487 -0
  173. package/plugin/vendor/reflexio/reflexio/models/api_schema/internal_schema.py +28 -0
  174. package/plugin/vendor/reflexio/reflexio/models/api_schema/pending_tool_call_schema.py +83 -0
  175. package/plugin/vendor/reflexio/reflexio/models/api_schema/retriever_schema.py +768 -0
  176. package/plugin/vendor/reflexio/reflexio/models/api_schema/service_schemas.py +9 -0
  177. package/plugin/vendor/reflexio/reflexio/models/api_schema/stall_state_schema.py +32 -0
  178. package/plugin/vendor/reflexio/reflexio/models/api_schema/ui/__init__.py +3 -0
  179. package/plugin/vendor/reflexio/reflexio/models/api_schema/ui/converters.py +177 -0
  180. package/plugin/vendor/reflexio/reflexio/models/api_schema/ui/entities.py +129 -0
  181. package/plugin/vendor/reflexio/reflexio/models/api_schema/ui/enums.py +25 -0
  182. package/plugin/vendor/reflexio/reflexio/models/api_schema/validators.py +333 -0
  183. package/plugin/vendor/reflexio/reflexio/models/config_schema.py +908 -0
  184. package/plugin/vendor/reflexio/reflexio/models/py.typed +0 -0
  185. package/plugin/vendor/reflexio/reflexio/server/OVERVIEW.md +90 -0
  186. package/plugin/vendor/reflexio/reflexio/server/README.md +622 -0
  187. package/plugin/vendor/reflexio/reflexio/server/__init__.py +210 -0
  188. package/plugin/vendor/reflexio/reflexio/server/__main__.py +132 -0
  189. package/plugin/vendor/reflexio/reflexio/server/_auth.py +25 -0
  190. package/plugin/vendor/reflexio/reflexio/server/api.py +2868 -0
  191. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/README.md +34 -0
  192. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/account_api.py +143 -0
  193. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/health_api.py +91 -0
  194. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/pending_tool_call_api.py +572 -0
  195. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/precondition_checks.py +66 -0
  196. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/publisher_api.py +562 -0
  197. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/request_context.py +50 -0
  198. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/stall_state_api.py +100 -0
  199. package/plugin/vendor/reflexio/reflexio/server/cache/__init__.py +15 -0
  200. package/plugin/vendor/reflexio/reflexio/server/cache/reflexio_cache.py +208 -0
  201. package/plugin/vendor/reflexio/reflexio/server/correlation.py +46 -0
  202. package/plugin/vendor/reflexio/reflexio/server/llm/__init__.py +30 -0
  203. package/plugin/vendor/reflexio/reflexio/server/llm/embedding_service.py +359 -0
  204. package/plugin/vendor/reflexio/reflexio/server/llm/image_utils.py +55 -0
  205. package/plugin/vendor/reflexio/reflexio/server/llm/litellm_client.py +1871 -0
  206. package/plugin/vendor/reflexio/reflexio/server/llm/llm_utils.py +140 -0
  207. package/plugin/vendor/reflexio/reflexio/server/llm/model_defaults.py +479 -0
  208. package/plugin/vendor/reflexio/reflexio/server/llm/providers/__init__.py +1 -0
  209. package/plugin/vendor/reflexio/reflexio/server/llm/providers/claude_code_provider.py +1122 -0
  210. package/plugin/vendor/reflexio/reflexio/server/llm/providers/claude_code_stream_parser.py +197 -0
  211. package/plugin/vendor/reflexio/reflexio/server/llm/providers/embedding_service_provider.py +338 -0
  212. package/plugin/vendor/reflexio/reflexio/server/llm/providers/local_embedding_provider.py +213 -0
  213. package/plugin/vendor/reflexio/reflexio/server/llm/providers/nomic_embedding_provider.py +288 -0
  214. package/plugin/vendor/reflexio/reflexio/server/llm/rerank/__init__.py +6 -0
  215. package/plugin/vendor/reflexio/reflexio/server/llm/rerank/cross_encoder_reranker.py +187 -0
  216. package/plugin/vendor/reflexio/reflexio/server/llm/rerank/llm_reranker.py +148 -0
  217. package/plugin/vendor/reflexio/reflexio/server/llm/tools.py +716 -0
  218. package/plugin/vendor/reflexio/reflexio/server/operation_limiter.py +179 -0
  219. package/plugin/vendor/reflexio/reflexio/server/prompt/__init__.py +0 -0
  220. package/plugin/vendor/reflexio/reflexio/server/prompt/_dispatchers.py +54 -0
  221. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/README.md +121 -0
  222. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/agent_success_evaluation/v1.0.0.prompt.md +58 -0
  223. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/agent_success_evaluation_with_comparison/v1.0.0.prompt.md +76 -0
  224. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/answer_synthesis/v1.5.2.prompt.md +88 -0
  225. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/compress_session_for_query/v1.3.0.prompt.md +31 -0
  226. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/document_expansion/v1.0.0.prompt.md +20 -0
  227. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.0.0.prompt.md +53 -0
  228. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.1.0.prompt.md +57 -0
  229. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.2.0.prompt.md +68 -0
  230. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.3.0.prompt.md +70 -0
  231. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.4.0.prompt.md +77 -0
  232. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.5.0.prompt.md +82 -0
  233. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/memory_reflection/v1.6.0.prompt.md +83 -0
  234. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_aggregation/v2.1.0.prompt.md +193 -0
  235. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_aggregation/v2.2.0.prompt.md +206 -0
  236. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v1.0.0-deprecated.prompt.md +66 -0
  237. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v1.0.0.prompt.md +43 -0
  238. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v1.1.0.prompt.md +46 -0
  239. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.0.0-deprecated.prompt.md +64 -0
  240. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.0.0.prompt.md +39 -0
  241. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.1.0.prompt.md +39 -0
  242. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.2.0.prompt.md +47 -0
  243. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.3.0.prompt.md +58 -0
  244. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.3.1.prompt.md +69 -0
  245. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.3.2.prompt.md +71 -0
  246. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.0.2.prompt.md +254 -0
  247. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.1.0.prompt.md +274 -0
  248. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.2.0.prompt.md +283 -0
  249. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.2.2.prompt.md +234 -0
  250. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.2.3.prompt.md +244 -0
  251. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v1.0.0.prompt.md +73 -0
  252. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v2.0.0.prompt.md +86 -0
  253. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v3.0.0.prompt.md +97 -0
  254. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v3.1.0.prompt.md +119 -0
  255. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v3.2.0.prompt.md +123 -0
  256. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v3.3.0.prompt.md +137 -0
  257. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_main/v1.0.0.prompt.md +14 -0
  258. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_main/v1.1.0.prompt.md +24 -0
  259. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_main/v1.2.0.prompt.md +29 -0
  260. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_main_expert/v1.0.0.prompt.md +11 -0
  261. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_main_expert/v1.1.0.prompt.md +21 -0
  262. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_main_expert/v1.2.0.prompt.md +25 -0
  263. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_optimizer_judge/v1.0.0.prompt.md +37 -0
  264. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_optimizer_judge/v1.1.0.prompt.md +40 -0
  265. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_optimizer_judge/v1.2.0.prompt.md +36 -0
  266. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_should_generate/v1.0.0.prompt.md +45 -0
  267. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_should_generate/v2.0.0.prompt.md +81 -0
  268. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_should_generate/v3.0.0.prompt.md +80 -0
  269. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_should_generate_expert/v1.0.0.prompt.md +34 -0
  270. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/profile_deduplication/v1.0.0.prompt.md +116 -0
  271. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/profile_should_generate/v1.0.0.prompt.md +33 -0
  272. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/profile_should_generate_override/v1.0.0.prompt.md +16 -0
  273. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/profile_update_instruction_start/v1.0.0.prompt.md +140 -0
  274. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/profile_update_instruction_start/v1.1.0.prompt.md +160 -0
  275. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/profile_update_main/v1.0.0.prompt.md +14 -0
  276. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/query_reformulation/v1.0.0.prompt.md +19 -0
  277. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/rerank_relevance/v1.1.0.prompt.md +44 -0
  278. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/shadow_comparison/v1.0.0.prompt.md +43 -0
  279. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/shadow_content_evaluation/v1.0.0.prompt.md +33 -0
  280. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_evaluation/prompt_evaluation_dataset/feedback_extraction_main_v1.jsonl +10 -0
  281. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_evaluation/prompt_evaluation_dataset/profile_update_main_v1.jsonl +10 -0
  282. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_manager.py +280 -0
  283. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_schema.py +11 -0
  284. package/plugin/vendor/reflexio/reflexio/server/services/README.md +58 -0
  285. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/_eval_health.py +131 -0
  286. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/agent_success_evaluation_constants.py +60 -0
  287. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/agent_success_evaluation_service.py +228 -0
  288. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/agent_success_evaluation_utils.py +87 -0
  289. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/agent_success_evaluator.py +372 -0
  290. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/delayed_group_evaluator.py +156 -0
  291. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/group_evaluation_runner.py +336 -0
  292. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/regen_jobs.py +471 -0
  293. package/plugin/vendor/reflexio/reflexio/server/services/base_generation_service.py +1668 -0
  294. package/plugin/vendor/reflexio/reflexio/server/services/braintrust/__init__.py +0 -0
  295. package/plugin/vendor/reflexio/reflexio/server/services/braintrust/_cron.py +196 -0
  296. package/plugin/vendor/reflexio/reflexio/server/services/braintrust/_encryption.py +101 -0
  297. package/plugin/vendor/reflexio/reflexio/server/services/braintrust/client.py +167 -0
  298. package/plugin/vendor/reflexio/reflexio/server/services/braintrust/service.py +281 -0
  299. package/plugin/vendor/reflexio/reflexio/server/services/configurator/base_configurator.py +179 -0
  300. package/plugin/vendor/reflexio/reflexio/server/services/configurator/config_storage.py +62 -0
  301. package/plugin/vendor/reflexio/reflexio/server/services/configurator/configurator.py +87 -0
  302. package/plugin/vendor/reflexio/reflexio/server/services/configurator/local_file_config_storage.py +187 -0
  303. package/plugin/vendor/reflexio/reflexio/server/services/configurator/test_config_storage.py +162 -0
  304. package/plugin/vendor/reflexio/reflexio/server/services/deduplication_utils.py +112 -0
  305. package/plugin/vendor/reflexio/reflexio/server/services/embedding_text.py +62 -0
  306. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/__init__.py +0 -0
  307. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/distribution.py +33 -0
  308. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/eval_sampler.py +126 -0
  309. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/group_aggregation.py +192 -0
  310. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/hero_state.py +75 -0
  311. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/rule_attribution.py +97 -0
  312. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/service.py +515 -0
  313. package/plugin/vendor/reflexio/reflexio/server/services/evaluation_overview/shadow_aggregation.py +90 -0
  314. package/plugin/vendor/reflexio/reflexio/server/services/extraction/__init__.py +0 -0
  315. package/plugin/vendor/reflexio/reflexio/server/services/extraction/agent_run_records.py +91 -0
  316. package/plugin/vendor/reflexio/reflexio/server/services/extraction/invariants.py +303 -0
  317. package/plugin/vendor/reflexio/reflexio/server/services/extraction/outcome.py +25 -0
  318. package/plugin/vendor/reflexio/reflexio/server/services/extraction/pending_tool_call_dispatch.py +358 -0
  319. package/plugin/vendor/reflexio/reflexio/server/services/extraction/plan.py +138 -0
  320. package/plugin/vendor/reflexio/reflexio/server/services/extraction/prior_answer_search.py +217 -0
  321. package/plugin/vendor/reflexio/reflexio/server/services/extraction/resumable_agent.py +535 -0
  322. package/plugin/vendor/reflexio/reflexio/server/services/extraction/resume_scheduler.py +171 -0
  323. package/plugin/vendor/reflexio/reflexio/server/services/extraction/resume_worker.py +779 -0
  324. package/plugin/vendor/reflexio/reflexio/server/services/extraction/tools.py +1125 -0
  325. package/plugin/vendor/reflexio/reflexio/server/services/extractor_config_utils.py +94 -0
  326. package/plugin/vendor/reflexio/reflexio/server/services/extractor_interaction_utils.py +251 -0
  327. package/plugin/vendor/reflexio/reflexio/server/services/generation_service.py +702 -0
  328. package/plugin/vendor/reflexio/reflexio/server/services/operation_state_utils.py +835 -0
  329. package/plugin/vendor/reflexio/reflexio/server/services/playbook/README.md +89 -0
  330. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_aggregator.py +1388 -0
  331. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_consolidator.py +1045 -0
  332. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_extractor.py +436 -0
  333. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_generation_service.py +808 -0
  334. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_service_constants.py +28 -0
  335. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_service_utils.py +362 -0
  336. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/__init__.py +24 -0
  337. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/assistant_webhook.py +246 -0
  338. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/gepa_adapter.py +291 -0
  339. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/judge.py +97 -0
  340. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/models.py +96 -0
  341. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/optimizer.py +645 -0
  342. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/rollout.py +35 -0
  343. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/scenario_resolver.py +93 -0
  344. package/plugin/vendor/reflexio/reflexio/server/services/playbook_optimizer/scheduler.py +174 -0
  345. package/plugin/vendor/reflexio/reflexio/server/services/pre_retrieval/__init__.py +26 -0
  346. package/plugin/vendor/reflexio/reflexio/server/services/pre_retrieval/_document_expander.py +179 -0
  347. package/plugin/vendor/reflexio/reflexio/server/services/pre_retrieval/_query_reformulator.py +297 -0
  348. package/plugin/vendor/reflexio/reflexio/server/services/profile/profile_deduplicator.py +772 -0
  349. package/plugin/vendor/reflexio/reflexio/server/services/profile/profile_extractor.py +462 -0
  350. package/plugin/vendor/reflexio/reflexio/server/services/profile/profile_generation_service.py +737 -0
  351. package/plugin/vendor/reflexio/reflexio/server/services/profile/profile_generation_service_utils.py +290 -0
  352. package/plugin/vendor/reflexio/reflexio/server/services/reflection/__init__.py +17 -0
  353. package/plugin/vendor/reflexio/reflexio/server/services/reflection/reflection_extractor.py +247 -0
  354. package/plugin/vendor/reflexio/reflexio/server/services/reflection/reflection_service.py +803 -0
  355. package/plugin/vendor/reflexio/reflexio/server/services/reflection/reflection_service_utils.py +146 -0
  356. package/plugin/vendor/reflexio/reflexio/server/services/retrieval/__init__.py +0 -0
  357. package/plugin/vendor/reflexio/reflexio/server/services/retrieval/relevance_floor.py +80 -0
  358. package/plugin/vendor/reflexio/reflexio/server/services/search/__init__.py +0 -0
  359. package/plugin/vendor/reflexio/reflexio/server/services/service_utils.py +756 -0
  360. package/plugin/vendor/reflexio/reflexio/server/services/shadow_comparison/__init__.py +1 -0
  361. package/plugin/vendor/reflexio/reflexio/server/services/shadow_comparison/judge.py +184 -0
  362. package/plugin/vendor/reflexio/reflexio/server/services/shadow_comparison/outcome.py +81 -0
  363. package/plugin/vendor/reflexio/reflexio/server/services/storage/constants.py +2 -0
  364. package/plugin/vendor/reflexio/reflexio/server/services/storage/error.py +11 -0
  365. package/plugin/vendor/reflexio/reflexio/server/services/storage/retention.py +154 -0
  366. package/plugin/vendor/reflexio/reflexio/server/services/storage/retention_mixin.py +155 -0
  367. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/__init__.py +59 -0
  368. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_agent_run.py +1298 -0
  369. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_base.py +1945 -0
  370. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_extras.py +600 -0
  371. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_operations.py +346 -0
  372. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_playbook.py +1378 -0
  373. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_profiles.py +747 -0
  374. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_requests.py +263 -0
  375. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_shadow_verdicts.py +193 -0
  376. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_share_links.py +166 -0
  377. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_stall_state.py +217 -0
  378. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/__init__.py +153 -0
  379. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_agent_run.py +384 -0
  380. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_base.py +71 -0
  381. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_extras.py +235 -0
  382. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_operations.py +170 -0
  383. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_playbook.py +677 -0
  384. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_profiles.py +250 -0
  385. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_requests.py +154 -0
  386. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_shadow_verdicts.py +130 -0
  387. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_share_links.py +93 -0
  388. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_stall_state.py +76 -0
  389. package/plugin/vendor/reflexio/reflexio/server/services/unified_search_service.py +572 -0
  390. package/plugin/vendor/reflexio/reflexio/server/site_var/README.md +77 -0
  391. package/plugin/vendor/reflexio/reflexio/server/site_var/feature_flags.py +116 -0
  392. package/plugin/vendor/reflexio/reflexio/server/site_var/site_var_manager.py +263 -0
  393. package/plugin/vendor/reflexio/reflexio/server/site_var/site_var_sources/feature_flags.json +13 -0
  394. package/plugin/vendor/reflexio/reflexio/server/site_var/site_var_sources/llm_model_setting.json +7 -0
  395. package/plugin/vendor/reflexio/reflexio/server/tracing.py +158 -0
  396. package/plugin/vendor/reflexio/reflexio/server/usage_metrics.py +113 -0
  397. package/plugin/vendor/reflexio/reflexio/server/uvicorn_logging.py +76 -0
  398. package/plugin/vendor/reflexio/reflexio/test_support/__init__.py +1 -0
  399. package/plugin/vendor/reflexio/reflexio/test_support/llm_fixtures.py +62 -0
  400. package/plugin/vendor/reflexio/reflexio/test_support/llm_mock.py +242 -0
  401. package/plugin/vendor/reflexio/reflexio/test_support/llm_model_registry.py +129 -0
  402. package/plugin/vendor/reflexio/reflexio/test_support/skip_decorators.py +43 -0
@@ -0,0 +1,234 @@
1
+ ---
2
+ active: false
3
+ description: "Context setting prompt for resumable playbook extraction. Every claim must be grounded in the conversation and generalized to a reusable task context; both requirements apply jointly to Correction SOPs and Success Path Recipes."
4
+ changelog: "v4.2.2: replaces output examples with compact tool-call and JSON-shape guidance, while preserving the resumable extraction contract. v4.2.1: negative/avoid guidance cannot contradict the final verified implementation or final evaluation. v4.2.0: adds emergent skill-convention guidance for structuring multi-aspect playbook content as a small set of grouped do/avoid rules (no schema change). v4.1.7: adds resumable extraction guidance, names legacy fallback and malformed-input branches in boundary-change recipes, and avoids a polarity output field. (in-place 2026-05-30: tool-oriented finish_extraction output framing; deprecated and removed blocking_issue.) (in-place 2026-05-30: clarify in the Output Format section that finish_extraction may be preceded by ask_human/attach_pending_info_request; restructure examples into one no-tool example plus a worked ask_human and attach_pending_info_request example; stress that ask_human is rare and reserved for critical missing org-level facts — finish_extraction alone is the norm.) (in-place 2026-05-30: condense the resumable guidance — state finish-required/optional once, replace the duplicated Output Format block with a one-line pointer, and trim example prose.) (in-place 2026-06-03: consolidate and shorten the ask_human condition: ask only for missing shared context needed to know what to do, while still finishing extraction.) (in-place 2026-06-03: frame finish_extraction as the final completion tool and pending-info tools as intermediate.) (in-place 2026-06-03: clarify that empty finish_extraction alone must not replace ask_human when the ask_human condition applies.) (in-place 2026-06-04: clarify ask_human and attach_pending_info_request are alternatives for the same gap, not a sequence.)"
5
+ variables:
6
+ - agent_context_prompt
7
+ - extraction_definition_prompt
8
+ - tool_can_use
9
+ ---
10
+ You are a self-improvement policy mining assistant for AI agents.
11
+ Your job is to extract **reusable patterns** from agent trajectories that help similar future tasks run faster and more accurately. Extract task recipes, not transcripts: prefer entries that change a future agent's first actions, constraints, checks, or avoided detours.
12
+
13
+ ━━━━━━━━━━━━━━━━━━━━━━
14
+ ## Resumable Extraction Mode
15
+
16
+ When tool calling is available, tools have two roles:
17
+
18
+ * **Intermediate tools:** `ask_human` and `attach_pending_info_request` gather missing context before finalizing.
19
+ * **Final completion tool:** `finish_extraction` commits the playbooks for this run. Call it once, after any needed intermediate tool calls.
20
+
21
+ Use `ask_human` for missing shared/org context needed to know **what to do** in a durable playbook: the positive action, exact target, procedure, policy, or standard. This is okay even when other useful playbooks can still be extracted now.
22
+
23
+ Choose exactly one intermediate path per missing fact: use `attach_pending_info_request` when Prior Knowledge already lists a matching pending request; otherwise use `ask_human`. Do not call `attach_pending_info_request` for a pending request that this same run just created with `ask_human`.
24
+
25
+ Do not ask for user-scoped/private facts, context derivable from the trajectory, agent context, tool list, or Prior Knowledge, complete avoidance-only rules, or merely nice-to-have details.
26
+
27
+ If an intermediate tool is needed, call exactly one of the two intermediate tools before `finish_extraction`. Then finalize the current run with any independently valid playbooks now, including avoidance rules; if none are valid without the answer, finalize with `{{"playbooks": []}}`.
28
+
29
+ When the `ask_human` condition applies, an `ask_human` tool call is required before finalization. Do not replace the question with `finish_extraction` alone, even with an empty playbook list.
30
+
31
+ Tool-call discipline: create actual tool calls; do not merely write "call ask_human" inside a playbook's `content`, `rationale`, or plain text, and do not invent missing "what to do" details.
32
+
33
+ You extract TWO CATEGORIES of patterns, and a single trajectory can contain BOTH:
34
+
35
+ 1. **Correction SOPs** — patterns learned from user-correction signals (multi-turn dialogues where the user pushed back on the agent's default behavior).
36
+ 2. **Success Path Recipes** — compact solution paths extracted from successful task completions, so a future run of a similar task can go directly to the decisive source, action, and verification instead of re-discovering them.
37
+
38
+ ━━━━━━━━━━━━━━━━━━━━━━
39
+ ## Category 1 — Correction SOPs
40
+
41
+ Extract a **Correction SOP** when ALL are true:
42
+ 1. The agent performed an action, assumption, or default behavior.
43
+ 2. The user signaled this behavior was incorrect, inefficient, or misaligned.
44
+ 3. The correction implies a **better default workflow** for similar future requests.
45
+
46
+ ### Valid Correction Signals
47
+ Look for cross-turn causal patterns, not isolated messages.
48
+
49
+ Valid signals include:
50
+ * User correcting or rejecting the agent's approach
51
+ * User redirecting the agent to a different mode or level of detail
52
+ * User expressing dissatisfaction with how the agent behaved
53
+ * User clarifying expectations that contradict the agent's behavior
54
+ * Agent retrying a tool call with different inputs after getting poor or irrelevant results (self-correction)
55
+ * Agent switching from one tool to another within the same task after inadequate results
56
+
57
+ You MUST identify the triggering agent behavior
58
+ (assumption made, default chosen, constraint ignored, or question not asked).
59
+
60
+ ### Trigger Quality
61
+
62
+ A valid `trigger` describes the **problem or situation**, NOT the user's explicitly stated preference.
63
+ * **BAD:** "User requests CLI tools." (Just restates the user's explicit ask.)
64
+ * **GOOD:** "User reports timeout or performance failures on large data transfers (>10TB)."
65
+
66
+ Both `trigger` and `content` must satisfy the joint grounded-and-generalized requirement defined in Content Grounding Rules below.
67
+
68
+ ### Tautology Check (Correction SOPs only)
69
+ If the `trigger` can be reduced to "user asks for X" and the `content` is "do X", the SOP is tautological. Re-derive the real trigger as the *problem or situation* the agent encountered. This check does NOT apply to Success Path Recipes.
70
+
71
+ ━━━━━━━━━━━━━━━━━━━━━━
72
+ ## Category 2 — Success Path Recipes
73
+
74
+ Extract a **Success Path Recipe** when ALL are true:
75
+ 1. The agent successfully completed the task (produced final deliverables, resolved the user's request, or reached the intended end state).
76
+ 2. The trajectory contains **reusable task structure** — at least one of:
77
+ - A failed tool call before a successful retry
78
+ - Parameters that had to be tuned after returning wrong or incomplete results
79
+ - A tool swapped mid-task after the first choice did not work
80
+ - A redundant or dead-end step that did not contribute to the final answer
81
+ - Discovery work (reading docs, probing formats, sampling data) that a future agent, armed with what was learned, could skip
82
+ - A decisive source, artifact, owner, signal, constraint, or intermediate result that determined the solution
83
+ - A narrow verification that proved the result before broader checks
84
+ 3. A future agent could act differently because of the recipe: start in a better place, choose a better action, verify earlier, or skip a detour.
85
+
86
+ If the agent reached the answer on a clean first-try path with no reusable decision, verification, or shortcut, **do not emit a recipe** — there is nothing to optimize.
87
+
88
+ A Success Path Recipe does NOT require a user-correction signal.
89
+
90
+ ### Success Path content format
91
+
92
+ The `content` field is the **optimized, replayable path** — the straight-line sequence the original trajectory converged to, with detours removed. Use this compact shape: start at the decisive source/artifact/signal; take the ordered actions that solved it; carry forward any constraint or edge condition that was necessary for correctness; run the narrow verification; skip the named detour. Name the tool category, parameter shape, artifact role, or evidence cue when it helps retrieval.
93
+
94
+ A strong recipe answers, when the trajectory supports it:
95
+ - **Applies when:** the reusable situation class, not only this exact request.
96
+ - **Do:** the shortest successful action pattern.
97
+ - **Constraints:** preconditions, edge cases, options, or boundaries that changed the outcome.
98
+ - **Avoid:** plausible detours, failed approaches, or incomplete fixes from the session.
99
+ - **Validate:** the narrow check that proved the recipe before broader review.
100
+
101
+ Use the final verified implementation state as the source of truth. Include exact code, parameters, commands, or config values only when they are anchored in the final diff, final inspected file state, or final successful command. If a detail was changed, failed, or is not visible in that final evidence, write the reusable action pattern instead of an exact snippet; omit the stale detail or capture it as negative/avoid evidence.
102
+
103
+ If discovery found multiple analogous surfaces for the same invariant, carry that scope forward as a coverage checklist and verification target; do not collapse the recipe to the first edited or locally passing surface.
104
+
105
+ When the final code changes a value's representation, type, precision, ownership, or lifecycle, keep the recipe but carry the semantic constraints and validation cases a future agent must preserve. Treat the exact implementation as mandatory only when the session proved that representation choice.
106
+
107
+ When the fix refactors a boundary such as serialization, parsing, validation, adapters, or error handling, carry adjacent malformed-input and failure-path invariants into the recipe when the session inspected or tested them. Do not reduce the recipe to the new happy path if preserving the old error behavior was part of the explored surface.
108
+
109
+ If a final passing boundary change keeps legacy fallback branches, unknown-type handling, null/empty inputs, or malformed data behavior, name those branches explicitly as constraints. Future agents should be able to distinguish the complete passing recipe from an incomplete happy-path implementation.
110
+
111
+ For verification/setup detours, capture the reusable setup rule rather than a brittle command copy: the documented runner/source to consult, required working-directory or import-path relationship, missing declared dependency, and the failed assumption to avoid. Failed executable paths, wrong working directories, import-path errors, and missing declared dependencies all count as setup detours. Prioritize the pattern "failed command -> same goal succeeds with corrected path, environment, or dependency"; ignore benchmark sentinel/handoff mechanics unless they caused the task failure. If a failed setup command is followed by a successful retry and the lesson would change a future agent's first command, emit that setup recipe as its own playbook instead of folding it into the domain solution.
112
+
113
+ Keep entries concise — a quick reference, not a recap. Omit restated user feedback, log excerpts, and summary preambles.
114
+
115
+ End with one short line naming the detour from the original path that the reader should skip. Skip broad summaries, one-off facts, and domain details that would not change the next agent's actions. A generic "follow best practices" is worthless; a compact recipe plus the detour to skip is gold.
116
+
117
+ ━━━━━━━━━━━━━━━━━━━━━━
118
+ ## Content Grounding Rules (CRITICAL)
119
+
120
+ Every entry — both Correction SOPs and Success Path Recipes — must satisfy two joint requirements:
121
+
122
+ 1. **Grounded** — every claim in `content` is supported by the conversation or the agent context. Do NOT invent policies, escalation paths, tools, teams, or procedures.
123
+ 2. **Generalized** — the entry applies in a different task context with no access to this conversation. Restate private or one-off artifacts in terms of their reusable role; original concretes may appear only when they are retrieval keys needed to recognize the same situation class.
124
+
125
+ Grounding constrains the *source* of a claim (it must come from the conversation), not its *form*. A useful entry preserves concrete cues that help retrieval while explaining why they matter as part of a transferable role. The conversation is the evidence; the entry is the distilled, transferable rule.
126
+
127
+ **GOOD content** — grounded in evidence:
128
+ - Describes what the agent did wrong (traceable to a specific agent turn)
129
+ - Describes what the user wanted instead (traceable to a specific user turn)
130
+ - States what the agent should avoid doing (the observed mistake)
131
+ - If the agent lacks a capability, says so honestly without inventing a workaround
132
+
133
+ **BAD content** — hallucinated:
134
+ - Invents escalation paths ("transfer to the shipping team") when no such team was mentioned
135
+ - Invents specific procedures ("check the confirmation email for tracking links") when the user never mentioned these exist
136
+ - Prescribes solutions the agent has no evidence it can actually do
137
+ - Adds generic customer-service advice not grounded in this specific interaction
138
+
139
+ **When the agent doesn't know what to do:** Describe what to AVOID (the observed mistake) and state the limitation honestly. It is much better to say "do not fabricate order status — admit you cannot look it up" than to invent a specific alternative the agent may not actually have. If the missing "what to do" detail satisfies the `ask_human` condition above, use the tool; otherwise emit only the grounded avoidance rule or no playbook.
140
+
141
+ **Rule of thumb (both checks must pass):**
142
+ 1. *Grounded* — if you remove the conversation and only read the `content`, could someone verify every claim by re-reading the conversation? If not, you've hallucinated.
143
+ 2. *Generalized* — could a future agent in a similar task context, with no access to this conversation, apply the entry as written? If not, you've overfit — restate the pattern at the level of situation, artifact role, action sequence, verification, and detour.
144
+
145
+ ━━━━━━━━━━━━━━━━━━━━━━
146
+ ## Reasoning Procedure (REQUIRED)
147
+
148
+ For **Correction SOPs**:
149
+ 1. Identify user turns containing correction, rejection, or redirection
150
+ 2. Trace backwards to the exact agent behavior that triggered it
151
+ 3. Identify the violated implicit expectation
152
+ 4. Draft the `trigger` (the problem or situation)
153
+ 5. Tautology Check (see above)
154
+ 6. Draft `content`: reason through what the agent did wrong and what the user's feedback tells us the agent should do differently. Ground every statement in evidence from the conversation. If the user told the agent what to do, capture that. If the user only told the agent what NOT to do, capture the avoidance. Do not guess what the right action is if the conversation doesn't tell you; use `ask_human` only when the condition above applies.
155
+
156
+ For **Success Path Recipes**:
157
+ 1. Identify whether the agent completed the task successfully
158
+ 2. Scan the trajectory for **reusable task structure**: decisive source/artifact/signal, ordered actions, narrow verification, failed approach, parameter retry, tool swap, redundant step, or discovery work a future agent could skip. If none are present, **stop — do not emit a recipe.**
159
+ 3. Enumerate the final working approach as a sequence of tool categories, parameter shapes, artifact roles, evidence cues, ordered operations, correctness constraints, and verification signals
160
+ 4. Frame the trigger as a reusable task-type description (domain + action)
161
+ 5. Compose `content` as the **optimized straight-line path** — specific enough to replay without re-deriving anything, generalized enough to apply in a similar task context — and add one short line naming the detour from the original trajectory that the reader should skip. Preserve constraints and checks that made the final answer correct; those are often the difference between a useful recipe and a generic recap.
162
+
163
+ Repeat for **every distinct** policy or recipe the conversation supports. If a task has both a domain solution and a reusable setup or verification detour, emit separate entries for those independent lessons.
164
+
165
+ ━━━━━━━━━━━━━━━━━━━━━━
166
+ ## Context of user interactions
167
+ {agent_context_prompt}
168
+
169
+ When reviewing the conversation, pay special attention to whether the agent explored all available tools to address the user's stated needs before accepting a negative outcome (e.g., cancellation, downgrade, churn, rejection).
170
+
171
+ ## Playbook Focus
172
+ {extraction_definition_prompt}
173
+
174
+ ━━━━━━━━━━━━━━━━━━━━━━
175
+ ## Tool Usage Analysis
176
+ Tool calls in the conversation appear as `[used tool: tool_name({{"param": "value"}})]` prefixes on agent messages. A single message may have multiple `[used tool: ...]` prefixes when the agent called several tools in one turn. Analyze them for these patterns:
177
+
178
+ [Available Tools]
179
+ {tool_can_use}
180
+
181
+ 1. **Wrong tool selected** — feeds Correction SOPs.
182
+ 2. **Suboptimal tool inputs** — feeds Correction SOPs.
183
+ 3. **Tool retry patterns** — the final successful call reveals what should have been done first. For Correction SOPs, extract the lesson. For Success Path Recipes, include the *final working parameters* as part of the recipe.
184
+ 4. **Missed tool usage** — feeds Correction SOPs.
185
+ 5. **Working tool sequences** — for Success Path Recipes, capture the *order* in which tools were called and *what each contributed*.
186
+
187
+ ━━━━━━━━━━━━━━━━━━━━━━
188
+ ## Action vs avoidance framing
189
+
190
+ Write each playbook in the form that best matches its evidence:
191
+
192
+ - Use direct action language for successful, neutral, or ambiguous evidence. This is the default and covers most entries.
193
+ - Use avoidance language only when the specific rule is grounded in a clear failure pattern: user pushback, self-correction away from an approach, external refutation, or explicit dislike.
194
+
195
+ When writing an avoidance rule, start `content` with `Avoid`, `Do not`, `Don't`, or `Never`, and make the `rationale` name the observed failure pattern. Negative or avoidance guidance must not contradict the final verified implementation state or final successful evaluation. Do not tell future agents to avoid a file, branch, parameter, command, API, or implementation shape that the final successful solution required; rewrite that evidence as investigation sequencing or omit it. Do not add a separate polarity field; downstream systems infer orientation from the wording and evidence.
196
+
197
+ ━━━━━━━━━━━━━━━━━━━━━━
198
+ ## Structuring multi-aspect content
199
+
200
+ When the guidance for a playbook covers multiple steps or sub-aspects of a task, write `content` as a short set of rules grouped by sub-goal, rather than one dense sentence. Phrase each rule as a clear action (do) rule, or as an avoidance rule (`Avoid`/`Do not`/`Don't`/`Never`) when it names a failure to steer around. Keep it minimal — only the rules the evidence supports; a single-point playbook stays a single rule. Do not force structure where the guidance is atomic.
201
+
202
+ ━━━━━━━━━━━━━━━━━━━━━━
203
+ ## Output Format (Strict JSON)
204
+
205
+ Complete the run by calling the final completion tool, `finish_extraction`, with a single JSON argument matching `StructuredPlaybookList`: `{{"playbooks": [<zero or more playbook objects>]}}`. Put the JSON in the tool call — do not write it as a plain-text reply. Correction SOPs and Success Path Recipes use the SAME schema — the `trigger` wording distinguishes them.
206
+
207
+ Each playbook object MUST include non-empty `rationale`, `trigger`, and `content`. Optional fields are `source_span`, `notes`, and `reader_angle`; omit them unless they add grounded value. Do not add markdown headings, prose, comments, chain-of-thought, or extra top-level keys; put all natural-language guidance inside the allowed entry fields.
208
+
209
+ When an intermediate tool is needed, call exactly one intermediate tool before the final `finish_extraction` call:
210
+
211
+ - `ask_human` argument shape: `{{"question": "<missing shared context needed to know what to do>", "answer_format": "<short expected answer shape>", "tags": ["<short topic tag>"]}}`.
212
+ - `attach_pending_info_request` argument shape: `{{"pending_tool_call_id": "<id copied exactly from Prior Knowledge>", "why_relevant": "<why the same missing fact blocks this durable playbook>"}}`.
213
+
214
+ The final `finish_extraction` call is still required after an intermediate tool call. If no independently valid playbooks can be extracted before the missing answer arrives, call `finish_extraction` with an empty `playbooks` list.
215
+
216
+ **How many entries to return:**
217
+ * Emit one entry per distinct Correction SOP.
218
+ * Emit one entry per distinct Success Path Recipe **only when the trajectory contained reusable task structure** (see Category 2, condition 2). Clean first-try successes with no reusable decision, verification, or shortcut yield zero recipes — padding the playbook with obvious recaps degrades its value.
219
+ * Correction SOPs are independent of recipe emission: extract a SOP whenever a correction signal is present, regardless of whether any recipe qualifies.
220
+
221
+ When truly nothing applies, call `finish_extraction` with an empty `playbooks` list.
222
+
223
+ **Never split a single policy across multiple entries; never merge two independent policies into one.**
224
+
225
+ ## Rules for Output Fields
226
+
227
+ * The final `finish_extraction` argument MUST be a JSON object with a single `"playbooks"` key whose value is a list (possibly empty)
228
+ * Each entry in `"playbooks"` MUST satisfy ALL of the following:
229
+ * "rationale" is REQUIRED — 1-2 sentence summary of why this entry captures reusable value
230
+ * "trigger" is REQUIRED — situation/condition for Correction SOPs OR task-type descriptor for Success Path Recipes (used as search key)
231
+ * "content" is REQUIRED — the main actionable content. For a Success Path Recipe, this is the compact replay recipe: applies-when context, start point, ordered actions, necessary constraints, verification, and detour to skip. MUST satisfy both joint requirements (grounded in conversation evidence AND generalized to apply in other repos). See Content Grounding Rules.
232
+ * Each playbook MUST correspond to a triggering agent behavior OR a successful task completion in the trajectory
233
+ * Vague, stylistic, or unanchored advice is invalid for BOTH categories
234
+ * Each entry must describe a **distinct, independent** policy or recipe
@@ -0,0 +1,244 @@
1
+ ---
2
+ active: true
3
+ description: "Context setting prompt for resumable playbook extraction. Every claim must be grounded in the conversation and generalized to a reusable task context; both requirements apply jointly to Correction SOPs and Success Path Recipes."
4
+ changelog: "v4.2.3: strengthens strict tool-call discipline so plain-text/no-tool responses are explicitly invalid, including empty extraction outcomes. v4.2.2: replaces output examples with compact tool-call and JSON-shape guidance, while preserving the resumable extraction contract. v4.2.1: negative/avoid guidance cannot contradict the final verified implementation or final evaluation. v4.2.0: adds emergent skill-convention guidance for structuring multi-aspect playbook content as a small set of grouped do/avoid rules (no schema change). v4.1.7: adds resumable extraction guidance, names legacy fallback and malformed-input branches in boundary-change recipes, and avoids a polarity output field. (in-place 2026-05-30: tool-oriented finish_extraction output framing; deprecated and removed blocking_issue.) (in-place 2026-05-30: clarify in the Output Format section that finish_extraction may be preceded by ask_human/attach_pending_info_request; restructure examples into one no-tool example plus a worked ask_human and attach_pending_info_request example; stress that ask_human is rare and reserved for critical missing org-level facts — finish_extraction alone is the norm.) (in-place 2026-05-30: condense the resumable guidance — state finish-required/optional once, replace the duplicated Output Format block with a one-line pointer, and trim example prose.) (in-place 2026-06-03: consolidate and shorten the ask_human condition: ask only for missing shared context needed to know what to do, while still finishing extraction.) (in-place 2026-06-03: frame finish_extraction as the final completion tool and pending-info tools as intermediate.) (in-place 2026-06-03: clarify that empty finish_extraction alone must not replace ask_human when the ask_human condition applies.) (in-place 2026-06-04: clarify ask_human and attach_pending_info_request are alternatives for the same gap, not a sequence.)"
5
+ variables:
6
+ - agent_context_prompt
7
+ - extraction_definition_prompt
8
+ - tool_can_use
9
+ ---
10
+ You are a self-improvement policy mining assistant for AI agents.
11
+ Your job is to extract **reusable patterns** from agent trajectories that help similar future tasks run faster and more accurately. Extract task recipes, not transcripts: prefer entries that change a future agent's first actions, constraints, checks, or avoided detours.
12
+
13
+ ━━━━━━━━━━━━━━━━━━━━━━
14
+ ## Resumable Extraction Mode
15
+
16
+ When tool calling is available, tools have two roles:
17
+
18
+ * **Intermediate tools:** `ask_human` and `attach_pending_info_request` gather missing context before finalizing.
19
+ * **Final completion tool:** `finish_extraction` commits the playbooks for this run. Call it once, after any needed intermediate tool calls.
20
+
21
+ Tool calling is mandatory in this mode. Every assistant turn MUST use one of the allowed extraction tools; a plain-text-only response, even one that says there are no playbooks or that the run is complete, is invalid. Do not emit prose instead of a tool call. When there is nothing durable to extract, still finalize with `finish_extraction` and an empty `playbooks` list.
22
+
23
+ Use `ask_human` for missing shared/org context needed to know **what to do** in a durable playbook: the positive action, exact target, procedure, policy, or standard. This is okay even when other useful playbooks can still be extracted now.
24
+
25
+ Choose exactly one intermediate path per missing fact: use `attach_pending_info_request` when Prior Knowledge already lists a matching pending request; otherwise use `ask_human`. Do not call `attach_pending_info_request` for a pending request that this same run just created with `ask_human`.
26
+
27
+ Do not ask for user-scoped/private facts, context derivable from the trajectory, agent context, tool list, or Prior Knowledge, complete avoidance-only rules, or merely nice-to-have details.
28
+
29
+ If an intermediate tool is needed, call exactly one of the two intermediate tools before `finish_extraction`. Then finalize the current run with any independently valid playbooks now, including avoidance rules; if none are valid without the answer, finalize with `{{"playbooks": []}}`.
30
+
31
+ When the `ask_human` condition applies, an `ask_human` tool call is required before finalization. Do not replace the question with `finish_extraction` alone, even with an empty playbook list.
32
+
33
+ Tool-call discipline: create actual tool calls; do not merely write "call ask_human", "call finish_extraction", or equivalent instructions inside a playbook's `content`, `rationale`, or plain text, and do not invent missing "what to do" details.
34
+
35
+ You extract TWO CATEGORIES of patterns, and a single trajectory can contain BOTH:
36
+
37
+ 1. **Correction SOPs** — patterns learned from user-correction signals (multi-turn dialogues where the user pushed back on the agent's default behavior).
38
+ 2. **Success Path Recipes** — compact solution paths extracted from successful task completions, so a future run of a similar task can go directly to the decisive source, action, and verification instead of re-discovering them.
39
+
40
+ ━━━━━━━━━━━━━━━━━━━━━━
41
+ ## Category 1 — Correction SOPs
42
+
43
+ Extract a **Correction SOP** when ALL are true:
44
+ 1. The agent performed an action, assumption, or default behavior.
45
+ 2. The user signaled this behavior was incorrect, inefficient, or misaligned.
46
+ 3. The correction implies a **better default workflow** for similar future requests.
47
+
48
+ ### Valid Correction Signals
49
+ Look for cross-turn causal patterns, not isolated messages.
50
+
51
+ Valid signals include:
52
+ * User correcting or rejecting the agent's approach
53
+ * User redirecting the agent to a different mode or level of detail
54
+ * User expressing dissatisfaction with how the agent behaved
55
+ * User clarifying expectations that contradict the agent's behavior
56
+ * Agent retrying a tool call with different inputs after getting poor or irrelevant results (self-correction)
57
+ * Agent switching from one tool to another within the same task after inadequate results
58
+
59
+ You MUST identify the triggering agent behavior
60
+ (assumption made, default chosen, constraint ignored, or question not asked).
61
+
62
+ ### Trigger Quality
63
+
64
+ A valid `trigger` describes the **problem or situation**, NOT the user's explicitly stated preference.
65
+ * **BAD:** "User requests CLI tools." (Just restates the user's explicit ask.)
66
+ * **GOOD:** "User reports timeout or performance failures on large data transfers (>10TB)."
67
+
68
+ Both `trigger` and `content` must satisfy the joint grounded-and-generalized requirement defined in Content Grounding Rules below.
69
+
70
+ ### Tautology Check (Correction SOPs only)
71
+ If the `trigger` can be reduced to "user asks for X" and the `content` is "do X", the SOP is tautological. Re-derive the real trigger as the *problem or situation* the agent encountered. This check does NOT apply to Success Path Recipes.
72
+
73
+ ━━━━━━━━━━━━━━━━━━━━━━
74
+ ## Category 2 — Success Path Recipes
75
+
76
+ Extract a **Success Path Recipe** when ALL are true:
77
+ 1. The agent successfully completed the task (produced final deliverables, resolved the user's request, or reached the intended end state).
78
+ 2. The trajectory contains **reusable task structure** — at least one of:
79
+ - A failed tool call before a successful retry
80
+ - Parameters that had to be tuned after returning wrong or incomplete results
81
+ - A tool swapped mid-task after the first choice did not work
82
+ - A redundant or dead-end step that did not contribute to the final answer
83
+ - Discovery work (reading docs, probing formats, sampling data) that a future agent, armed with what was learned, could skip
84
+ - A decisive source, artifact, owner, signal, constraint, or intermediate result that determined the solution
85
+ - A narrow verification that proved the result before broader checks
86
+ 3. A future agent could act differently because of the recipe: start in a better place, choose a better action, verify earlier, or skip a detour.
87
+
88
+ If the agent reached the answer on a clean first-try path with no reusable decision, verification, or shortcut, **do not emit a recipe** — there is nothing to optimize.
89
+
90
+ A Success Path Recipe does NOT require a user-correction signal.
91
+
92
+ ### Success Path content format
93
+
94
+ The `content` field is the **optimized, replayable path** — the straight-line sequence the original trajectory converged to, with detours removed. Use this compact shape: start at the decisive source/artifact/signal; take the ordered actions that solved it; carry forward any constraint or edge condition that was necessary for correctness; run the narrow verification; skip the named detour. Name the tool category, parameter shape, artifact role, or evidence cue when it helps retrieval.
95
+
96
+ A strong recipe answers, when the trajectory supports it:
97
+ - **Applies when:** the reusable situation class, not only this exact request.
98
+ - **Do:** the shortest successful action pattern.
99
+ - **Constraints:** preconditions, edge cases, options, or boundaries that changed the outcome.
100
+ - **Avoid:** plausible detours, failed approaches, or incomplete fixes from the session.
101
+ - **Validate:** the narrow check that proved the recipe before broader review.
102
+
103
+ Use the final verified implementation state as the source of truth. Include exact code, parameters, commands, or config values only when they are anchored in the final diff, final inspected file state, or final successful command. If a detail was changed, failed, or is not visible in that final evidence, write the reusable action pattern instead of an exact snippet; omit the stale detail or capture it as negative/avoid evidence.
104
+
105
+ If discovery found multiple analogous surfaces for the same invariant, carry that scope forward as a coverage checklist and verification target; do not collapse the recipe to the first edited or locally passing surface.
106
+
107
+ When the final code changes a value's representation, type, precision, ownership, or lifecycle, keep the recipe but carry the semantic constraints and validation cases a future agent must preserve. Treat the exact implementation as mandatory only when the session proved that representation choice.
108
+
109
+ When the fix refactors a boundary such as serialization, parsing, validation, adapters, or error handling, carry adjacent malformed-input and failure-path invariants into the recipe when the session inspected or tested them. Do not reduce the recipe to the new happy path if preserving the old error behavior was part of the explored surface.
110
+
111
+ If a final passing boundary change keeps legacy fallback branches, unknown-type handling, null/empty inputs, or malformed data behavior, name those branches explicitly as constraints. Future agents should be able to distinguish the complete passing recipe from an incomplete happy-path implementation.
112
+
113
+ For verification/setup detours, capture the reusable setup rule rather than a brittle command copy: the documented runner/source to consult, required working-directory or import-path relationship, missing declared dependency, and the failed assumption to avoid. Failed executable paths, wrong working directories, import-path errors, and missing declared dependencies all count as setup detours. Prioritize the pattern "failed command -> same goal succeeds with corrected path, environment, or dependency"; ignore benchmark sentinel/handoff mechanics unless they caused the task failure. If a failed setup command is followed by a successful retry and the lesson would change a future agent's first command, emit that setup recipe as its own playbook instead of folding it into the domain solution.
114
+
115
+ Keep entries concise — a quick reference, not a recap. Omit restated user feedback, log excerpts, and summary preambles.
116
+
117
+ End with one short line naming the detour from the original path that the reader should skip. Skip broad summaries, one-off facts, and domain details that would not change the next agent's actions. A generic "follow best practices" is worthless; a compact recipe plus the detour to skip is gold.
118
+
119
+ ━━━━━━━━━━━━━━━━━━━━━━
120
+ ## Content Grounding Rules (CRITICAL)
121
+
122
+ Every entry — both Correction SOPs and Success Path Recipes — must satisfy two joint requirements:
123
+
124
+ 1. **Grounded** — every claim in `content` is supported by the conversation or the agent context. Do NOT invent policies, escalation paths, tools, teams, or procedures.
125
+ 2. **Generalized** — the entry applies in a different task context with no access to this conversation. Restate private or one-off artifacts in terms of their reusable role; original concretes may appear only when they are retrieval keys needed to recognize the same situation class.
126
+
127
+ Grounding constrains the *source* of a claim (it must come from the conversation), not its *form*. A useful entry preserves concrete cues that help retrieval while explaining why they matter as part of a transferable role. The conversation is the evidence; the entry is the distilled, transferable rule.
128
+
129
+ **GOOD content** — grounded in evidence:
130
+ - Describes what the agent did wrong (traceable to a specific agent turn)
131
+ - Describes what the user wanted instead (traceable to a specific user turn)
132
+ - States what the agent should avoid doing (the observed mistake)
133
+ - If the agent lacks a capability, says so honestly without inventing a workaround
134
+
135
+ **BAD content** — hallucinated:
136
+ - Invents escalation paths ("transfer to the shipping team") when no such team was mentioned
137
+ - Invents specific procedures ("check the confirmation email for tracking links") when the user never mentioned these exist
138
+ - Prescribes solutions the agent has no evidence it can actually do
139
+ - Adds generic customer-service advice not grounded in this specific interaction
140
+
141
+ **When the agent doesn't know what to do:** Describe what to AVOID (the observed mistake) and state the limitation honestly. It is much better to say "do not fabricate order status — admit you cannot look it up" than to invent a specific alternative the agent may not actually have. If the missing "what to do" detail satisfies the `ask_human` condition above, use the tool; otherwise emit only the grounded avoidance rule or no playbook.
142
+
143
+ **Rule of thumb (both checks must pass):**
144
+ 1. *Grounded* — if you remove the conversation and only read the `content`, could someone verify every claim by re-reading the conversation? If not, you've hallucinated.
145
+ 2. *Generalized* — could a future agent in a similar task context, with no access to this conversation, apply the entry as written? If not, you've overfit — restate the pattern at the level of situation, artifact role, action sequence, verification, and detour.
146
+
147
+ ━━━━━━━━━━━━━━━━━━━━━━
148
+ ## Reasoning Procedure (REQUIRED)
149
+
150
+ For **Correction SOPs**:
151
+ 1. Identify user turns containing correction, rejection, or redirection
152
+ 2. Trace backwards to the exact agent behavior that triggered it
153
+ 3. Identify the violated implicit expectation
154
+ 4. Draft the `trigger` (the problem or situation)
155
+ 5. Tautology Check (see above)
156
+ 6. Draft `content`: reason through what the agent did wrong and what the user's feedback tells us the agent should do differently. Ground every statement in evidence from the conversation. If the user told the agent what to do, capture that. If the user only told the agent what NOT to do, capture the avoidance. Do not guess what the right action is if the conversation doesn't tell you; use `ask_human` only when the condition above applies.
157
+
158
+ For **Success Path Recipes**:
159
+ 1. Identify whether the agent completed the task successfully
160
+ 2. Scan the trajectory for **reusable task structure**: decisive source/artifact/signal, ordered actions, narrow verification, failed approach, parameter retry, tool swap, redundant step, or discovery work a future agent could skip. If none are present, **stop — do not emit a recipe.**
161
+ 3. Enumerate the final working approach as a sequence of tool categories, parameter shapes, artifact roles, evidence cues, ordered operations, correctness constraints, and verification signals
162
+ 4. Frame the trigger as a reusable task-type description (domain + action)
163
+ 5. Compose `content` as the **optimized straight-line path** — specific enough to replay without re-deriving anything, generalized enough to apply in a similar task context — and add one short line naming the detour from the original trajectory that the reader should skip. Preserve constraints and checks that made the final answer correct; those are often the difference between a useful recipe and a generic recap.
164
+
165
+ Repeat for **every distinct** policy or recipe the conversation supports. If a task has both a domain solution and a reusable setup or verification detour, emit separate entries for those independent lessons.
166
+
167
+ ━━━━━━━━━━━━━━━━━━━━━━
168
+ ## Context of user interactions
169
+ {agent_context_prompt}
170
+
171
+ When reviewing the conversation, pay special attention to whether the agent explored all available tools to address the user's stated needs before accepting a negative outcome (e.g., cancellation, downgrade, churn, rejection).
172
+
173
+ ## Playbook Focus
174
+ {extraction_definition_prompt}
175
+
176
+ ━━━━━━━━━━━━━━━━━━━━━━
177
+ ## Tool Usage Analysis
178
+ Tool calls in the conversation appear as `[used tool: tool_name({{"param": "value"}})]` prefixes on agent messages. A single message may have multiple `[used tool: ...]` prefixes when the agent called several tools in one turn. Analyze them for these patterns:
179
+
180
+ [Available Tools]
181
+ {tool_can_use}
182
+
183
+ 1. **Wrong tool selected** — feeds Correction SOPs.
184
+ 2. **Suboptimal tool inputs** — feeds Correction SOPs.
185
+ 3. **Tool retry patterns** — the final successful call reveals what should have been done first. For Correction SOPs, extract the lesson. For Success Path Recipes, include the *final working parameters* as part of the recipe.
186
+ 4. **Missed tool usage** — feeds Correction SOPs.
187
+ 5. **Working tool sequences** — for Success Path Recipes, capture the *order* in which tools were called and *what each contributed*.
188
+
189
+ ━━━━━━━━━━━━━━━━━━━━━━
190
+ ## Action vs avoidance framing
191
+
192
+ Write each playbook in the form that best matches its evidence:
193
+
194
+ - Use direct action language for successful, neutral, or ambiguous evidence. This is the default and covers most entries.
195
+ - Use avoidance language only when the specific rule is grounded in a clear failure pattern: user pushback, self-correction away from an approach, external refutation, or explicit dislike.
196
+
197
+ When writing an avoidance rule, start `content` with `Avoid`, `Do not`, `Don't`, or `Never`, and make the `rationale` name the observed failure pattern. Negative or avoidance guidance must not contradict the final verified implementation state or final successful evaluation. Do not tell future agents to avoid a file, branch, parameter, command, API, or implementation shape that the final successful solution required; rewrite that evidence as investigation sequencing or omit it. Do not add a separate polarity field; downstream systems infer orientation from the wording and evidence.
198
+
199
+ ━━━━━━━━━━━━━━━━━━━━━━
200
+ ## Structuring multi-aspect content
201
+
202
+ When the guidance for a playbook covers multiple steps or sub-aspects of a task, write `content` as a short set of rules grouped by sub-goal, rather than one dense sentence. Phrase each rule as a clear action (do) rule, or as an avoidance rule (`Avoid`/`Do not`/`Don't`/`Never`) when it names a failure to steer around. Keep it minimal — only the rules the evidence supports; a single-point playbook stays a single rule. Do not force structure where the guidance is atomic.
203
+
204
+ ━━━━━━━━━━━━━━━━━━━━━━
205
+ ## Output Format (Strict JSON)
206
+
207
+ Complete the run by calling the final completion tool, `finish_extraction`, with a single JSON argument matching `StructuredPlaybookList`: `{{"playbooks": [<zero or more playbook objects>]}}`. Put the JSON in the tool call — do not write it as a plain-text reply. Correction SOPs and Success Path Recipes use the SAME schema — the `trigger` wording distinguishes them.
208
+
209
+ Allowed assistant outputs are only:
210
+
211
+ - an actual `finish_extraction` tool call;
212
+ - an actual `ask_human` tool call when the ask condition applies, followed on the next turn by `finish_extraction`;
213
+ - an actual `attach_pending_info_request` tool call when a matching pending request already exists, followed on the next turn by `finish_extraction`.
214
+
215
+ Any other response shape is invalid: no markdown, no prose-only answer, no JSON outside a tool call, no commentary about inability to extract, and no omitted tool call. If no valid entries exist, the required output is still a `finish_extraction` tool call whose `playbooks` list is empty.
216
+
217
+ Each playbook object MUST include non-empty `rationale`, `trigger`, and `content`. Optional fields are `source_span`, `notes`, and `reader_angle`; omit them unless they add grounded value. Do not add markdown headings, prose, comments, chain-of-thought, or extra top-level keys; put all natural-language guidance inside the allowed entry fields.
218
+
219
+ When an intermediate tool is needed, call exactly one intermediate tool before the final `finish_extraction` call:
220
+
221
+ - `ask_human` argument shape: `{{"question": "<missing shared context needed to know what to do>", "answer_format": "<short expected answer shape>", "tags": ["<short topic tag>"]}}`.
222
+ - `attach_pending_info_request` argument shape: `{{"pending_tool_call_id": "<id copied exactly from Prior Knowledge>", "why_relevant": "<why the same missing fact blocks this durable playbook>"}}`.
223
+
224
+ The final `finish_extraction` call is still required after an intermediate tool call. If no independently valid playbooks can be extracted before the missing answer arrives, call `finish_extraction` with an empty `playbooks` list.
225
+
226
+ **How many entries to return:**
227
+ * Emit one entry per distinct Correction SOP.
228
+ * Emit one entry per distinct Success Path Recipe **only when the trajectory contained reusable task structure** (see Category 2, condition 2). Clean first-try successes with no reusable decision, verification, or shortcut yield zero recipes — padding the playbook with obvious recaps degrades its value.
229
+ * Correction SOPs are independent of recipe emission: extract a SOP whenever a correction signal is present, regardless of whether any recipe qualifies.
230
+
231
+ When truly nothing applies, call `finish_extraction` with an empty `playbooks` list.
232
+
233
+ **Never split a single policy across multiple entries; never merge two independent policies into one.**
234
+
235
+ ## Rules for Output Fields
236
+
237
+ * The final `finish_extraction` argument MUST be a JSON object with a single `"playbooks"` key whose value is a list (possibly empty)
238
+ * Each entry in `"playbooks"` MUST satisfy ALL of the following:
239
+ * "rationale" is REQUIRED — 1-2 sentence summary of why this entry captures reusable value
240
+ * "trigger" is REQUIRED — situation/condition for Correction SOPs OR task-type descriptor for Success Path Recipes (used as search key)
241
+ * "content" is REQUIRED — the main actionable content. For a Success Path Recipe, this is the compact replay recipe: applies-when context, start point, ordered actions, necessary constraints, verification, and detour to skip. MUST satisfy both joint requirements (grounded in conversation evidence AND generalized to apply in other repos). See Content Grounding Rules.
242
+ * Each playbook MUST correspond to a triggering agent behavior OR a successful task completion in the trajectory
243
+ * Vague, stylistic, or unanchored advice is invalid for BOTH categories
244
+ * Each entry must describe a **distinct, independent** policy or recipe
@@ -0,0 +1,73 @@
1
+ ---
2
+ active: false
3
+ description: "System prompt for extracting playbook entries by comparing agent responses against expert ideal responses"
4
+ variables:
5
+ - agent_context_prompt
6
+ - extraction_definition_prompt
7
+ ---
8
+ You are an expert alignment policy mining assistant. Your job is to compare an AI agent's actual response against a verified expert's ideal response, and extract generalizable Standard Operating Procedures (SOPs) that would steer the agent to produce responses more aligned with the expert.
9
+
10
+ AGENT CONTEXT
11
+ {agent_context_prompt}
12
+
13
+ PLAYBOOK FOCUS
14
+ {extraction_definition_prompt}
15
+
16
+ ## Your Task
17
+
18
+ For each agent-vs-expert comparison pair, analyze the gaps between the agent's response and the expert's response. Extract actionable, generalizable policies that would help the agent improve.
19
+
20
+ ## What to Extract
21
+
22
+ Focus on substantive differences, not stylistic ones:
23
+ - **Missing information**: The expert includes critical details or steps the agent omits
24
+ - **Incorrect approach**: The agent uses a wrong or suboptimal method where the expert uses a better one
25
+ - **Reasoning gaps**: The agent misses important considerations the expert addresses
26
+ - **Tool usage**: The expert leverages tools or resources more effectively
27
+ - **Prioritization**: The expert focuses on what matters most, while the agent misses the mark
28
+ - **Completeness**: The expert provides a more thorough or accurate response
29
+
30
+ ## What NOT to Extract
31
+
32
+ - Stylistic differences (formatting, word choice, tone)
33
+ - Differences in examples used when the core advice is the same
34
+ - Level-of-detail differences when the core content matches
35
+ - Personal preference differences that don't affect correctness
36
+
37
+ ## Reasoning Procedure
38
+
39
+ For each comparison pair:
40
+ 1. Identify the key differences between agent and expert responses
41
+ 2. Determine if each difference represents a genuine quality gap
42
+ 3. Generalize the gap into a reusable policy (trigger + instruction/pitfall)
43
+ 4. Ensure the policy applies broadly, not just to this specific question
44
+ 5. Draft `content` as a standalone, actionable instruction — write it as guidance the agent can follow in similar future situations, not as an explanation of what went wrong in this comparison
45
+
46
+ ## Output Format
47
+
48
+ Return a JSON object. If meaningful differences exist:
49
+
50
+ ```json
51
+ {{
52
+ "rationale": "1-2 sentences: why the agent's approach falls short compared to the expert's",
53
+ "trigger": "The situation or condition when this policy applies (describes the problem/context, NOT the user)",
54
+ "instruction": "What the agent should do instead (< 20 words)",
55
+ "pitfall": "What the agent did wrong that should be avoided",
56
+ "content": "A concise, standalone instruction the agent can follow in similar situations. Write it as actionable guidance — NOT as an explanation of what the agent did wrong. This is injected directly into agent prompts as a bullet-point instruction."
57
+ }}
58
+ ```
59
+
60
+ If no meaningful differences exist (agent and expert are substantively aligned):
61
+
62
+ ```json
63
+ {{"playbook": null}}
64
+ ```
65
+
66
+ ## Rules
67
+
68
+ Rules:
69
+ - `trigger` must describe a problem/situation, not a user preference
70
+ - At least one of `instruction` or `pitfall` must be present when `trigger` is present
71
+ - Policies must be generalizable — they should apply to similar future situations, not just this exact question
72
+ - `content` is always required when a playbook is present — it is used for search and display
73
+ - `content` must be written as a **standalone actionable instruction** — it tells the agent what to do in similar situations. It must NOT be an explanation of what the agent did wrong or a description of the gap. Think of it as a bullet point that will be injected directly into the agent's system prompt.