stacktrace-cli 0.2.3__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (220) hide show
  1. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/PKG-INFO +4 -4
  2. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/README.md +3 -3
  3. stacktrace_cli-0.3.0/docs/releases/v0.3.0.md +53 -0
  4. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/aidr.md +8 -8
  5. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/detection-upload.md +1 -1
  6. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/detector.md +37 -37
  7. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/monitor.md +22 -22
  8. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/pyproject.toml +1 -1
  9. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/__init__.py +1 -1
  10. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/analysis.py +9 -9
  11. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/cli.py +22 -21
  12. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/analyzer.py +1 -1
  13. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/cache.py +15 -15
  14. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/deterministic.py +1 -1
  15. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/finding.py +3 -3
  16. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/markers.py +1 -1
  17. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/priors.py +21 -21
  18. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/reasoning.py +74 -63
  19. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/render.py +14 -14
  20. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/rules.py +3 -3
  21. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/run.py +65 -65
  22. stacktrace_cli-0.2.3/src/stacktrace_cli/monitor/escalate.py → stacktrace_cli-0.3.0/src/stacktrace_cli/monitor/reasoning.py +13 -13
  23. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/render.py +2 -2
  24. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/server.py +37 -35
  25. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/app.js +22 -22
  26. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/styles.css +4 -4
  27. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/state.py +3 -3
  28. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/verdicts.py +3 -3
  29. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/watch.py +7 -7
  30. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/cli.py +8 -8
  31. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/sync_detect.py +8 -8
  32. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/feed_harness.mjs +14 -14
  33. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_feed_model.py +4 -4
  34. stacktrace_cli-0.2.3/tests/monitor/test_escalation_gates.py → stacktrace_cli-0.3.0/tests/monitor/test_reasoning_gates.py +76 -74
  35. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_render.py +17 -17
  36. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_server.py +17 -17
  37. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_site_assets.py +2 -2
  38. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_verdicts.py +3 -3
  39. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_watch.py +10 -10
  40. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_cli.py +2 -2
  41. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_sync_detect.py +3 -3
  42. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_agent_instance_id.py +1 -1
  43. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_analysis.py +5 -5
  44. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_cli.py +7 -7
  45. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_analyzer_contract.py +1 -1
  46. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_cache.py +18 -18
  47. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_end_to_end.py +16 -16
  48. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_markers.py +1 -1
  49. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_priors.py +33 -33
  50. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_reasoning.py +67 -69
  51. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_render.py +19 -19
  52. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_run.py +70 -68
  53. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_fingerprint_covers_what_the_analyzer_reads.py +31 -31
  54. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_monitor_cli.py +2 -2
  55. stacktrace_cli-0.2.3/tests/test_no_live_document_states_the_old_escalate_default.py → stacktrace_cli-0.3.0/tests/test_no_live_document_states_the_old_stage_three_default.py +13 -6
  56. stacktrace_cli-0.2.3/tests/test_stored_verdicts_are_never_gated_on_escalate.py → stacktrace_cli-0.3.0/tests/test_stored_verdicts_are_never_gated_on_reasoning.py +7 -7
  57. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_untrusted_content_never_travels.py +7 -7
  58. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/uv.lock +1 -1
  59. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.agents/skills/release-stacktrace/SKILL.md +0 -0
  60. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.claude/settings.json +0 -0
  61. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.claude/skills/release-stacktrace/SKILL.md +0 -0
  62. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.codex/hooks.json +0 -0
  63. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.github/workflows/autofix.yml +0 -0
  64. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.github/workflows/ci.yml +0 -0
  65. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.github/workflows/claude.yml +0 -0
  66. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.github/workflows/publish-pypi.yml +0 -0
  67. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/.gitignore +0 -0
  68. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/AGENTS.md +0 -0
  69. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/CLAUDE.md +0 -0
  70. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0001-session-telemetry-as-input.md +0 -0
  71. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0002-session-collection-in-openaidr.md +0 -0
  72. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0003-runtime-edges.md +0 -0
  73. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0004-three-stage-detector.md +0 -0
  74. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0005-detection-family-and-report-assembly.md +0 -0
  75. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0006-trust-boundary-and-detection-upload.md +0 -0
  76. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0007-proprietary-package-on-open-dependencies.md +0 -0
  77. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0008-console-in-separate-demo-package.md +0 -0
  78. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0010-detection-severity-and-confidence-ladders.md +0 -0
  79. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0011-verdict-cache.md +0 -0
  80. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0012-observation-evidence-kinds-and-transport.md +0 -0
  81. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0013-rule-catalogue-triage-and-per-rule-context.md +0 -0
  82. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0014-declared-project-mapping.md +0 -0
  83. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0015-narrow-the-security-catalogue.md +0 -0
  84. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0016-stall-grouping-is-session-wide.md +0 -0
  85. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0017-a-declined-repeat-is-a-stall.md +0 -0
  86. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0018-remote-sync-config-and-facade-consumption.md +0 -0
  87. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0019-delegated-commands-are-openaca-objects.md +0 -0
  88. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0020-openaca-consumption-boundary.md +0 -0
  89. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0021-two-command-kinds.md +0 -0
  90. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0022-detection-scope-is-a-catalogue-column.md +0 -0
  91. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0023-sync-detect-collects-and-does-not-escalate.md +0 -0
  92. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0024-the-upload-carries-observations.md +0 -0
  93. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0025-a-denied-call-is-activity-never-an-invocation.md +0 -0
  94. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0026-the-console-becomes-a-product-surface.md +0 -0
  95. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0027-a-finding-may-name-what-it-could-not-place.md +0 -0
  96. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0028-the-console-shows-what-it-placed.md +0 -0
  97. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/0029-stage-three-returns-to-opt-in.md +0 -0
  98. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/HOOK-PROMPT.md +0 -0
  99. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/INDEX.md +0 -0
  100. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/adrs/TEMPLATE.md +0 -0
  101. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/cutover-openaca-remote.md +0 -0
  102. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/002-session-input.md +0 -0
  103. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/003-correlation.md +0 -0
  104. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/004-detector.md +0 -0
  105. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/005-remote-sync.md +0 -0
  106. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/006-cli-composition.md +0 -0
  107. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/007-detection-upload.md +0 -0
  108. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/plans/008-monitor.md +0 -0
  109. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/releases/v0.0.1.md +0 -0
  110. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/releases/v0.1.0.md +0 -0
  111. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/releases/v0.2.0.md +0 -0
  112. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/releases/v0.2.1.md +0 -0
  113. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/releases/v0.2.2.md +0 -0
  114. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/releases/v0.2.3.md +0 -0
  115. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/cli-composition.md +0 -0
  116. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/correlation.md +0 -0
  117. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/remote-sync.md +0 -0
  118. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/docs/specs/session-input.md +0 -0
  119. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/scripts/git-hooks/pre-push +0 -0
  120. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/scripts/install-hooks.sh +0 -0
  121. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/__main__.py +0 -0
  122. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/__init__.py +0 -0
  123. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/acquire.py +0 -0
  124. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/composition.py +0 -0
  125. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/join.py +0 -0
  126. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/observed.py +0 -0
  127. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/orchestrate.py +0 -0
  128. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/project_map.py +0 -0
  129. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/record.py +0 -0
  130. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/render.py +0 -0
  131. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/__init__.py +0 -0
  132. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/__init__.py +0 -0
  133. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/exclusions.md +0 -0
  134. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/framing.md +0 -0
  135. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/stacktrace-deceptive-completion.md +0 -0
  136. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/stacktrace-injected-instruction-followed.md +0 -0
  137. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/stacktrace-intent-drift.md +0 -0
  138. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/secrets.py +0 -0
  139. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/verdict.py +0 -0
  140. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/__init__.py +0 -0
  141. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/OFL.txt +0 -0
  142. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/dm-mono-400-latin.woff2 +0 -0
  143. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/dm-mono-500-latin.woff2 +0 -0
  144. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/dm-sans-latin.woff2 +0 -0
  145. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/index.html +0 -0
  146. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/__init__.py +0 -0
  147. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/client.py +0 -0
  148. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/config.py +0 -0
  149. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/detect_payload.py +0 -0
  150. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/payload.py +0 -0
  151. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/policy.py +0 -0
  152. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/redact.py +0 -0
  153. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/spool.py +0 -0
  154. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/sync.py +0 -0
  155. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/upload_contract.py +0 -0
  156. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/__init__.py +0 -0
  157. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/access.py +0 -0
  158. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/outcome.py +0 -0
  159. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/protocols.py +0 -0
  160. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/render.py +0 -0
  161. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/__init__.py +0 -0
  162. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/agent_bom_fixture.py +0 -0
  163. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/detector_session_fixture.py +0 -0
  164. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/__init__.py +0 -0
  165. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/monitor/test_state.py +0 -0
  166. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/__init__.py +0 -0
  167. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/helpers.py +0 -0
  168. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_cli.py +0 -0
  169. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_client.py +0 -0
  170. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_config.py +0 -0
  171. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_activity.py +0 -0
  172. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_client.py +0 -0
  173. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_contract_is_exhaustive.py +0 -0
  174. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_gate_matches_the_cloud.py +0 -0
  175. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_layers_compose.py +0 -0
  176. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_payload_reaches_the_cloud_model.py +0 -0
  177. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_redaction.py +0 -0
  178. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_spool.py +0 -0
  179. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_upload_contract.py +0 -0
  180. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_detect_wire_payload.py +0 -0
  181. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_facade_contract.py +0 -0
  182. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_payload.py +0 -0
  183. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_policy.py +0 -0
  184. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_redact_payload.py +0 -0
  185. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_seam.py +0 -0
  186. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_spool.py +0 -0
  187. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_sync.py +0 -0
  188. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/remote/test_upload_contract.py +0 -0
  189. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_acquire.py +0 -0
  190. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_bom_shape.py +0 -0
  191. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_call_status_conventions.py +0 -0
  192. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_composition.py +0 -0
  193. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_correlate_orchestrate.py +0 -0
  194. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_correlate_render.py +0 -0
  195. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_correlated_session.py +0 -0
  196. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detection_components.py +0 -0
  197. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detection_subject.py +0 -0
  198. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_analyzer_live.py +0 -0
  199. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_deterministic.py +0 -0
  200. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_finding.py +0 -0
  201. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_liveness.py +0 -0
  202. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_secrets.py +0 -0
  203. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_detector_verdict.py +0 -0
  204. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_help_sections.py +0 -0
  205. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_join.py +0 -0
  206. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_observed.py +0 -0
  207. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_openaca_contract.py +0 -0
  208. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_openaidr_contract.py +0 -0
  209. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_option_arity_matches_the_binary.py +0 -0
  210. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_project_map.py +0 -0
  211. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_readme_examples_are_real_output.py +0 -0
  212. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_release_readiness.py +0 -0
  213. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_seam_boundary.py +0 -0
  214. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_sessions_access.py +0 -0
  215. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_sessions_end_to_end.py +0 -0
  216. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_sessions_protocol_typing.py +0 -0
  217. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_sessions_protocols.py +0 -0
  218. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_sessions_render.py +0 -0
  219. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_shell_separators_match_the_shell.py +0 -0
  220. {stacktrace_cli-0.2.3 → stacktrace_cli-0.3.0}/tests/test_verification_subcommands_match_the_binary.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: stacktrace-cli
3
- Version: 0.2.3
3
+ Version: 0.3.0
4
4
  Summary: CLI for Stacktrace — Detection and Response platform for AI Agents.
5
5
  Project-URL: Homepage, https://stacktrace.ai
6
6
  Author-email: "Stacktrace AI, Inc" <founders@stacktrace.ai>
@@ -128,13 +128,13 @@ because it is installed.
128
128
  Two of `detect`'s three stages run entirely locally and need no model or
129
129
  credential, and they are the two a bare `detect` runs. The third sends flagged
130
130
  sessions to the agent's *own* CLI — the provider that produced the transcript,
131
- never a different one — and runs only when you pass `--escalate`, capped by
131
+ never a different one — and runs only when you pass `--reasoning`, capped by
132
132
  `--budget`.
133
133
 
134
134
  `sessions` omits prompts, tool arguments and results unless you pass
135
135
  `--include-content`. `monitor` binds to loopback only, refuses a non-loopback
136
- address rather than warning about it, and escalates nothing unless `--escalate`
137
- is given.
136
+ address rather than warning about it, and analyses nothing with a model unless
137
+ `--reasoning` is given.
138
138
 
139
139
  ## Status
140
140
 
@@ -103,13 +103,13 @@ because it is installed.
103
103
  Two of `detect`'s three stages run entirely locally and need no model or
104
104
  credential, and they are the two a bare `detect` runs. The third sends flagged
105
105
  sessions to the agent's *own* CLI — the provider that produced the transcript,
106
- never a different one — and runs only when you pass `--escalate`, capped by
106
+ never a different one — and runs only when you pass `--reasoning`, capped by
107
107
  `--budget`.
108
108
 
109
109
  `sessions` omits prompts, tool arguments and results unless you pass
110
110
  `--include-content`. `monitor` binds to loopback only, refuses a non-loopback
111
- address rather than warning about it, and escalates nothing unless `--escalate`
112
- is given.
111
+ address rather than warning about it, and analyses nothing with a model unless
112
+ `--reasoning` is given.
113
113
 
114
114
  ## Status
115
115
 
@@ -0,0 +1,53 @@
1
+ # 0.3.0 — the third stage is asked for by name
2
+
3
+ `--escalate` is now `--reasoning`. The old name described the pipeline's own
4
+ mechanic — stage two promoting a session to stage three — rather than what you
5
+ get for it, which is a reasoning model reading the session. The flag now says
6
+ that, on `stacktrace detect`, `stacktrace monitor` and `remote sync detect`
7
+ alike.
8
+
9
+ **This release renames a flag and two JSON fields with no aliases.** Everything
10
+ you need to change is in the table below; nothing else about the pipeline moved,
11
+ and the defaults are exactly what 0.2.3 shipped.
12
+
13
+ ## What to change
14
+
15
+ | 0.2.3 | 0.3.0 |
16
+ | --- | --- |
17
+ | `detect --escalate` | `detect --reasoning` |
18
+ | `monitor --escalate` | `monitor --reasoning` |
19
+ | `remote sync detect --escalate` | `remote sync detect --reasoning` |
20
+ | `--no-escalate` | *(nothing — omit `--reasoning`)* |
21
+ | `summary.escalated` | `summary.reasoning_requested` |
22
+ | `unknowns[].escalated_on` | `unknowns[].reasoning_reasons` |
23
+ | `POST /api/escalate` | `POST /api/reasoning` |
24
+
25
+ **If you typed `--no-escalate`, ignore what the CLI suggests.** Click offers
26
+ `--no-cache` as the nearest match, which is the wrong flag and an expensive one
27
+ to take: it disables the verdict cache, so a run re-buys answers you have
28
+ already paid for. There is no replacement for `--no-escalate`. Stage three has
29
+ been opt-in since 0.2.3, so leaving `--reasoning` off is the whole of it.
30
+
31
+ ## Highlights
32
+
33
+ - **`--reasoning` is a plain flag, not an on/off pair.** `--no-reasoning` does
34
+ not exist, because there is nothing to turn off: stage three is opt-in on all
35
+ three commands (ADR-0029). A pair would imply a default worth overriding.
36
+
37
+ - **The vocabulary moved with the flag, so one word covers the whole path.**
38
+ The stage that answers was already called `reasoning`; the signal that reaches
39
+ it is now a `ReasoningRequest` rather than an `Escalation`. `--help`, the text
40
+ output and `--format json` agree. A run that declined now reads `re-run with
41
+ --reasoning to analyse them`, and the summary line reads `3 for reasoning, 2
42
+ analysed`.
43
+
44
+ - **Your stored verdicts survive the upgrade.** The verdict cache is keyed on
45
+ what the prompt is a function of, not on the names this codebase uses
46
+ internally, so a session graded under 0.2.3 is still answered from cache under
47
+ 0.3.0. The stage-three prompt is unchanged and still recorded as `v1`, which
48
+ is what keeps two machines' answers comparable across the rename.
49
+
50
+ - **The monitor's page and route follow.** `POST /api/reasoning` replaces
51
+ `POST /api/escalate`, and the page's button reads against the same budget it
52
+ always did. A monitor and a CLI of the same version are required, as before —
53
+ the page is served by the process it is talking to.
@@ -42,8 +42,8 @@ version in a named manifest.
42
42
  | A discovery tool | OpenACA's existing scan already inventories agent tooling |
43
43
  | A fleet service | Only detector findings may travel, agent-keyed and content-free; sessions never leave the machine (ADR-0006) |
44
44
 
45
- One exception, named rather than rounded away: escalating a session for semantic
46
- analysis sends it to the provider that session already came from. See
45
+ One exception, named rather than rounded away: requesting reasoning on a session
46
+ sends it to the provider that session already came from. See
47
47
  [Trust boundary](#trust-boundary).
48
48
 
49
49
  ## Core tenets
@@ -56,7 +56,7 @@ enforced rather than merely intended.
56
56
  |---|---|---|
57
57
  | **Trust** — sensitive data does not cross the trust boundary | Transcripts, correlation records and session models are read only on the machine that produced them and have no upload path. What may travel is a detector finding: a verdict, component coordinates, confidence and observed capabilities, keyed to the agent — never conversation, and never an evidence excerpt | ADR-0006, and the local-only markers carried in the session model |
58
58
  | **Quality** — precision first; recall from a second source | A false alarm spends a human hour, so precision is a product decision, not a tuning parameter. Recall is bought by **adding an independent kind of evidence**, never by loosening a threshold | The detector's principles; the match outcome carried through correlation |
59
- | **Cost** — scalable unit economics | Deterministic rules first, then scoring against the component graph. Neither uses a model. **Only what they cannot resolve reaches inference** | The three-stage detector; no local model; a bounded escalation budget |
59
+ | **Cost** — scalable unit economics | Deterministic rules first, then scoring against the component graph. Neither uses a model. **Only what they cannot resolve reaches inference** | The three-stage detector; no local model; a bounded reasoning budget |
60
60
  | **Security** — read-only sensing | The JSONL and SQLite caches agents already write, opened read-only. No process hooks, no proxy, no injected agent, no listener in the sensing path | Collection design; ADR-0001 |
61
61
 
62
62
  Two of these deserve their consequences stated plainly, because they are the ones
@@ -151,7 +151,7 @@ Each is specified where its component is, alongside the sub-component it joins:
151
151
  | Contract | Inside | Joins | Specified in |
152
152
  |---|---|---|---|
153
153
  | **Collection interface** | OpenAIDR | The per-kind readers to OpenAIDR's collector | [One contract, per agent kind](https://github.com/open-agent-security/openaidr/blob/main/docs/specs/session-collection.md#one-contract-per-agent-kind) |
154
- | **Escalation signal** | Detector | `priors` to `reasoning` — the detector's second and third stages | [The escalation signal](detector.md#the-escalation-signal) |
154
+ | **Reasoning request** | Detector | `priors` to `reasoning` — the detector's second and third stages | [The reasoning request](detector.md#the-reasoning-request) |
155
155
  | **Analyzer adapter** | Detector | `reasoning` to the per-kind command-line agent it invokes | [The analyzer adapter](detector.md#the-analyzer-adapter) |
156
156
 
157
157
  The per-kind rows are pluggable slots, not components: supporting another agent
@@ -223,7 +223,7 @@ go; session input, the Correlator and the Detector do identical work in both.
223
223
  | Serves | Production: agent-keyed findings to Stacktrace Cloud (ADR-0006) | Development and demonstration: the local trace console |
224
224
  | Trigger | One run over a window of sessions | Changed files, as agents work |
225
225
  | Emit | Run to quiescence, emit complete findings | Draw a span when parsed, enrich it in place |
226
- | Escalation | Looser cap, off-peak | Tight cap, background, never inline |
226
+ | Reasoning request | Looser cap, off-peak | Tight cap, background, never inline |
227
227
 
228
228
  **Mode lives in the driver, never inside a component.** No component can ask which
229
229
  mode it is running in, and nothing in the session model, correlation record or
@@ -235,12 +235,12 @@ cold pass without the tail — which is what stops two modes becoming two pipeli
235
235
  | Collect, project to spans | sub-second per changed file | Every change, or once per scheduled run |
236
236
  | Correlate | fast **only** against a cached graph | Graph rebuilt on manifest change |
237
237
  | `deterministic`, `priors` | milliseconds per evaluation | On each new span, against the session's accumulated evidence — never one span in isolation |
238
- | `reasoning` | seconds | Background, flagged sessions only, within the escalation budget |
238
+ | `reasoning` | seconds | Background, flagged sessions only, within the reasoning budget |
239
239
 
240
240
  **Rule: nothing waits for the slowest stage.** Interactively, a tool call appears
241
241
  when parsed and its outcome, identity and findings attach later as enrichments to
242
242
  a row that already exists. On a scheduled run the same rule means one slow
243
- escalation does not hold the whole run.
243
+ reasoning request does not hold the whole run.
244
244
 
245
245
  Three consequences that would otherwise look arbitrary. None of them is a
246
246
  concession to the console — each holds in both modes:
@@ -259,7 +259,7 @@ concession to the console — each holds in both modes:
259
259
  to re-read. Incremental collection is what makes that affordable: full pass at
260
260
  cold start, then changed files only.
261
261
 
262
- **The escalation budget is the one genuine per-mode parameter**, and the driver
262
+ **The reasoning budget is the one genuine per-mode parameter**, and the driver
263
263
  sets it. `reasoning` is the only stage that costs seconds, a network call and
264
264
  money, so how much of the flagged set may reach it is a consumer decision, not a
265
265
  property of the detector. Everything else about the two modes is scheduling.
@@ -291,7 +291,7 @@ Detecting before registering keeps a failed run from creating an asset.
291
291
  |---|---|---|
292
292
  | `--agent-kind`, `--bom`, `--budget`, `--sample-budget`, `--cache/--no-cache`, `--project-map`, `--root` | as `stacktrace detect` | |
293
293
  | `--since` | `7d` | |
294
- | `--escalate/--no-escalate` | **`--no-escalate`** | Stage 3 spends the developer's provider quota and crosses a boundary; unattended is the wrong place for that to be implicit |
294
+ | `--reasoning` | **off** | Stage 3 spends the developer's provider quota and crosses a boundary; unattended is the wrong place for that to be implicit |
295
295
  | `--dry-run` | off | NDJSON via the same builder the real path uses, so the preview **is** the payload. ADR-0006 constraint 4 requires it |
296
296
  | `--quiet`, `--allow-offline-cache` | off | as `sync endpoint` |
297
297
 
@@ -52,28 +52,28 @@ folds them together with OpenACA's three families into one envelope we assemble
52
52
 
53
53
  **`deterministic` and `priors` need no model, network or credential.** That is
54
54
  what lets AIDR be adopted and evaluated before anyone debates language models on
55
- developer machines — `--no-escalate` leaves exactly those two running, and they
55
+ developer machines — a bare `detect` leaves exactly those two running, and they
56
56
  are the stages a first evaluation should be judged on.
57
57
 
58
58
  **That is a claim about the stages, not about the run.** Correlation happens
59
59
  before any of them and matches advisories over what was invoked, which is a
60
- network round trip to osv.dev (`docs/specs/correlation.md`). So `--no-escalate`
60
+ network round trip to osv.dev (`docs/specs/correlation.md`). So omitting `--reasoning`
61
61
  is not an offline mode, and the CLI's help says so: what it turns off is the
62
62
  only stage that sends anything *session-derived* off the machine. Package
63
63
  coordinates travel either way; prompts, arguments and results travel only when
64
- escalation runs, and then to the session's own provider and no other.
64
+ reasoning runs, and then to the session's own provider and no other.
65
65
 
66
66
  **`reasoning` is opt-in and capped** (ADR-0029). A stage that spends the
67
67
  developer's quota and crosses that content boundary is asked for rather than
68
- assumed, so it runs only under `--escalate`. Three things bound its cost when it
68
+ assumed, so it runs only under `--reasoning`. Three things bound its cost when it
69
69
  is asked for: the `--budget` cap, the verdict cache — so an unchanged session is
70
70
  never paid for twice — and the flag itself.
71
71
 
72
72
  **The ordering is the unit economics.** Microseconds, then milliseconds, then —
73
73
  only for the residue neither could resolve — seconds of inference. The per-session
74
74
  cost of the ordinary case is effectively zero, so session volume does not scale
75
- the model bill. Escalation is capped per run for the same reason: an unbounded
76
- escalation path would make cost a function of how noisy a day was.
75
+ the model bill. Reasoning is capped per run for the same reason: an unbounded
76
+ reasoning path would make cost a function of how noisy a day was.
77
77
 
78
78
  **And the ordering is also the precision story.** Each stage emits only what it
79
79
  can defend at its own level of evidence. `deterministic` does not guess at intent;
@@ -205,8 +205,8 @@ development machine, running stages one and two.
205
205
  | `stacktrace-tool-shadowing` | 0 | Requires `outcome == "ambiguous"`. **The join produced 0 ambiguous resolutions in 27,280.** Structurally unfireable, not under-tuned |
206
206
  | `stacktrace-unpinned-invoked` | 0 findings, recorded on 47 sessions | Escalated only *in combination*, and its only possible partner was `tool-shadowing`. So the threshold of two could never be met and every firing was discarded. It is also a static property of the BOM, which is OpenACA's posture surface rather than a runtime signal |
207
207
 
208
- **The combination escalation route went with them.** It required
209
- `len(escalating) >= 2` over a set with exactly two members, one of which could
208
+ **The combination route went with them.** It required
209
+ `len(requesting) >= 2` over a set with exactly two members, one of which could
210
210
  not fire — arithmetic that cannot come out. `docs/adrs/0013` records the
211
211
  decision; the deferred table below records what each rule would need.
212
212
 
@@ -448,14 +448,14 @@ left half moved to the declared side, where the graph is authoritative. This is
448
448
  the rule that most needs the join to exist, and the reason it is worth the two
449
449
  halves.
450
450
 
451
- **Escalation has four routes, and none is a single supply-chain prior.** One
451
+ **Reasoning request has four routes, and none is a single supply-chain prior.** One
452
452
  unpinned component is Tuesday; an unpinned component whose tool name is also
453
- shadowed is seconds of inference well spent — so two or more escalating priors is
454
- the first route. The second is a stage-1 injection marker, which escalates on its
453
+ shadowed is seconds of inference well spent — so two or more requesting priors is
454
+ the first route. The second is a stage-1 injection marker, which requests reasoning on its
455
455
  own: `stacktrace-injected-instruction-followed` exists precisely to ask whether
456
456
  the agent acted on that marker, and gating it behind unrelated supply-chain
457
- priors would make the rule unreachable in its defining case. A single escalating
458
- prior firing alone is recorded as a reason a later escalation may cite, and
457
+ priors would make the rule unreachable in its defining case. A single requesting
458
+ prior firing alone is recorded as a reason a later reasoning request may cite, and
459
459
  otherwise dropped.
460
460
 
461
461
  The third and fourth routes exist because the first two both select for
@@ -479,7 +479,7 @@ analysed, and a shipped rule would fire only by accident:
479
479
  that are already covered, which is the blind spot it exists for. Its budget
480
480
  bounds how many quiet sessions compete for a slot, not how many extra
481
481
  invocations a run may make -- selected samples join the one pending list the
482
- run-level budget caps, ranked below every evidence-backed escalation.
482
+ run-level budget caps, ranked below every evidence-backed reasoning request.
483
483
 
484
484
  **The session is passed whole, and the bound is a tripwire rather than a
485
485
  window.** An earlier draft of this spec specified a window around the spans of
@@ -513,10 +513,10 @@ fired on presence would score near-zero precision against a corpus that is 100%
513
513
  injected.
514
514
 
515
515
  **Stage three emits; it never withdraws.** There is no suppression verdict — no
516
- "no finding" answer that cancels the escalation which produced it. A stage that
516
+ "no finding" answer that cancels the reasoning request which produced it. A stage that
517
517
  reasons about a transcript must not be able to talk the pipeline out of a fact,
518
- and every escalation route is a stage-one fact, so there is nothing a verdict
519
- could honestly withdraw. What would bring suppression back is an escalating prior
518
+ and every route into reasoning is a stage-one fact, so there is nothing a verdict
519
+ could honestly withdraw. What would bring suppression back is a requesting prior
520
520
  that is a *reading* rather than a fact (ADR-0013).
521
521
 
522
522
  **`entrypoint` changes what drift means.** A programmatic entrypoint means no
@@ -532,7 +532,7 @@ reported to the analyzer as neither — never defaulted to interactive, which
532
532
  would present a machine-generated `user` turn as a person's authorisation and is
533
533
  the failure this clause exists to prevent.
534
534
 
535
- ## The escalation signal
535
+ ## The reasoning request
536
536
 
537
537
  The contract between `priors` and `reasoning`, stated so either can be built
538
538
  without the other.
@@ -544,14 +544,14 @@ without the other.
544
544
  | Stage-1 findings on this session | Their spans are where a semantic question is worth asking. **Descriptors only** — see the payload rule below |
545
545
  | Spans of interest | So a verdict can cite them rather than describe them |
546
546
  | Coverage | So a verdict knows how much of the session `priors` could actually read |
547
- | Budget | What remains of this run's escalation allowance |
547
+ | Budget | What remains of this run's reasoning allowance |
548
548
 
549
- **`priors` remains the only producer.** Stage 1 does not escalate: it emits, and
549
+ **`priors` remains the only producer.** Stage 1 does not request reasoning: it emits, and
550
550
  `priors` composes the signal from whatever stage 1 already emitted on that
551
551
  session. One producer, one consumer, and a deterministic finding can still raise
552
- a session's priority without inventing a second escalation path.
552
+ a session's priority without inventing a second route into reasoning.
553
553
 
554
- **There are two routes, and both are stage-one facts.** Each escalates because
554
+ **There are two routes, and both are stage-one facts.** Each requests reasoning because
555
555
  a fact is established and its *meaning* is not. A third — a drift precursor, a
556
556
  denied call whose capability the agent then reached another way — went with
557
557
  `refusal-circumvention` in ADR-0015; `denial_kind` is still on the seam, so the
@@ -562,7 +562,7 @@ route is cheap to restore if drift proves under-reached without it.
562
562
  | `stacktrace-injection-marker` | Instruction-shaped content reached the session. Whether the agent *acted on it* is unanswerable deterministically |
563
563
  | `stacktrace-credential-egress` | A credential-shaped string reached an outbound call. Whether it is a credential or a fixture's placeholder is the single largest false-positive class `exclusions.md` names — an admission that this is the model's question |
564
564
 
565
- **Reliability findings never escalate.** `progress-stall` and `hung-call` are
565
+ **Reliability findings never request reasoning.** `progress-stall` and `hung-call` are
566
566
  facts about whether the session got stuck, and no amount of inference makes a
567
567
  repeated call more or less repeated. They are also excluded from the *security*
568
568
  test the quiet-session sample applies, so a session whose only finding is a
@@ -570,7 +570,7 @@ stall stays sample-eligible — otherwise the sample would skip exactly the
570
570
  sessions where something went wrong in a way no security rule reads.
571
571
 
572
572
  **Attacker-derived content never enters a high-trust slot.** A decoded injection
573
- payload is named by class and span in the escalation signal — never quoted into
573
+ payload is named by class and span in the reasoning request — never quoted into
574
574
  it. The prompt's *why this was escalated* position is read by the analyzer as
575
575
  pipeline testimony, and a quoted payload sitting there is the same attacker text
576
576
  promoted from data to instruction. It reaches evidence and logs; it does not reach
@@ -578,7 +578,7 @@ the prompt. Prefacing it with *this is not an instruction* does not help, and ad
578
578
  no signal the transcript does not already carry.
579
579
 
580
580
  A stage that cannot run is not a stage that found nothing: if `reasoning` is
581
- unavailable, the escalation is recorded as *unknown* with the priors that
581
+ unavailable, the reasoning request is recorded as *unknown* with the priors that
582
582
  triggered it, so the reason for concern survives even when the analysis does not.
583
583
 
584
584
  ## The reasoning stage
@@ -651,7 +651,7 @@ cannot be attributed, which is worse for a stage whose findings must be
651
651
  checkable.
652
652
 
653
653
  There is no fourth invocation. A suppression question would have nothing to ask
654
- about: measured on 74 escalation-eligible sessions it was **never once
654
+ about: measured on 74 reasoning-eligible sessions it was **never once
655
655
  reachable**, because every route is either a stage-one fact — out of
656
656
  suppression's reach by design — or the evidence-free `sampled` route, which
657
657
  carries no prior to withdraw.
@@ -909,7 +909,7 @@ expect, and is deliberately absent (ADR-0004).
909
909
  | Its work is already split | Compositional questions → `priors`, better because the graph is authoritative rather than inferred. Semantic questions → `reasoning`, better because the model is stronger |
910
910
  | The cost is not the weights | A multi-gigabyte artifact through a path built for a small pure-Python package; an inference runtime absent from typical machines; resource governance; quantized inference varying across hardware, so a confidence value would mean different things on different laptops |
911
911
 
912
- Deferred, not refused forever. If escalation volume ever justifies a cheap local
912
+ Deferred, not refused forever. If reasoning volume ever justifies a cheap local
913
913
  filter, this is where it goes.
914
914
 
915
915
  ## Rules not in the catalogue
@@ -943,17 +943,17 @@ narrowing of scope rather than a judgement on any rule's quality.
943
943
 
944
944
  | Rule | Stage | Returns when |
945
945
  |---|---|---|
946
- | `stacktrace-guardrail-modification` | deterministic | **A reason to emit rather than escalate.** The 53 real config writes are worth a person's attention and the 84 authoring writes are not, and the split is already known — what is missing is a route that asks *did someone ask for this* instead of asserting it happened. See ADR-0013 |
946
+ | `stacktrace-guardrail-modification` | deterministic | **A reason to emit rather than request reasoning.** The 53 real config writes are worth a person's attention and the 84 authoring writes are not, and the split is already known — what is missing is a route that asks *did someone ask for this* instead of asserting it happened. See ADR-0013 |
947
947
  | `stacktrace-destructive-action` | deterministic | Nothing: the escaped-workspace half is implementable today and produced 1 real finding in 30 days. It is withheld because the `history_discarded` half produced the other 12 and a rule is re-added whole or not at all. Re-add scoped to `recursive_delete_outside_workspace` alone |
948
948
  | `stacktrace-tool-shadowing` | priors | **An `ambiguous` resolution.** The join has produced none in 27,280. Either the join never emits one, in which case shadowing needs a different definition, or it does and the corpus has not exercised it — a question for `correlate/join.py` before this returns |
949
- | `stacktrace-unpinned-invoked` | priors | **A second escalating prior to combine with,** or a reason to escalate alone. Neither exists, and as a static BOM property it belongs to OpenACA's posture rules |
949
+ | `stacktrace-unpinned-invoked` | priors | **A second requesting prior to combine with,** or a reason to request reasoning alone. Neither exists, and as a static BOM property it belongs to OpenACA's posture rules |
950
950
  | `stacktrace-unsanctioned-mcp-tool-use` | priors | A decision to widen the catalogue again (ADR-0015). Nothing technical blocks it: it fired 4 times on the measured corpus and its input — an `absent_from_composition` resolution keyed `mcp_server` — is unchanged |
951
951
  | `stacktrace-unattended-privileged-action` | deterministic | The same. It was the largest surviving security rule at 12 findings, and its inputs (`entrypoint`, `permission_mode`, observed capabilities) all remain on the seam |
952
952
  | `stacktrace-provider-refusal` | deterministic | The same; `provider_refusals` still arrives on the session seam and fired 3 times |
953
- | `stacktrace-refusal-circumvention` | deterministic | The same. Restoring it also restores the third escalation route, which is the reason to want it back before the other four |
953
+ | `stacktrace-refusal-circumvention` | deterministic | The same. Restoring it also restores the third route into reasoning, which is the reason to want it back before the other four |
954
954
  | `stacktrace-capability-crossing` | priors | The same, plus the input it always lacked: resolutions carrying a component candidate that declares sensitive access (275 of 27,280 carried any candidate at all) |
955
955
  Suppression was withdrawn with them, and is not a rule: it returns only when an
956
- escalating prior is a *reading* rather than a fact, because only then is there
956
+ requesting prior is a *reading* rather than a fact, because only then is there
957
957
  something a verdict could honestly withdraw (ADR-0013).
958
958
 
959
959
  One wiring note remains:
@@ -984,7 +984,7 @@ Three things a detection carries that the other families do not:
984
984
  |---|---|
985
985
  | **Session reference** | Identity, agent kind, start, turn count — and **no content**. It exists to identify a session, never to transport conversation. Drift here would breach the local-only boundary from inside the data model, which is why it is stated as a constraint rather than left to reviewers |
986
986
  | **Observed capabilities** | Drawn from the existing closed capability taxonomy (OpenACA ADR-0041), naming what the session was seen exercising. This puts behavioural evidence on the same axis as the static *declared* / *curated* / *inferred* descriptors. The taxonomy is **not widened** to accommodate it |
987
- | **Provenance of the verdict** | Which stage produced it, and for escalated findings which analyzer, model and prompt version. Without this, two verdicts from two machines are not comparable |
987
+ | **Provenance of the verdict** | Which stage produced it, and for requested findings which analyzer, model and prompt version. Without this, two verdicts from two machines are not comparable |
988
988
 
989
989
  ## What this component owns
990
990
 
@@ -994,12 +994,12 @@ Three things a detection carries that the other families do not:
994
994
  | Combined report assembly | New — work ADR-0005 got for free by living upstream | Read OpenACA's three families, merge, sort and render four into one envelope in JSON, SARIF and text (ADR-0005) |
995
995
  | The `stacktrace-*` rule namespace and its validation | New, small | Fails loudly on an unknown identifier |
996
996
  | Payload redaction before upload | New | Ours, so OpenACA's upload contract stays family-agnostic (ADR-0006, ADR-0005) |
997
- | Escalation to an external process | New capability, new risk surface | Adapter per agent kind, read-only invocation, explicit unknown on absence — see the safety rules above |
997
+ | Reasoning request to an external process | New capability, new risk surface | Adapter per agent kind, read-only invocation, explicit unknown on absence — see the safety rules above |
998
998
 
999
999
  **Nothing here is contributed upstream.** OpenACA gains no family, no output row
1000
1000
  type, no scan flag and no rule namespace (ADR-0005). The envelope shape is reused
1001
1001
  rather than redefined, so one report still carries behaviour and composition
1002
- together. The genuinely new surface is escalation, which is why the constraints
1002
+ together. The genuinely new surface is reasoning, which is why the constraints
1003
1003
  on it are stated as rules rather than guidance.
1004
1004
 
1005
1005
  ## Where this stands
@@ -1015,7 +1015,7 @@ arm, because evaluation is out of scope for the first pass.
1015
1015
  | Security findings | **20** |
1016
1016
  | Reliability findings | 17 |
1017
1017
  | Largest single rule's share | 5 of the 20 |
1018
- | Sessions escalated to `reasoning` | 3 |
1018
+ | Sessions sent to `reasoning` | 3 |
1019
1019
  | `capability-crossing` findings | **0** |
1020
1020
  | `uninventoried_component` unknowns | 6 |
1021
1021
  | Resolutions carrying any BOM candidate | 275 of 27,280 (**1%**) |
@@ -1035,8 +1035,8 @@ Four of those rows are the ones to read, and none of them is the headline:
1035
1035
  - **The 6 `uninventoried_component` unknowns were new noise of a different
1036
1036
  kind** — honest, and not free. They were what `unsanctioned-mcp-tool-use`
1037
1037
  produced instead of a finding it could not place, and both went in ADR-0015.
1038
- - **The 3 escalations no longer trace to a single test fixture.** Every
1039
- escalation on the pre-triage run came from one session reading another
1038
+ - **The 3 reasoning requests no longer trace to a single test fixture.** Every
1039
+ reasoning request on the pre-triage run came from one session reading another
1040
1040
  security tool's own detector; requiring that injection characters were
1041
1041
  *placed* rather than merely present removed that class.
1042
1042
 
@@ -18,8 +18,8 @@ Supersedes the console's separation from the engine (ADR-0008 → ADR-0026).
18
18
  per refresh, no parsing our own stdout.
19
19
  3. Leaving the page open all day is free: no model, no provider quota, and
20
20
  osv.dev is asked once per component rather than once per tick. The single
21
- exception is a person clicking escalate, which § *Escalation* gates.
22
- 4. The one action the page offers — escalating a session to stage 3 — cannot be
21
+ exception is a person clicking the reasoning button, which § *Reasoning* gates.
22
+ 4. The one action the page offers — sending a session to stage 3 — cannot be
23
23
  taken by anything but a person who asked for it.
24
24
 
25
25
  ## Non-goals
@@ -37,7 +37,7 @@ Supersedes the console's separation from the engine (ADR-0008 → ADR-0026).
37
37
 
38
38
  ```
39
39
  stacktrace monitor [--port N] [--host 127.0.0.1] [--interval 2]
40
- [--escalate] [--budget 10] [--no-open]
40
+ [--reasoning] [--budget 10] [--no-open]
41
41
  [--agent-kind K]... [--since 7d] [--bom P]... [--root P]
42
42
  [--project-map OLD=NEW]...
43
43
  ```
@@ -50,8 +50,8 @@ is monitor's.
50
50
  | `--port` | ephemeral | A fixed port is for bookmarking; the default avoids a collision with whatever else is on 8000 |
51
51
  | `--host` | `127.0.0.1` | Loopback. A non-loopback value is **refused**, not warned about: this page renders working directories and matched values, and *remote access* is a non-goal above. The flag exists so that `--host 0.0.0.0` fails with a reason instead of appearing to work |
52
52
  | `--interval` | `2` (seconds) | The floor between ticks, not a guarantee. A tick that takes longer than the interval does not queue another |
53
- | `--escalate` | off | Whether the escalate endpoint **exists**. Off, so the page has no spending action unless asked for (ADR-0023's reasoning, one surface over). **This flag alone is the whole opt-in** |
54
- | `--budget` | `10` | Escalations this monitor process may spend, total — not per tick, not per session. `detect`'s own default, and nobody is expected to pass it: it is a ceiling for the ordinary case, not a decision the user makes each run. Lower it for a shared or long-lived session |
53
+ | `--reasoning` | off | Whether the reasoning endpoint **exists**. Off, so the page has no spending action unless asked for (ADR-0023's reasoning, one surface over). **This flag alone is the whole opt-in** |
54
+ | `--budget` | `10` | Reasoning requests this monitor process may spend, total — not per tick, not per session. `detect`'s own default, and nobody is expected to pass it: it is a ceiling for the ordinary case, not a decision the user makes each run. Lower it for a shared or long-lived session |
55
55
  | `--no-open` | off | Print the URL instead of opening a browser |
56
56
 
57
57
  Exit codes: `1` on a usage error or a port that cannot be bound. SIGINT is how
@@ -75,7 +75,7 @@ caller rather than its third copy.
75
75
  # src/stacktrace_cli/analysis.py — the composition root for the read pipeline
76
76
  def analyse(
77
77
  *, agent_kinds, since, bom_paths, project_map, root,
78
- escalate=False, budget=..., sample_budget=..., cache=None,
78
+ reasoning=False, budget=..., sample_budget=..., cache=None,
79
79
  advisories=None,
80
80
  ) -> Analysis # .run: DetectorRun, .view: CorrelatedView, plus the window
81
81
  ```
@@ -231,7 +231,7 @@ the stages before it carried, so a render that replaced the columns threw away
231
231
  the reader's place in them several times a second. Activity dedupes on `span`,
232
232
  findings on `(rule_id, session_id, cited spans)` — the detector is
233
233
  deterministic, so those three identify a finding across stages. Only the
234
- summaries, which are counts, are recomputed. The escalate button therefore
234
+ summaries, which are counts, are recomputed. The reasoning button therefore
235
235
  lives in its own element inside each card, repainted as the budget moves
236
236
  without rebuilding the card around it.
237
237
 
@@ -293,7 +293,7 @@ a client needs to remember.
293
293
  "tool_name": "get_issue", "status": "succeeded"}]
294
294
  }],
295
295
  "alerts": [ /* an entry of `render_json`'s `detections[]`, verbatim */ ],
296
- "escalate": {"enabled": false, "remaining": 0, "running": []},
296
+ "reasoning": {"enabled": false, "remaining": 0, "running": []},
297
297
  "banners": ["claude-code: transcript directory unreadable"]
298
298
  }
299
299
  ```
@@ -314,7 +314,7 @@ unplaceable session, a tick that raised. A page with an empty column and no
314
314
  banner is indistinguishable from a working page with nothing to show, which is
315
315
  the one confusion this document refuses.
316
316
 
317
- `escalate.running` carries the session ids currently out for analysis, so a
317
+ `reasoning.running` carries the session ids currently out for analysis, so a
318
318
  second click on the same session is refused by the page rather than by the
319
319
  budget — a stage-3 run takes tens of seconds and a spinner is not a lock.
320
320
 
@@ -329,12 +329,12 @@ not carry. It is also why `--host` warns.
329
329
  |---|---|
330
330
  | `GET /` and `/{app.js,styles.css}` | Static, from package data. `Cache-Control: no-store` |
331
331
  | `GET /api/state?since=N` | Long-poll. Returns immediately when `revision > N`; otherwise blocks until the next tick or a timeout, then returns the current snapshot |
332
- | `POST /api/escalate` | Only registered when `--escalate` was passed. § *Escalation* |
332
+ | `POST /api/reasoning` | Only registered when `--reasoning` was passed. § *Reasoning* |
333
333
 
334
334
  The long-poll returns on timeout rather than hanging forever, so a laptop that
335
335
  slept does not leave a socket the server is still holding.
336
336
 
337
- ## Escalation
337
+ ## Reasoning
338
338
 
339
339
  ### Why the button exists at all
340
340
 
@@ -348,13 +348,13 @@ and 2 can see that a marker arrived and that a component was reached; only stage
348
348
  Which puts the reader in the position the button removes. They are watching a
349
349
  session where something looks wrong, the page can tell them a marker was
350
350
  injected, and the next question — *did the agent follow it?* — is answerable only
351
- by leaving the page, finding the session id, and running `detect --escalate` in a
351
+ by leaving the page, finding the session id, and running `detect --reasoning` in a
352
352
  terminal. The finding they want is one click and forty seconds away, and the
353
353
  interface makes them assemble a command instead.
354
354
 
355
355
  **So the button is there because the question arises while looking at the page,
356
356
  and nowhere else.** That is also the whole of its justification: it is not a
357
- convenience for running the detector, and monitor never escalates on its own.
357
+ convenience for running the detector, and monitor never requests reasoning on its own.
358
358
 
359
359
  ### Why it is not merely a button
360
360
 
@@ -368,10 +368,10 @@ Four gates, all required:
368
368
 
369
369
  | Gate | Stops |
370
370
  |---|---|
371
- | The endpoint is registered only under `--escalate` | Spending by a monitor nobody asked to spend |
371
+ | The endpoint is registered only under `--reasoning` | Spending by a monitor nobody asked to spend |
372
372
  | `Origin`/`Referer` must be absent or loopback | A remote page: the browser sets `Origin` on cross-origin POSTs and a page cannot forge it. Absent is allowed because a local non-browser client could run the CLI anyway |
373
373
  | A token minted at startup, carried in the opened URL, echoed in a request header | A local *browser* page that guessed the port |
374
- | `--budget` (default 10), decremented per escalation, disabling at zero | An unbounded bill from a held-down button. At 24–61 s and $0.013–$0.038 per session, the default caps a monitor's whole lifetime at well under a dollar — which is why it can have a default rather than being asked for |
374
+ | `--budget` (default 10), decremented per reasoning request, disabling at zero | An unbounded bill from a held-down button. At 24–61 s and $0.013–$0.038 per session, the default caps a monitor's whole lifetime at well under a dollar — which is why it can have a default rather than being asked for |
375
375
 
376
376
  ### The prompt
377
377
 
@@ -390,7 +390,7 @@ rather than left to the render layer, because what it says *is* the consent:
390
390
  > - Answers the three questions the other stages cannot: whether an injected
391
391
  > instruction was acted on, whether the agent drifted from the request, and
392
392
  > whether a completion claims work it did not do.
393
- > - **7 of 10** escalations left for this monitor session.
393
+ > - **7 of 10** reasoning run(s) left for this monitor session.
394
394
  >
395
395
  > *Already analysed once? A session that has not changed since is answered from
396
396
  > the local verdict cache and costs nothing (ADR-0011).*
@@ -404,7 +404,7 @@ turns into repeat clicks. It **says what the reader gets**, so the trade is
404
404
  legible rather than a leap of faith. And it **shows the remaining budget**, so
405
405
  running out is not a surprise at click seven.
406
406
 
407
- A refused escalation says which gate refused it — a silent no-op reads as a
407
+ A refused reasoning request says which gate refused it — a silent no-op reads as a
408
408
  broken page. A refusal for budget says so in the button's own label rather than
409
409
  only on click, so the ceiling is visible before it is hit.
410
410
 
@@ -426,7 +426,7 @@ src/stacktrace_cli/
426
426
  monitor/
427
427
  state.py Snapshot, revision, the waiter set the long-poll blocks on
428
428
  watch.py The tick: mtime gate, one analyse() call, snapshot publication
429
- server.py Handler, routes, the four escalation gates
429
+ server.py Handler, routes, the four reasoning gates
430
430
  site/ index.html, app.js, styles.css — package data
431
431
  app.js owns the feed's grouping, pacing and ordering
432
432
  cli.py EDITED — detect calls analyse(); monitor registered
@@ -451,13 +451,13 @@ they test without a socket. `server.py` is tested against a real
451
451
  ## Robustness bar
452
452
 
453
453
  Aims to get right: the page never shows a finding the detector did not produce;
454
- no tick spends provider quota without a person having passed `--escalate` and
454
+ no tick spends provider quota without a person having passed `--reasoning` and
455
455
  clicked through the prompt; the server never binds anything but loopback; an
456
456
  empty or failed collection is visibly empty or failed rather than blank; and
457
457
  `detect` and `sync detect` behave exactly as they did before the extraction.
458
458
 
459
459
  Above the bar however small: a page that can invent, alter or suppress a
460
- detection; an escalation reachable without all four gates; a bind to a
460
+ detection; a reasoning request reachable without all four gates; a bind to a
461
461
  non-loopback address; and any behaviour change in the two commands this work
462
462
  refactors — the extraction is proven by their existing tests passing
463
463
  *unedited*, so a diff that touches them is the finding.
@@ -468,8 +468,8 @@ text, not the design.
468
468
  ## References
469
469
 
470
470
  - ADR-0026 — the console becomes a product surface (supersedes ADR-0008)
471
- - ADR-0004 — provider affinity for escalation
471
+ - ADR-0004 — provider affinity for reasoning
472
472
  - ADR-0006 — the trust boundary, and why a path may appear locally
473
473
  - ADR-0011 — the verdict cache, read here and never written
474
- - ADR-0023 — `sync detect` does not escalate; the same reasoning gates `--escalate`
474
+ - ADR-0023 — `sync detect` does not escalate; the same reasoning gates `--reasoning`
475
475
  - ADR-0024 — one source per fact, applied to `alerts`
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "stacktrace-cli"
3
- version = "0.2.3"
3
+ version = "0.3.0"
4
4
  description = "CLI for Stacktrace — Detection and Response platform for AI Agents."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -1,3 +1,3 @@
1
1
  """The `stacktrace` command-line interface — detection and response for AI agents."""
2
2
 
3
- __version__ = "0.2.3"
3
+ __version__ = "0.3.0"