stacktrace-cli 0.2.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/PKG-INFO +7 -6
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/README.md +6 -5
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0004-three-stage-detector.md +2 -0
- stacktrace_cli-0.3.0/docs/adrs/0029-stage-three-returns-to-opt-in.md +115 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/INDEX.md +1 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/007-detection-upload.md +2 -2
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/008-monitor.md +1 -1
- stacktrace_cli-0.3.0/docs/releases/v0.2.3.md +102 -0
- stacktrace_cli-0.3.0/docs/releases/v0.3.0.md +53 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/aidr.md +8 -8
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/detection-upload.md +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/detector.md +41 -41
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/monitor.md +22 -22
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/pyproject.toml +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/__init__.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/analysis.py +9 -9
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/cli.py +37 -28
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/acquire.py +154 -2
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/analyzer.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/cache.py +15 -15
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/deterministic.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/finding.py +3 -3
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/markers.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/priors.py +21 -21
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/reasoning.py +74 -63
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/render.py +22 -25
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/rules.py +3 -3
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/run.py +65 -65
- stacktrace_cli-0.2.2/src/stacktrace_cli/monitor/escalate.py → stacktrace_cli-0.3.0/src/stacktrace_cli/monitor/reasoning.py +13 -13
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/render.py +2 -2
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/server.py +37 -35
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/app.js +22 -22
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/index.html +1 -5
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/styles.css +4 -9
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/state.py +3 -3
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/verdicts.py +3 -3
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/watch.py +7 -7
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/cli.py +8 -9
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/sync_detect.py +13 -7
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/feed_harness.mjs +14 -14
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_feed_model.py +4 -4
- stacktrace_cli-0.2.2/tests/monitor/test_escalation_gates.py → stacktrace_cli-0.3.0/tests/monitor/test_reasoning_gates.py +76 -74
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_render.py +17 -17
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_server.py +17 -17
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_site_assets.py +2 -2
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_verdicts.py +3 -3
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_watch.py +10 -10
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_cli.py +5 -5
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_sync_detect.py +90 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_acquire.py +299 -2
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_agent_instance_id.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_analysis.py +5 -5
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_cli.py +62 -22
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_analyzer_contract.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_cache.py +18 -18
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_end_to_end.py +16 -16
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_markers.py +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_priors.py +33 -33
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_reasoning.py +67 -69
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_render.py +46 -29
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_run.py +70 -68
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_fingerprint_covers_what_the_analyzer_reads.py +31 -31
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_monitor_cli.py +2 -2
- stacktrace_cli-0.3.0/tests/test_no_live_document_states_the_old_stage_three_default.py +165 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_openaca_contract.py +7 -1
- stacktrace_cli-0.3.0/tests/test_stored_verdicts_are_never_gated_on_reasoning.py +74 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_untrusted_content_never_travels.py +7 -7
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/uv.lock +1 -1
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.agents/skills/release-stacktrace/SKILL.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.claude/settings.json +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.claude/skills/release-stacktrace/SKILL.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.codex/hooks.json +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.github/workflows/autofix.yml +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.github/workflows/ci.yml +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.github/workflows/claude.yml +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.github/workflows/publish-pypi.yml +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/.gitignore +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/AGENTS.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/CLAUDE.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0001-session-telemetry-as-input.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0002-session-collection-in-openaidr.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0003-runtime-edges.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0005-detection-family-and-report-assembly.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0006-trust-boundary-and-detection-upload.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0007-proprietary-package-on-open-dependencies.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0008-console-in-separate-demo-package.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0010-detection-severity-and-confidence-ladders.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0011-verdict-cache.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0012-observation-evidence-kinds-and-transport.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0013-rule-catalogue-triage-and-per-rule-context.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0014-declared-project-mapping.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0015-narrow-the-security-catalogue.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0016-stall-grouping-is-session-wide.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0017-a-declined-repeat-is-a-stall.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0018-remote-sync-config-and-facade-consumption.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0019-delegated-commands-are-openaca-objects.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0020-openaca-consumption-boundary.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0021-two-command-kinds.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0022-detection-scope-is-a-catalogue-column.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0023-sync-detect-collects-and-does-not-escalate.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0024-the-upload-carries-observations.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0025-a-denied-call-is-activity-never-an-invocation.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0026-the-console-becomes-a-product-surface.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0027-a-finding-may-name-what-it-could-not-place.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/0028-the-console-shows-what-it-placed.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/HOOK-PROMPT.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/adrs/TEMPLATE.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/cutover-openaca-remote.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/002-session-input.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/003-correlation.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/004-detector.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/005-remote-sync.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/plans/006-cli-composition.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/releases/v0.0.1.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/releases/v0.1.0.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/releases/v0.2.0.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/releases/v0.2.1.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/releases/v0.2.2.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/cli-composition.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/correlation.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/remote-sync.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/docs/specs/session-input.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/scripts/git-hooks/pre-push +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/scripts/install-hooks.sh +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/__main__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/composition.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/join.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/observed.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/orchestrate.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/project_map.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/record.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/correlate/render.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/exclusions.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/framing.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/stacktrace-deceptive-completion.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/stacktrace-injected-instruction-followed.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/prompts/v1/stacktrace-intent-drift.md +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/secrets.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/detector/verdict.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/OFL.txt +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/dm-mono-400-latin.woff2 +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/dm-mono-500-latin.woff2 +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/monitor/site/fonts/dm-sans-latin.woff2 +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/client.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/config.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/detect_payload.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/payload.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/policy.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/redact.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/spool.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/sync.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/remote/upload_contract.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/access.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/outcome.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/protocols.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/src/stacktrace_cli/sessions/render.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/agent_bom_fixture.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/detector_session_fixture.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/monitor/test_state.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/__init__.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/helpers.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_cli.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_client.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_config.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_activity.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_client.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_contract_is_exhaustive.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_gate_matches_the_cloud.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_layers_compose.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_payload_reaches_the_cloud_model.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_redaction.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_spool.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_upload_contract.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_detect_wire_payload.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_facade_contract.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_payload.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_policy.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_redact_payload.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_seam.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_spool.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_sync.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/remote/test_upload_contract.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_bom_shape.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_call_status_conventions.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_composition.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_correlate_orchestrate.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_correlate_render.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_correlated_session.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detection_components.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detection_subject.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_analyzer_live.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_deterministic.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_finding.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_liveness.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_secrets.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_detector_verdict.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_help_sections.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_join.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_observed.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_openaidr_contract.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_option_arity_matches_the_binary.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_project_map.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_readme_examples_are_real_output.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_release_readiness.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_seam_boundary.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_sessions_access.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_sessions_end_to_end.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_sessions_protocol_typing.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_sessions_protocols.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_sessions_render.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_shell_separators_match_the_shell.py +0 -0
- {stacktrace_cli-0.2.2 → stacktrace_cli-0.3.0}/tests/test_verification_subcommands_match_the_binary.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: stacktrace-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: CLI for Stacktrace — Detection and Response platform for AI Agents.
|
|
5
5
|
Project-URL: Homepage, https://stacktrace.ai
|
|
6
6
|
Author-email: "Stacktrace AI, Inc" <founders@stacktrace.ai>
|
|
@@ -126,14 +126,15 @@ because it is installed.
|
|
|
126
126
|
## What leaves your machine
|
|
127
127
|
|
|
128
128
|
Two of `detect`'s three stages run entirely locally and need no model or
|
|
129
|
-
credential
|
|
130
|
-
|
|
131
|
-
|
|
129
|
+
credential, and they are the two a bare `detect` runs. The third sends flagged
|
|
130
|
+
sessions to the agent's *own* CLI — the provider that produced the transcript,
|
|
131
|
+
never a different one — and runs only when you pass `--reasoning`, capped by
|
|
132
|
+
`--budget`.
|
|
132
133
|
|
|
133
134
|
`sessions` omits prompts, tool arguments and results unless you pass
|
|
134
135
|
`--include-content`. `monitor` binds to loopback only, refuses a non-loopback
|
|
135
|
-
address rather than warning about it, and
|
|
136
|
-
is given.
|
|
136
|
+
address rather than warning about it, and analyses nothing with a model unless
|
|
137
|
+
`--reasoning` is given.
|
|
137
138
|
|
|
138
139
|
## Status
|
|
139
140
|
|
|
@@ -101,14 +101,15 @@ because it is installed.
|
|
|
101
101
|
## What leaves your machine
|
|
102
102
|
|
|
103
103
|
Two of `detect`'s three stages run entirely locally and need no model or
|
|
104
|
-
credential
|
|
105
|
-
|
|
106
|
-
|
|
104
|
+
credential, and they are the two a bare `detect` runs. The third sends flagged
|
|
105
|
+
sessions to the agent's *own* CLI — the provider that produced the transcript,
|
|
106
|
+
never a different one — and runs only when you pass `--reasoning`, capped by
|
|
107
|
+
`--budget`.
|
|
107
108
|
|
|
108
109
|
`sessions` omits prompts, tool arguments and results unless you pass
|
|
109
110
|
`--include-content`. `monitor` binds to loopback only, refuses a non-loopback
|
|
110
|
-
address rather than warning about it, and
|
|
111
|
-
is given.
|
|
111
|
+
address rather than warning about it, and analyses nothing with a model unless
|
|
112
|
+
`--reasoning` is given.
|
|
112
113
|
|
|
113
114
|
## Status
|
|
114
115
|
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: 0029
|
|
3
|
+
title: Stage three returns to opt-in
|
|
4
|
+
status: accepted
|
|
5
|
+
date: 2026-09-09
|
|
6
|
+
supersedes: null
|
|
7
|
+
superseded-by: null
|
|
8
|
+
amends: 0004
|
|
9
|
+
amended-by: null
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
## Context
|
|
13
|
+
|
|
14
|
+
ADR-0004 was amended on 2026-09-03 to default `--escalate` **on** for
|
|
15
|
+
`stacktrace detect`. The reasoning was sound and the amendment said so plainly:
|
|
16
|
+
a stage nobody turns on never runs, and rules that depend on it — `injection-
|
|
17
|
+
marker` above all — can only ever be evaluated against zero verdicts. It also
|
|
18
|
+
named its own exit: *"Revisit once there is a measured escalation rate and
|
|
19
|
+
verdict yield to argue from; if the yield does not justify the spend, the
|
|
20
|
+
honest move is back to opt-in rather than a narrower cap."*
|
|
21
|
+
|
|
22
|
+
What forces the revisit now is not the yield. It is that the default was only
|
|
23
|
+
ever changed in one of the three places that carry it, and the other two — plus
|
|
24
|
+
everything we publish — kept saying the opposite:
|
|
25
|
+
|
|
26
|
+
| Surface | Default |
|
|
27
|
+
| --- | --- |
|
|
28
|
+
| `run_detector()` | `escalate=False` |
|
|
29
|
+
| `stacktrace monitor` | `False` — *"a page left open is the wrong place for that to be implicit"* |
|
|
30
|
+
| `remote sync detect` | `False` (ADR-0023) |
|
|
31
|
+
| `stacktrace detect` | `True` |
|
|
32
|
+
|
|
33
|
+
The published quickstart tells a new user to run `stacktrace detect
|
|
34
|
+
--no-escalate`, annotated *"# Investigate without model escalation"* — a
|
|
35
|
+
documented first command whose only purpose is to undo the default. The site's
|
|
36
|
+
data-handling section makes the same promise in prose: *"Start with detection
|
|
37
|
+
that doesn't send session content to a model."* `docs/plans/004-detector.md`
|
|
38
|
+
still describes the stage as opt-in. One surface out of four, contradicted by
|
|
39
|
+
its own documentation, is not a default; it is a divergence.
|
|
40
|
+
|
|
41
|
+
Be honest about the evidence this ADR does **not** have. The escalation rate is
|
|
42
|
+
low — an audit on 2026-09-02 measured 3 escalations across 635 sessions, 1 on
|
|
43
|
+
evidence and 2 sampled — but that audit predates the amendment and so describes
|
|
44
|
+
behaviour under the opt-in default it is being used to restore. Verdict yield,
|
|
45
|
+
the other half of what ADR-0004 asked for, has not been measured. Stage 3 has
|
|
46
|
+
also still never run against a real analyzer: the `live_cli` test is opt-in and
|
|
47
|
+
deselected. So this is not the evidence-based revisit ADR-0004 anticipated. It
|
|
48
|
+
is a consistency and consent decision taken ahead of that evidence.
|
|
49
|
+
|
|
50
|
+
## Decision
|
|
51
|
+
|
|
52
|
+
`stacktrace detect` defaults `--escalate` off, matching `run_detector`,
|
|
53
|
+
`monitor` and `sync detect`. A bare `stacktrace detect` sends no session content
|
|
54
|
+
off the machine.
|
|
55
|
+
|
|
56
|
+
Two things travel with it. The verdict cache is no longer gated on `escalate`:
|
|
57
|
+
`run_detector` already states that a verdict already paid for is an answer this
|
|
58
|
+
run *has*, and that `escalate` governs whether new ones may be commissioned, not
|
|
59
|
+
whether old ones may be read — `cli.py` contradicted that, and flipping the
|
|
60
|
+
default without fixing it would have made every stored verdict silently
|
|
61
|
+
unreadable on the default path. And a run that declines to escalate still counts
|
|
62
|
+
and names the sessions that qualified, as `reasoning_not_requested`, so a run
|
|
63
|
+
without stage 3 does not read as a clean one.
|
|
64
|
+
|
|
65
|
+
## Alternatives considered
|
|
66
|
+
|
|
67
|
+
- **Leave it on and fix only the docs.** Rejected: the quickstart, the
|
|
68
|
+
data-handling copy and the plan all describe opt-in, and three of four code
|
|
69
|
+
surfaces implement it. Changing four things to match one is the wrong
|
|
70
|
+
direction, and the one is the surface that spends the user's money.
|
|
71
|
+
- **Keep it on until verdict yield is measured, as ADR-0004 asked.** Rejected on
|
|
72
|
+
consent rather than evidence: the measurement requires a bare `detect` to keep
|
|
73
|
+
sending transcript content to a provider by default, and the same measurement
|
|
74
|
+
is obtainable from users who opt in. The amendment accepted that cost
|
|
75
|
+
deliberately; what it did not anticipate is that we would simultaneously
|
|
76
|
+
publish the opposite promise.
|
|
77
|
+
- **A narrower cap instead of opt-in.** Rejected by ADR-0004 itself, in advance:
|
|
78
|
+
*"the honest move is back to opt-in rather than a narrower cap."*
|
|
79
|
+
- **Prompt interactively on first run.** Rejected: `detect` is scriptable and
|
|
80
|
+
runs unattended; a prompt makes the non-interactive path either hang or pick a
|
|
81
|
+
default, which is this decision again with extra machinery.
|
|
82
|
+
|
|
83
|
+
## Consequences
|
|
84
|
+
|
|
85
|
+
A bare `stacktrace detect` is now local plus the osv.dev coordinate lookup, and
|
|
86
|
+
nothing else leaves. That is the promise the site already makes, kept.
|
|
87
|
+
|
|
88
|
+
The cost is the one the amendment identified and it returns in full: stage 3
|
|
89
|
+
runs only when asked, so verdict yield and the rules that depend on it stay
|
|
90
|
+
unmeasured unless we go and measure them. That work does not disappear — it
|
|
91
|
+
moves from "collected passively from every user" to "collected deliberately",
|
|
92
|
+
and it should be done, because `injection-marker` cannot be evaluated at all
|
|
93
|
+
without it.
|
|
94
|
+
|
|
95
|
+
Watch for the stage going dark again. `reasoning_not_requested` in the reasoning
|
|
96
|
+
accounting is the signal: if it is the dominant outcome across real runs for a
|
|
97
|
+
release or two, the stage is not being reached and the rules behind it are still
|
|
98
|
+
untuned. That is the same failure the 2026-09-03 amendment reacted to, and the
|
|
99
|
+
answer will need to be something other than flipping the default a second time.
|
|
100
|
+
|
|
101
|
+
Users relying on the implicit default silently lose stage 3. `--no-escalate`
|
|
102
|
+
keeps working, so no invocation breaks; the release note carries the change.
|
|
103
|
+
|
|
104
|
+
## When to revisit
|
|
105
|
+
|
|
106
|
+
- **Verdict yield gets measured** and justifies the spend — then this is a
|
|
107
|
+
candidate to revert, but with the docs and the other three surfaces changed in
|
|
108
|
+
the same commit, which is what went wrong the first time.
|
|
109
|
+
- **`reasoning_not_requested` dominates** real runs for two releases, per above.
|
|
110
|
+
- **Stage 3 runs against a real analyzer** and the `live_cli` test stops being
|
|
111
|
+
deselected — until then the default-on path was shipping the least-validated
|
|
112
|
+
stage.
|
|
113
|
+
- **The escalation stops costing the user.** The whole argument is that the
|
|
114
|
+
stage spends someone else's quota and crosses a content boundary. A stage that
|
|
115
|
+
did neither would not need to be asked for.
|
|
@@ -61,3 +61,4 @@ supersedes anything yet.
|
|
|
61
61
|
- [0025](0025-a-denied-call-is-activity-never-an-invocation.md) — **A denied call is activity, and never an invocation.** ADR-0017 makes a declined repeat a stall while `outcome.py` keeps a denied call out of every count, so a denied-only session produced a finding with no activity and the Cloud rejected the whole run; a separate `denied` counter resolves it without either rule bending, and coverage keeps summing `invocations` alone. Read before changing what `activity[]` counts, or before folding `denied` into a coverage-like ratio.
|
|
62
62
|
- [0027](0027-a-finding-may-name-what-it-could-not-place.md) — **A finding may name the component it could not place.** Narrows ADR-0006 constraint 3 for one field: an unplaced subject carries the name the session addressed it by, never as an identity. Read before removing a name from a finding on trust-boundary grounds, or before widening this to a second field.
|
|
63
63
|
- [0028](0028-the-console-shows-what-it-placed.md) — **The console shows what it placed, and states no coverage.** Unplaceable rows, the `N/M placed` header fraction and the red on refusals are gone from `stacktrace monitor`: an unplaced row names no component, and `outcome.py`'s rule about a denied call makes reddening one the page contradicting its own counters. The snapshot contract, the upload and every rule are unchanged. Read before adding a page-side filter, before restoring a coverage ratio to the console, or if `unsanctioned-mcp-tool-use` returns.
|
|
64
|
+
- [0029](0029-stage-three-returns-to-opt-in.md) — **Stage three returns to opt-in.** Amends ADR-0004: `stacktrace detect` defaults `--escalate` off, matching `run_detector`, `monitor` and `sync detect`, all of which never changed — the 2026-09-03 amendment landed in one of four surfaces while the quickstart, the data-handling copy and `plans/004` kept promising opt-in. Taken on consistency and consent, not on the verdict yield 0004 asked for, which is still unmeasured. The verdict cache is ungated from `escalate` in the same change. Read before changing an escalation default, or before arguing stage 3 should be on again.
|
|
@@ -243,8 +243,8 @@ rewrite*.
|
|
|
243
243
|
- [x] **Step 1: Keep `sync endpoint`'s side-effect order.** Load config, refuse
|
|
244
244
|
if unconfigured, client, replay spool, **detect**, register, upload per kind.
|
|
245
245
|
Detecting before registering keeps a failed run from creating an asset.
|
|
246
|
-
- [x] **Step 2: `--escalate` defaults off
|
|
247
|
-
(ADR-0023).
|
|
246
|
+
- [x] **Step 2: `--escalate` defaults off** — a scheduled unattended run must
|
|
247
|
+
not spend provider quota or cross a content boundary implicitly (ADR-0023).
|
|
248
248
|
- [x] **Step 3: `--dry-run` prints NDJSON from the same builder.**
|
|
249
249
|
- [x] **Step 4: Cap evidence, refuse everything else.** Keep `evidence[0]` —
|
|
250
250
|
the anchor the Cloud identifies a call-scoped detection by — and order, with
|
|
@@ -177,7 +177,7 @@ def analyse(
|
|
|
177
177
|
return Analysis(run=run, view=acquired.view, window_start=window_start, window_end=window_end)
|
|
178
178
|
```
|
|
179
179
|
|
|
180
|
-
- [x] **Step 4: Rewire `cli.py`'s `detect`.** Replace the three calls with one `analyse(...)`; keep the `try/except ValueError → ClickException` around it and the `VerdictCache(default_directory()) if cache and
|
|
180
|
+
- [x] **Step 4: Rewire `cli.py`'s `detect`.** Replace the three calls with one `analyse(...)`; keep the `try/except ValueError → ClickException` around it and the `VerdictCache(default_directory()) if cache else None` argument — the cache is opened whenever `--cache` allows it and never gated on `escalate`, which governs whether new stage-3 work may be commissioned and not whether a verdict already paid for may be read. Keep `_parse_since` as a thin alias if anything else imports it; delete it if nothing does.
|
|
181
181
|
|
|
182
182
|
- [x] **Step 5: Rewire `sync_detect.py`'s `_collect_detect_run`.** One `analyse(...)` call; keep `SyncError` translation, keep the working-directory collection and the per-kind activity walk, and read `window_start`/`window_end`/`collection_failures` off the `Analysis` instead of computing them.
|
|
183
183
|
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# 0.2.3 — stage three is opt-in
|
|
2
|
+
|
|
3
|
+
`stacktrace detect` no longer sends anything session-derived off your machine
|
|
4
|
+
unless you ask it to. The third stage — the one that hands flagged sessions to
|
|
5
|
+
the agent's own CLI — is now behind `--escalate`, which is what the quickstart,
|
|
6
|
+
the data-handling page and three of the four code surfaces already said it was.
|
|
7
|
+
Alongside it, a fix worth reading if you installed with `install.sh`: `correlate`
|
|
8
|
+
was reaching whichever OpenACA your PATH happened to name rather than the one
|
|
9
|
+
this package pins.
|
|
10
|
+
|
|
11
|
+
## Highlights
|
|
12
|
+
|
|
13
|
+
- **`stacktrace detect` defaults `--escalate` off.** A bare run is the two
|
|
14
|
+
stages that need no model and no credential, plus the osv.dev advisory
|
|
15
|
+
lookup, which sends package coordinates only. `--escalate` turns the third
|
|
16
|
+
stage on and `--budget` still caps it. The default had been on since
|
|
17
|
+
2026-09-03 while `run_detector`, `stacktrace monitor` and `remote sync detect`
|
|
18
|
+
all defaulted it off and the published quickstart told new users to pass
|
|
19
|
+
`--no-escalate`; one surface out of four, contradicted by its own
|
|
20
|
+
documentation, was a divergence rather than a default. ADR-0029 records the
|
|
21
|
+
decision and is explicit that it is a consent decision taken ahead of the
|
|
22
|
+
verdict-yield evidence ADR-0004 asked for, which is still unmeasured.
|
|
23
|
+
|
|
24
|
+
A run that declines to escalate is not a clean run. Sessions that qualified
|
|
25
|
+
are counted in the reasoning accounting as `reasoning_not_requested`; the
|
|
26
|
+
session ids are in `--format json`, not the text output, which aggregates
|
|
27
|
+
them under the reason.
|
|
28
|
+
|
|
29
|
+
- **A verdict you already paid for is readable on the default path.** Both
|
|
30
|
+
`detect` and `remote sync detect` built their verdict cache only when
|
|
31
|
+
escalation was on, so with the new default every stored verdict would have
|
|
32
|
+
gone unread and the same sessions would have been billed twice on the next
|
|
33
|
+
`--escalate` run. `run_detector` has always held that the flag governs whether
|
|
34
|
+
new verdicts may be commissioned, not whether old ones may be read; the two
|
|
35
|
+
call sites disagreed with it. An exhaustiveness test now covers every cache
|
|
36
|
+
construction site rather than the two that were found by hand.
|
|
37
|
+
|
|
38
|
+
- **`correlate` runs the OpenACA this package pins.** Acquisition invoked the
|
|
39
|
+
bare name `openaca`, a PATH lookup — and `uv tool install stacktrace-cli`,
|
|
40
|
+
which is what `install.sh` runs, links only the `stacktrace` entry point, so
|
|
41
|
+
the pinned OpenACA sat one directory away, reachable but unnamed. A machine
|
|
42
|
+
with a stale global copy silently scanned with the wrong version; a machine
|
|
43
|
+
with none got a bare `FileNotFoundError`. `stacktrace --version` could not
|
|
44
|
+
reveal it, because it reports the metadata inside the environment while the
|
|
45
|
+
subprocess ran something else. The console script is now located through
|
|
46
|
+
`importlib.metadata`, authenticated against the digest its own install
|
|
47
|
+
recorded, and fails closed with a legible error where ownership cannot be
|
|
48
|
+
established. Verified against four real install topologies. The ADR-0007 seam
|
|
49
|
+
is unchanged.
|
|
50
|
+
|
|
51
|
+
- **One default, stated once.** The escalation default had also drifted into
|
|
52
|
+
`docs/specs/detector.md`, `docs/specs/detection-upload.md`, test descriptions,
|
|
53
|
+
wrapped help text, and `docs/plans/008-monitor.md`, which still instructed the
|
|
54
|
+
cache gate this release removes. A test now reads the active specs, plans and
|
|
55
|
+
help text and fails on a stale wording, in any of the three names the repo
|
|
56
|
+
uses for the stage.
|
|
57
|
+
|
|
58
|
+
- **The console names the log column `Agent Session Trace`,** dropping the
|
|
59
|
+
third word from `Agent Session Log Trace` — session, log and trace were three
|
|
60
|
+
words for one thing. The one-item legend above the columns (`MCP = a link to
|
|
61
|
+
an outside service`) is gone: an MCP row already carries its category tag and
|
|
62
|
+
names the server it reached.
|
|
63
|
+
|
|
64
|
+
- **`detect` stops printing the per-reason "Could not settle" breakdown.** It
|
|
65
|
+
listed internal reason ids — `mcp_transport_unavailable` dominates on a normal
|
|
66
|
+
machine — at a reader who cannot act on them one call at a time. The unsettled
|
|
67
|
+
count stays in the summary line, `--format json` still carries every `Unknown`
|
|
68
|
+
whole with its spans and detail, and the reasoning stage still names its own
|
|
69
|
+
outcomes.
|
|
70
|
+
|
|
71
|
+
## Install
|
|
72
|
+
|
|
73
|
+
`uv tool install stacktrace-cli==0.2.3` or `pip install stacktrace-cli==0.2.3`.
|
|
74
|
+
|
|
75
|
+
## Compatibility
|
|
76
|
+
|
|
77
|
+
**One behaviour change, and the patch version does not signal it: `stacktrace
|
|
78
|
+
detect` no longer escalates by default.** If you relied on a bare `detect`
|
|
79
|
+
reaching stage 3, pass `--escalate` to restore it. Nothing breaks — the
|
|
80
|
+
`--no-escalate` you may already be passing still means what it meant — but a
|
|
81
|
+
default run now produces no reasoning-stage findings, and sessions that would
|
|
82
|
+
have been analysed appear as `reasoning_not_requested` instead. `monitor` and
|
|
83
|
+
`remote sync detect` are unaffected; they already defaulted off.
|
|
84
|
+
|
|
85
|
+
If you had accumulated verdicts from earlier `--escalate` runs, they are served
|
|
86
|
+
again on the default path rather than ignored, so a `detect` after upgrading may
|
|
87
|
+
report reasoning findings without spending anything.
|
|
88
|
+
|
|
89
|
+
Text output changes in two places: the `Could not settle:` block is gone, and
|
|
90
|
+
the reasoning accounting's not-requested row names a flag rather than a reason
|
|
91
|
+
id. `--format json` is unchanged — no field added, removed or renamed — and the
|
|
92
|
+
upload payload is untouched, so anything parsing either sees what 0.2.2 served.
|
|
93
|
+
|
|
94
|
+
On the console, the log column is renamed and the legend line is removed. No
|
|
95
|
+
rule, finding, severity or payload field moves.
|
|
96
|
+
|
|
97
|
+
If `install.sh` put stacktrace on your machine and you had separately installed
|
|
98
|
+
or upgraded OpenACA to work around a scan failure, that workaround is no longer
|
|
99
|
+
needed; the pinned 0.6.0 is what runs now.
|
|
100
|
+
|
|
101
|
+
Still pre-alpha, with no back-compat hedging: interfaces can change between
|
|
102
|
+
releases without a deprecation window.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# 0.3.0 — the third stage is asked for by name
|
|
2
|
+
|
|
3
|
+
`--escalate` is now `--reasoning`. The old name described the pipeline's own
|
|
4
|
+
mechanic — stage two promoting a session to stage three — rather than what you
|
|
5
|
+
get for it, which is a reasoning model reading the session. The flag now says
|
|
6
|
+
that, on `stacktrace detect`, `stacktrace monitor` and `remote sync detect`
|
|
7
|
+
alike.
|
|
8
|
+
|
|
9
|
+
**This release renames a flag and two JSON fields with no aliases.** Everything
|
|
10
|
+
you need to change is in the table below; nothing else about the pipeline moved,
|
|
11
|
+
and the defaults are exactly what 0.2.3 shipped.
|
|
12
|
+
|
|
13
|
+
## What to change
|
|
14
|
+
|
|
15
|
+
| 0.2.3 | 0.3.0 |
|
|
16
|
+
| --- | --- |
|
|
17
|
+
| `detect --escalate` | `detect --reasoning` |
|
|
18
|
+
| `monitor --escalate` | `monitor --reasoning` |
|
|
19
|
+
| `remote sync detect --escalate` | `remote sync detect --reasoning` |
|
|
20
|
+
| `--no-escalate` | *(nothing — omit `--reasoning`)* |
|
|
21
|
+
| `summary.escalated` | `summary.reasoning_requested` |
|
|
22
|
+
| `unknowns[].escalated_on` | `unknowns[].reasoning_reasons` |
|
|
23
|
+
| `POST /api/escalate` | `POST /api/reasoning` |
|
|
24
|
+
|
|
25
|
+
**If you typed `--no-escalate`, ignore what the CLI suggests.** Click offers
|
|
26
|
+
`--no-cache` as the nearest match, which is the wrong flag and an expensive one
|
|
27
|
+
to take: it disables the verdict cache, so a run re-buys answers you have
|
|
28
|
+
already paid for. There is no replacement for `--no-escalate`. Stage three has
|
|
29
|
+
been opt-in since 0.2.3, so leaving `--reasoning` off is the whole of it.
|
|
30
|
+
|
|
31
|
+
## Highlights
|
|
32
|
+
|
|
33
|
+
- **`--reasoning` is a plain flag, not an on/off pair.** `--no-reasoning` does
|
|
34
|
+
not exist, because there is nothing to turn off: stage three is opt-in on all
|
|
35
|
+
three commands (ADR-0029). A pair would imply a default worth overriding.
|
|
36
|
+
|
|
37
|
+
- **The vocabulary moved with the flag, so one word covers the whole path.**
|
|
38
|
+
The stage that answers was already called `reasoning`; the signal that reaches
|
|
39
|
+
it is now a `ReasoningRequest` rather than an `Escalation`. `--help`, the text
|
|
40
|
+
output and `--format json` agree. A run that declined now reads `re-run with
|
|
41
|
+
--reasoning to analyse them`, and the summary line reads `3 for reasoning, 2
|
|
42
|
+
analysed`.
|
|
43
|
+
|
|
44
|
+
- **Your stored verdicts survive the upgrade.** The verdict cache is keyed on
|
|
45
|
+
what the prompt is a function of, not on the names this codebase uses
|
|
46
|
+
internally, so a session graded under 0.2.3 is still answered from cache under
|
|
47
|
+
0.3.0. The stage-three prompt is unchanged and still recorded as `v1`, which
|
|
48
|
+
is what keeps two machines' answers comparable across the rename.
|
|
49
|
+
|
|
50
|
+
- **The monitor's page and route follow.** `POST /api/reasoning` replaces
|
|
51
|
+
`POST /api/escalate`, and the page's button reads against the same budget it
|
|
52
|
+
always did. A monitor and a CLI of the same version are required, as before —
|
|
53
|
+
the page is served by the process it is talking to.
|
|
@@ -42,8 +42,8 @@ version in a named manifest.
|
|
|
42
42
|
| A discovery tool | OpenACA's existing scan already inventories agent tooling |
|
|
43
43
|
| A fleet service | Only detector findings may travel, agent-keyed and content-free; sessions never leave the machine (ADR-0006) |
|
|
44
44
|
|
|
45
|
-
One exception, named rather than rounded away:
|
|
46
|
-
|
|
45
|
+
One exception, named rather than rounded away: requesting reasoning on a session
|
|
46
|
+
sends it to the provider that session already came from. See
|
|
47
47
|
[Trust boundary](#trust-boundary).
|
|
48
48
|
|
|
49
49
|
## Core tenets
|
|
@@ -56,7 +56,7 @@ enforced rather than merely intended.
|
|
|
56
56
|
|---|---|---|
|
|
57
57
|
| **Trust** — sensitive data does not cross the trust boundary | Transcripts, correlation records and session models are read only on the machine that produced them and have no upload path. What may travel is a detector finding: a verdict, component coordinates, confidence and observed capabilities, keyed to the agent — never conversation, and never an evidence excerpt | ADR-0006, and the local-only markers carried in the session model |
|
|
58
58
|
| **Quality** — precision first; recall from a second source | A false alarm spends a human hour, so precision is a product decision, not a tuning parameter. Recall is bought by **adding an independent kind of evidence**, never by loosening a threshold | The detector's principles; the match outcome carried through correlation |
|
|
59
|
-
| **Cost** — scalable unit economics | Deterministic rules first, then scoring against the component graph. Neither uses a model. **Only what they cannot resolve reaches inference** | The three-stage detector; no local model; a bounded
|
|
59
|
+
| **Cost** — scalable unit economics | Deterministic rules first, then scoring against the component graph. Neither uses a model. **Only what they cannot resolve reaches inference** | The three-stage detector; no local model; a bounded reasoning budget |
|
|
60
60
|
| **Security** — read-only sensing | The JSONL and SQLite caches agents already write, opened read-only. No process hooks, no proxy, no injected agent, no listener in the sensing path | Collection design; ADR-0001 |
|
|
61
61
|
|
|
62
62
|
Two of these deserve their consequences stated plainly, because they are the ones
|
|
@@ -151,7 +151,7 @@ Each is specified where its component is, alongside the sub-component it joins:
|
|
|
151
151
|
| Contract | Inside | Joins | Specified in |
|
|
152
152
|
|---|---|---|---|
|
|
153
153
|
| **Collection interface** | OpenAIDR | The per-kind readers to OpenAIDR's collector | [One contract, per agent kind](https://github.com/open-agent-security/openaidr/blob/main/docs/specs/session-collection.md#one-contract-per-agent-kind) |
|
|
154
|
-
| **
|
|
154
|
+
| **Reasoning request** | Detector | `priors` to `reasoning` — the detector's second and third stages | [The reasoning request](detector.md#the-reasoning-request) |
|
|
155
155
|
| **Analyzer adapter** | Detector | `reasoning` to the per-kind command-line agent it invokes | [The analyzer adapter](detector.md#the-analyzer-adapter) |
|
|
156
156
|
|
|
157
157
|
The per-kind rows are pluggable slots, not components: supporting another agent
|
|
@@ -223,7 +223,7 @@ go; session input, the Correlator and the Detector do identical work in both.
|
|
|
223
223
|
| Serves | Production: agent-keyed findings to Stacktrace Cloud (ADR-0006) | Development and demonstration: the local trace console |
|
|
224
224
|
| Trigger | One run over a window of sessions | Changed files, as agents work |
|
|
225
225
|
| Emit | Run to quiescence, emit complete findings | Draw a span when parsed, enrich it in place |
|
|
226
|
-
|
|
|
226
|
+
| Reasoning request | Looser cap, off-peak | Tight cap, background, never inline |
|
|
227
227
|
|
|
228
228
|
**Mode lives in the driver, never inside a component.** No component can ask which
|
|
229
229
|
mode it is running in, and nothing in the session model, correlation record or
|
|
@@ -235,12 +235,12 @@ cold pass without the tail — which is what stops two modes becoming two pipeli
|
|
|
235
235
|
| Collect, project to spans | sub-second per changed file | Every change, or once per scheduled run |
|
|
236
236
|
| Correlate | fast **only** against a cached graph | Graph rebuilt on manifest change |
|
|
237
237
|
| `deterministic`, `priors` | milliseconds per evaluation | On each new span, against the session's accumulated evidence — never one span in isolation |
|
|
238
|
-
| `reasoning` | seconds | Background, flagged sessions only, within the
|
|
238
|
+
| `reasoning` | seconds | Background, flagged sessions only, within the reasoning budget |
|
|
239
239
|
|
|
240
240
|
**Rule: nothing waits for the slowest stage.** Interactively, a tool call appears
|
|
241
241
|
when parsed and its outcome, identity and findings attach later as enrichments to
|
|
242
242
|
a row that already exists. On a scheduled run the same rule means one slow
|
|
243
|
-
|
|
243
|
+
reasoning request does not hold the whole run.
|
|
244
244
|
|
|
245
245
|
Three consequences that would otherwise look arbitrary. None of them is a
|
|
246
246
|
concession to the console — each holds in both modes:
|
|
@@ -259,7 +259,7 @@ concession to the console — each holds in both modes:
|
|
|
259
259
|
to re-read. Incremental collection is what makes that affordable: full pass at
|
|
260
260
|
cold start, then changed files only.
|
|
261
261
|
|
|
262
|
-
**The
|
|
262
|
+
**The reasoning budget is the one genuine per-mode parameter**, and the driver
|
|
263
263
|
sets it. `reasoning` is the only stage that costs seconds, a network call and
|
|
264
264
|
money, so how much of the flagged set may reach it is a consumer decision, not a
|
|
265
265
|
property of the detector. Everything else about the two modes is scheduling.
|
|
@@ -291,7 +291,7 @@ Detecting before registering keeps a failed run from creating an asset.
|
|
|
291
291
|
|---|---|---|
|
|
292
292
|
| `--agent-kind`, `--bom`, `--budget`, `--sample-budget`, `--cache/--no-cache`, `--project-map`, `--root` | as `stacktrace detect` | |
|
|
293
293
|
| `--since` | `7d` | |
|
|
294
|
-
| `--
|
|
294
|
+
| `--reasoning` | **off** | Stage 3 spends the developer's provider quota and crosses a boundary; unattended is the wrong place for that to be implicit |
|
|
295
295
|
| `--dry-run` | off | NDJSON via the same builder the real path uses, so the preview **is** the payload. ADR-0006 constraint 4 requires it |
|
|
296
296
|
| `--quiet`, `--allow-offline-cache` | off | as `sync endpoint` |
|
|
297
297
|
|