atif-scan 0.7.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- atif_scan-0.7.3/.github/dependabot.yml +22 -0
- atif_scan-0.7.3/.github/workflows/ci.yml +21 -0
- atif_scan-0.7.3/.gitignore +23 -0
- atif_scan-0.7.3/AGENTS.md +21 -0
- atif_scan-0.7.3/LICENSE +21 -0
- atif_scan-0.7.3/PKG-INFO +12 -0
- atif_scan-0.7.3/README.md +111 -0
- atif_scan-0.7.3/SECURITY.md +153 -0
- atif_scan-0.7.3/docs/design.md +248 -0
- atif_scan-0.7.3/docs/detectors.md +645 -0
- atif_scan-0.7.3/docs/improvement-loop.md +156 -0
- atif_scan-0.7.3/docs/private-browser.md +168 -0
- atif_scan-0.7.3/docs/reports.md +466 -0
- atif_scan-0.7.3/docs/review.md +330 -0
- atif_scan-0.7.3/docs/runs.md +348 -0
- atif_scan-0.7.3/docs/usage-accounting.md +53 -0
- atif_scan-0.7.3/examples/demo_pack.py +43 -0
- atif_scan-0.7.3/examples/policy.json +71 -0
- atif_scan-0.7.3/examples/synthetic.json +56 -0
- atif_scan-0.7.3/pyproject.toml +102 -0
- atif_scan-0.7.3/src/atif_scan/__init__.py +55 -0
- atif_scan-0.7.3/src/atif_scan/__main__.py +3 -0
- atif_scan-0.7.3/src/atif_scan/access.py +82 -0
- atif_scan-0.7.3/src/atif_scan/browser/__init__.py +1 -0
- atif_scan-0.7.3/src/atif_scan/browser/answers.py +93 -0
- atif_scan-0.7.3/src/atif_scan/browser/export.py +458 -0
- atif_scan-0.7.3/src/atif_scan/browser/feedback.py +202 -0
- atif_scan-0.7.3/src/atif_scan/browser/findings.py +170 -0
- atif_scan-0.7.3/src/atif_scan/browser/focus.py +59 -0
- atif_scan-0.7.3/src/atif_scan/browser/highlights/highlights.css +53 -0
- atif_scan-0.7.3/src/atif_scan/browser/highlights/highlights.js +186 -0
- atif_scan-0.7.3/src/atif_scan/browser/highlights/index.html +46 -0
- atif_scan-0.7.3/src/atif_scan/browser/highlights.py +202 -0
- atif_scan-0.7.3/src/atif_scan/browser/server.py +303 -0
- atif_scan-0.7.3/src/atif_scan/browser/session.py +326 -0
- atif_scan-0.7.3/src/atif_scan/browser/static/browser.css +62 -0
- atif_scan-0.7.3/src/atif_scan/browser/static/browser.js +543 -0
- atif_scan-0.7.3/src/atif_scan/browser/static/index.html +82 -0
- atif_scan-0.7.3/src/atif_scan/browser/tokens.css +44 -0
- atif_scan-0.7.3/src/atif_scan/browser/viewer/evidence.js +482 -0
- atif_scan-0.7.3/src/atif_scan/browser/viewer/index.html +67 -0
- atif_scan-0.7.3/src/atif_scan/browser/viewer/viewer.css +140 -0
- atif_scan-0.7.3/src/atif_scan/browser/viewer/viewer.js +725 -0
- atif_scan-0.7.3/src/atif_scan/browser/web.py +74 -0
- atif_scan-0.7.3/src/atif_scan/cache.py +125 -0
- atif_scan-0.7.3/src/atif_scan/checks.py +257 -0
- atif_scan-0.7.3/src/atif_scan/cli/__init__.py +79 -0
- atif_scan-0.7.3/src/atif_scan/cli/args.py +460 -0
- atif_scan-0.7.3/src/atif_scan/cli/bench.py +81 -0
- atif_scan-0.7.3/src/atif_scan/cli/browse.py +73 -0
- atif_scan-0.7.3/src/atif_scan/cli/emit.py +83 -0
- atif_scan-0.7.3/src/atif_scan/cli/fast_agent.py +80 -0
- atif_scan-0.7.3/src/atif_scan/cli/highlights.py +41 -0
- atif_scan-0.7.3/src/atif_scan/cli/hunt.py +180 -0
- atif_scan-0.7.3/src/atif_scan/cli/images.py +307 -0
- atif_scan-0.7.3/src/atif_scan/cli/inputs.py +260 -0
- atif_scan-0.7.3/src/atif_scan/cli/inspect.py +64 -0
- atif_scan-0.7.3/src/atif_scan/cli/labels.py +531 -0
- atif_scan-0.7.3/src/atif_scan/cli/scan.py +322 -0
- atif_scan-0.7.3/src/atif_scan/cli/viewer.py +78 -0
- atif_scan-0.7.3/src/atif_scan/data/__init__.py +19 -0
- atif_scan-0.7.3/src/atif_scan/data/account_ids.py +151 -0
- atif_scan-0.7.3/src/atif_scan/data/accounting.py +108 -0
- atif_scan-0.7.3/src/atif_scan/data/content.py +82 -0
- atif_scan-0.7.3/src/atif_scan/data/credentials.py +322 -0
- atif_scan-0.7.3/src/atif_scan/data/facts.py +431 -0
- atif_scan-0.7.3/src/atif_scan/data/jslit.py +486 -0
- atif_scan-0.7.3/src/atif_scan/data/jsonval.py +103 -0
- atif_scan-0.7.3/src/atif_scan/data/loader.py +424 -0
- atif_scan-0.7.3/src/atif_scan/data/media.py +94 -0
- atif_scan-0.7.3/src/atif_scan/data/model.py +454 -0
- atif_scan-0.7.3/src/atif_scan/data/pairing.py +131 -0
- atif_scan-0.7.3/src/atif_scan/data/paths.py +63 -0
- atif_scan-0.7.3/src/atif_scan/data/shell.py +579 -0
- atif_scan-0.7.3/src/atif_scan/data/submission.py +118 -0
- atif_scan-0.7.3/src/atif_scan/data/terminal.py +36 -0
- atif_scan-0.7.3/src/atif_scan/data/tools.py +481 -0
- atif_scan-0.7.3/src/atif_scan/data/usage.py +272 -0
- atif_scan-0.7.3/src/atif_scan/data/web_activity.py +157 -0
- atif_scan-0.7.3/src/atif_scan/data/web_gaps.py +69 -0
- atif_scan-0.7.3/src/atif_scan/data/web_inputs.py +77 -0
- atif_scan-0.7.3/src/atif_scan/data/web_results.py +234 -0
- atif_scan-0.7.3/src/atif_scan/detectors/__init__.py +13 -0
- atif_scan-0.7.3/src/atif_scan/detectors/account_ids.py +81 -0
- atif_scan-0.7.3/src/atif_scan/detectors/awareness.py +176 -0
- atif_scan-0.7.3/src/atif_scan/detectors/builtin.py +367 -0
- atif_scan-0.7.3/src/atif_scan/detectors/catalog.py +139 -0
- atif_scan-0.7.3/src/atif_scan/detectors/context.py +42 -0
- atif_scan-0.7.3/src/atif_scan/detectors/discovery.py +145 -0
- atif_scan-0.7.3/src/atif_scan/detectors/harness.py +35 -0
- atif_scan-0.7.3/src/atif_scan/detectors/installs.py +325 -0
- atif_scan-0.7.3/src/atif_scan/detectors/integrity.py +602 -0
- atif_scan-0.7.3/src/atif_scan/detectors/lookup.py +718 -0
- atif_scan-0.7.3/src/atif_scan/detectors/priming.py +158 -0
- atif_scan-0.7.3/src/atif_scan/detectors/provenance.py +125 -0
- atif_scan-0.7.3/src/atif_scan/detectors/proximity.py +133 -0
- atif_scan-0.7.3/src/atif_scan/detectors/recall.py +211 -0
- atif_scan-0.7.3/src/atif_scan/detectors/side_channel.py +379 -0
- atif_scan-0.7.3/src/atif_scan/detectors/tamper.py +298 -0
- atif_scan-0.7.3/src/atif_scan/detectors/text.py +294 -0
- atif_scan-0.7.3/src/atif_scan/detectors/vocabulary.py +190 -0
- atif_scan-0.7.3/src/atif_scan/engine.py +212 -0
- atif_scan-0.7.3/src/atif_scan/evidence/__init__.py +3 -0
- atif_scan-0.7.3/src/atif_scan/evidence/cite.py +283 -0
- atif_scan-0.7.3/src/atif_scan/evidence/extract.py +391 -0
- atif_scan-0.7.3/src/atif_scan/evidence/highlights.py +99 -0
- atif_scan-0.7.3/src/atif_scan/evidence/history.py +215 -0
- atif_scan-0.7.3/src/atif_scan/output/__init__.py +3 -0
- atif_scan-0.7.3/src/atif_scan/output/brief.py +649 -0
- atif_scan-0.7.3/src/atif_scan/output/brief_view/__init__.py +169 -0
- atif_scan-0.7.3/src/atif_scan/output/brief_view/colour.py +90 -0
- atif_scan-0.7.3/src/atif_scan/output/brief_view/evidence.py +310 -0
- atif_scan-0.7.3/src/atif_scan/output/brief_view/run.py +479 -0
- atif_scan-0.7.3/src/atif_scan/output/brief_view/usage.py +540 -0
- atif_scan-0.7.3/src/atif_scan/output/brief_view/words.py +225 -0
- atif_scan-0.7.3/src/atif_scan/output/bundle.py +202 -0
- atif_scan-0.7.3/src/atif_scan/output/document.py +367 -0
- atif_scan-0.7.3/src/atif_scan/output/estimates.py +684 -0
- atif_scan-0.7.3/src/atif_scan/output/overview.py +755 -0
- atif_scan-0.7.3/src/atif_scan/output/rich.py +181 -0
- atif_scan-0.7.3/src/atif_scan/output/summary.py +224 -0
- atif_scan-0.7.3/src/atif_scan/output/text.py +237 -0
- atif_scan-0.7.3/src/atif_scan/output/views.py +30 -0
- atif_scan-0.7.3/src/atif_scan/packs/__init__.py +152 -0
- atif_scan-0.7.3/src/atif_scan/packs/deepswe.py +486 -0
- atif_scan-0.7.3/src/atif_scan/packs/local_models.py +86 -0
- atif_scan-0.7.3/src/atif_scan/packs/markers.py +120 -0
- atif_scan-0.7.3/src/atif_scan/packs/reference.py +505 -0
- atif_scan-0.7.3/src/atif_scan/packs/tb21.py +890 -0
- atif_scan-0.7.3/src/atif_scan/packs/tb4.py +473 -0
- atif_scan-0.7.3/src/atif_scan/policy.py +97 -0
- atif_scan-0.7.3/src/atif_scan/review/__init__.py +2 -0
- atif_scan-0.7.3/src/atif_scan/review/answers.py +249 -0
- atif_scan-0.7.3/src/atif_scan/review/catalogue.py +1078 -0
- atif_scan-0.7.3/src/atif_scan/review/coverage.py +100 -0
- atif_scan-0.7.3/src/atif_scan/review/inspect_server.py +195 -0
- atif_scan-0.7.3/src/atif_scan/review/labels.py +447 -0
- atif_scan-0.7.3/src/atif_scan/review/prompts.py +729 -0
- atif_scan-0.7.3/src/atif_scan/review/provenance.py +227 -0
- atif_scan-0.7.3/src/atif_scan/rules.py +151 -0
- atif_scan-0.7.3/src/atif_scan/sources/__init__.py +35 -0
- atif_scan-0.7.3/src/atif_scan/sources/bench.py +151 -0
- atif_scan-0.7.3/src/atif_scan/sources/harbor/__init__.py +2 -0
- atif_scan-0.7.3/src/atif_scan/sources/harbor/files.py +499 -0
- atif_scan-0.7.3/src/atif_scan/sources/harbor/hub.py +684 -0
- atif_scan-0.7.3/src/atif_scan/sources/harbor/listing.py +220 -0
- atif_scan-0.7.3/src/atif_scan/sources/harbor/runs.py +353 -0
- atif_scan-0.7.3/src/atif_scan/sources/inputs.py +621 -0
- atif_scan-0.7.3/src/atif_scan/sources/layout.py +268 -0
- atif_scan-0.7.3/src/atif_scan/sources/selection.py +457 -0
- atif_scan-0.7.3/src/atif_scan/sources/sync.py +308 -0
- atif_scan-0.7.3/tests/browser_live.test.cjs +249 -0
- atif_scan-0.7.3/tests/browser_static.test.cjs +461 -0
- atif_scan-0.7.3/tests/conftest.py +76 -0
- atif_scan-0.7.3/tests/fixtures/prompts/tasks/demo-task/environment/Dockerfile +1 -0
- atif_scan-0.7.3/tests/fixtures/prompts/tasks/demo-task/instruction.md +1 -0
- atif_scan-0.7.3/tests/fixtures/prompts/tasks/demo-task/tests/test_outputs.py +7 -0
- atif_scan-0.7.3/tests/fixtures/prompts/trace.json +33 -0
- atif_scan-0.7.3/tests/golden/prompts/awareness_hunt.blind.md +150 -0
- atif_scan-0.7.3/tests/golden/prompts/awareness_hunt.md +173 -0
- atif_scan-0.7.3/tests/golden/prompts/fabrication_hunt.blind.md +131 -0
- atif_scan-0.7.3/tests/golden/prompts/fabrication_hunt.md +167 -0
- atif_scan-0.7.3/tests/golden/prompts/hack_hunt.blind.md +159 -0
- atif_scan-0.7.3/tests/golden/prompts/hack_hunt.deepswe.md +198 -0
- atif_scan-0.7.3/tests/golden/prompts/hack_hunt.md +195 -0
- atif_scan-0.7.3/tests/golden/prompts/harness_message_hint.md +99 -0
- atif_scan-0.7.3/tests/golden/prompts/history_probe.md +154 -0
- atif_scan-0.7.3/tests/golden/prompts/lookup_used.md +132 -0
- atif_scan-0.7.3/tests/golden/prompts/lure_response.md +132 -0
- atif_scan-0.7.3/tests/golden/prompts/model_call_legitimacy.md +98 -0
- atif_scan-0.7.3/tests/golden/prompts/network_outcome.md +164 -0
- atif_scan-0.7.3/tests/golden/prompts/recall_or_derivation.md +136 -0
- atif_scan-0.7.3/tests/golden/prompts/skill_task_specific.md +115 -0
- atif_scan-0.7.3/tests/golden/prompts/test_access_intent.md +137 -0
- atif_scan-0.7.3/tests/golden/prompts/value_provenance.md +142 -0
- atif_scan-0.7.3/tests/golden/prompts/verification_hunt.blind.md +155 -0
- atif_scan-0.7.3/tests/golden/prompts/verification_hunt.md +173 -0
- atif_scan-0.7.3/tests/golden/prompts/versions.json +82 -0
- atif_scan-0.7.3/tests/golden/prompts/web_provenance.md +179 -0
- atif_scan-0.7.3/tests/test_access.py +129 -0
- atif_scan-0.7.3/tests/test_account_ids.py +285 -0
- atif_scan-0.7.3/tests/test_accounting_clarity.py +301 -0
- atif_scan-0.7.3/tests/test_allowances_reports_sources.py +375 -0
- atif_scan-0.7.3/tests/test_architecture.py +152 -0
- atif_scan-0.7.3/tests/test_argument_accounting.py +123 -0
- atif_scan-0.7.3/tests/test_atif_defects.py +352 -0
- atif_scan-0.7.3/tests/test_awareness_proximity.py +108 -0
- atif_scan-0.7.3/tests/test_bench_runs.py +233 -0
- atif_scan-0.7.3/tests/test_benchmark_query_evidence.py +202 -0
- atif_scan-0.7.3/tests/test_brief.py +277 -0
- atif_scan-0.7.3/tests/test_brief_cache.py +659 -0
- atif_scan-0.7.3/tests/test_browser_answers.py +85 -0
- atif_scan-0.7.3/tests/test_browser_cli.py +354 -0
- atif_scan-0.7.3/tests/test_browser_feedback.py +229 -0
- atif_scan-0.7.3/tests/test_browser_focus.py +181 -0
- atif_scan-0.7.3/tests/test_browser_server.py +381 -0
- atif_scan-0.7.3/tests/test_browser_session.py +369 -0
- atif_scan-0.7.3/tests/test_browser_static.py +22 -0
- atif_scan-0.7.3/tests/test_browser_web_gaps.py +199 -0
- atif_scan-0.7.3/tests/test_cite_summary.py +556 -0
- atif_scan-0.7.3/tests/test_code_mode.py +263 -0
- atif_scan-0.7.3/tests/test_compacted_token_messaging.py +64 -0
- atif_scan-0.7.3/tests/test_cost_integrity.py +465 -0
- atif_scan-0.7.3/tests/test_cost_ledger.py +163 -0
- atif_scan-0.7.3/tests/test_coverage_integrity.py +856 -0
- atif_scan-0.7.3/tests/test_coverage_reporting.py +224 -0
- atif_scan-0.7.3/tests/test_credentials_context.py +297 -0
- atif_scan-0.7.3/tests/test_detectors_review.py +671 -0
- atif_scan-0.7.3/tests/test_downstream_provenance.py +243 -0
- atif_scan-0.7.3/tests/test_e2e.py +71 -0
- atif_scan-0.7.3/tests/test_environment_dump.py +82 -0
- atif_scan-0.7.3/tests/test_evaluation_discovery.py +146 -0
- atif_scan-0.7.3/tests/test_facts.py +104 -0
- atif_scan-0.7.3/tests/test_failed_trials.py +191 -0
- atif_scan-0.7.3/tests/test_fast_agent_runner.py +78 -0
- atif_scan-0.7.3/tests/test_finding_index.py +160 -0
- atif_scan-0.7.3/tests/test_grok_build_tokens.py +139 -0
- atif_scan-0.7.3/tests/test_grounded_accounting_claims.py +121 -0
- atif_scan-0.7.3/tests/test_harbor_accounting.py +293 -0
- atif_scan-0.7.3/tests/test_harbor_files.py +338 -0
- atif_scan-0.7.3/tests/test_harbor_overview.py +680 -0
- atif_scan-0.7.3/tests/test_harbor_retries.py +150 -0
- atif_scan-0.7.3/tests/test_harbor_setups.py +150 -0
- atif_scan-0.7.3/tests/test_harness_sensitivity.py +132 -0
- atif_scan-0.7.3/tests/test_highlights.py +119 -0
- atif_scan-0.7.3/tests/test_history_archives.py +353 -0
- atif_scan-0.7.3/tests/test_image_model.py +365 -0
- atif_scan-0.7.3/tests/test_inspect.py +382 -0
- atif_scan-0.7.3/tests/test_inspect_mcp.py +120 -0
- atif_scan-0.7.3/tests/test_installs.py +617 -0
- atif_scan-0.7.3/tests/test_jslit.py +125 -0
- atif_scan-0.7.3/tests/test_labels.py +331 -0
- atif_scan-0.7.3/tests/test_lookup_receipts.py +205 -0
- atif_scan-0.7.3/tests/test_measure.py +143 -0
- atif_scan-0.7.3/tests/test_observation_pairing.py +359 -0
- atif_scan-0.7.3/tests/test_pack_deepswe.py +358 -0
- atif_scan-0.7.3/tests/test_pack_tb21.py +963 -0
- atif_scan-0.7.3/tests/test_pack_tb4.py +392 -0
- atif_scan-0.7.3/tests/test_packs_auto.py +110 -0
- atif_scan-0.7.3/tests/test_packs_review.py +419 -0
- atif_scan-0.7.3/tests/test_paths.py +41 -0
- atif_scan-0.7.3/tests/test_priming.py +106 -0
- atif_scan-0.7.3/tests/test_prompt_golden.py +136 -0
- atif_scan-0.7.3/tests/test_questions.py +1106 -0
- atif_scan-0.7.3/tests/test_questions_network_model.py +433 -0
- atif_scan-0.7.3/tests/test_questions_web_provenance.py +277 -0
- atif_scan-0.7.3/tests/test_raw_input_tools.py +98 -0
- atif_scan-0.7.3/tests/test_reasoning_token_alias.py +67 -0
- atif_scan-0.7.3/tests/test_recall.py +663 -0
- atif_scan-0.7.3/tests/test_report_review.py +315 -0
- atif_scan-0.7.3/tests/test_reprice.py +174 -0
- atif_scan-0.7.3/tests/test_review.py +355 -0
- atif_scan-0.7.3/tests/test_review_evidence.py +117 -0
- atif_scan-0.7.3/tests/test_scan.py +621 -0
- atif_scan-0.7.3/tests/test_scoped_usage.py +230 -0
- atif_scan-0.7.3/tests/test_selection.py +327 -0
- atif_scan-0.7.3/tests/test_shell.py +146 -0
- atif_scan-0.7.3/tests/test_shell_search.py +184 -0
- atif_scan-0.7.3/tests/test_side_channel.py +291 -0
- atif_scan-0.7.3/tests/test_sources_review.py +347 -0
- atif_scan-0.7.3/tests/test_submission.py +159 -0
- atif_scan-0.7.3/tests/test_sync.py +166 -0
- atif_scan-0.7.3/tests/test_sync_integrity.py +339 -0
- atif_scan-0.7.3/tests/test_tamper.py +431 -0
- atif_scan-0.7.3/tests/test_tb4_local_models.py +170 -0
- atif_scan-0.7.3/tests/test_terminal_screens.py +133 -0
- atif_scan-0.7.3/tests/test_titles.py +129 -0
- atif_scan-0.7.3/tests/test_tools.py +88 -0
- atif_scan-0.7.3/tests/test_totals_last_call.py +91 -0
- atif_scan-0.7.3/tests/test_trial_ledger.py +120 -0
- atif_scan-0.7.3/tests/test_unread.py +89 -0
- atif_scan-0.7.3/tests/test_value_provenance.py +114 -0
- atif_scan-0.7.3/tests/test_viewer_export.py +543 -0
- atif_scan-0.7.3/tests/test_walltime.py +121 -0
- atif_scan-0.7.3/tests/test_web_activity_reporting.py +336 -0
- atif_scan-0.7.3/tests/test_web_reference_provenance.py +154 -0
- atif_scan-0.7.3/tests/test_web_result_wrappers.py +225 -0
- atif_scan-0.7.3/tests/viewer_live.test.cjs +166 -0
- atif_scan-0.7.3/tests/viewer_static.test.cjs +293 -0
- atif_scan-0.7.3/tools/finding_index.py +77 -0
- atif_scan-0.7.3/tools/gold.py +144 -0
- atif_scan-0.7.3/tools/judge_review/index.html +38 -0
- atif_scan-0.7.3/tools/judge_review/review.css +48 -0
- atif_scan-0.7.3/tools/judge_review/review.js +197 -0
- atif_scan-0.7.3/tools/judge_review.py +195 -0
- atif_scan-0.7.3/tools/reprice.py +356 -0
- atif_scan-0.7.3/tools/research/README.md +23 -0
- atif_scan-0.7.3/tools/research/tb21_eval.py +260 -0
- atif_scan-0.7.3/tools/research/tb21_inventory.py +253 -0
- atif_scan-0.7.3/tools/research/tb21_judge.py +308 -0
- atif_scan-0.7.3/tools/research/tb21_labels.py +142 -0
- atif_scan-0.7.3/tools/research/tb4_hunt.py +187 -0
- atif_scan-0.7.3/tools/research/tb4_inventory.py +326 -0
- atif_scan-0.7.3/tools/task_diff.py +218 -0
- atif_scan-0.7.3/tools/viewer_shots.py +197 -0
- atif_scan-0.7.3/uv.lock +1085 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Added by hf-security-analysis when this repository was created.
|
|
2
|
+
# Actions are pinned to commit SHAs across the organisation; a pinned
|
|
3
|
+
# SHA is only safe while something raises it, and that something is
|
|
4
|
+
# Dependabot. Edit this file freely — the bot writes it once and does
|
|
5
|
+
# not come back.
|
|
6
|
+
version: 2
|
|
7
|
+
updates:
|
|
8
|
+
- package-ecosystem: "github-actions"
|
|
9
|
+
directory: "/"
|
|
10
|
+
schedule:
|
|
11
|
+
interval: "weekly"
|
|
12
|
+
# Releases younger than this are not adopted. The hours right after a
|
|
13
|
+
# maintainer account is compromised are exactly when a fresh release
|
|
14
|
+
# is most dangerous — tj-actions and the reviewdog actions both
|
|
15
|
+
# travelled that way.
|
|
16
|
+
cooldown:
|
|
17
|
+
default-days: 7
|
|
18
|
+
# One pull request a week for all of them, rather than one per action.
|
|
19
|
+
groups:
|
|
20
|
+
actions:
|
|
21
|
+
patterns:
|
|
22
|
+
- "*"
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: checks
|
|
2
|
+
on: [push, pull_request]
|
|
3
|
+
permissions:
|
|
4
|
+
contents: read
|
|
5
|
+
jobs:
|
|
6
|
+
test:
|
|
7
|
+
runs-on: ubuntu-latest
|
|
8
|
+
strategy:
|
|
9
|
+
matrix:
|
|
10
|
+
python: ['3.11', '3.14']
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
13
|
+
- uses: astral-sh/setup-uv@c18668ad3cf93ea998bef934396af7bb5c839dc7 # v10.2.0
|
|
14
|
+
with:
|
|
15
|
+
python-version: ${{ matrix.python }}
|
|
16
|
+
- run: uv sync --locked --group dev
|
|
17
|
+
- run: uv run pytest -q
|
|
18
|
+
- run: uv run ruff check .
|
|
19
|
+
- run: uv run ruff format --check .
|
|
20
|
+
- run: uv run ty check
|
|
21
|
+
- run: uv build
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
.venv/
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
.pytest_cache/
|
|
5
|
+
.ruff_cache/
|
|
6
|
+
dist/
|
|
7
|
+
.env
|
|
8
|
+
*.auth.json
|
|
9
|
+
traces/
|
|
10
|
+
reports/
|
|
11
|
+
# Harbor writes job output to ./jobs when run from here: real traces.
|
|
12
|
+
/jobs/
|
|
13
|
+
.fast-agent/
|
|
14
|
+
|
|
15
|
+
# Private review output belongs outside Git; these patterns are only a backstop.
|
|
16
|
+
/review/
|
|
17
|
+
/reviews/
|
|
18
|
+
/judge/
|
|
19
|
+
/judges/
|
|
20
|
+
/j[0-9]*/
|
|
21
|
+
/q/
|
|
22
|
+
*.answer.json
|
|
23
|
+
*.review.atif.json
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# atif-scan development
|
|
2
|
+
|
|
3
|
+
<!-- fast-agent subagents -->
|
|
4
|
+
|
|
5
|
+
Read README.md and SECURITY.md first. Keep the core (`data`, `checks`/`rules`/`policy`/`engine`, `detectors`, `packs`) stdlib-only
|
|
6
|
+
and offline; `huggingface_hub` and `rich` are imported lazily, only by `sources` (hf://
|
|
7
|
+
inputs) and `output` (`rich`, `brief_view`). `tests/test_architecture.py` enforces the layers.
|
|
8
|
+
Do not execute trajectory commands or URLs. Never commit real traces, auth files,
|
|
9
|
+
raw findings or copied benchmark solutions. Use synthetic fixtures only.
|
|
10
|
+
|
|
11
|
+
Keep model/parsing, detector predicates, rule logic, evaluation, and report export
|
|
12
|
+
separate. Plugins return typed results. Reports are explicitly allowlisted and
|
|
13
|
+
contain no snippets. Unknown evidence is not a negative result. Severity is review
|
|
14
|
+
priority, not cheating probability. Make task selection explicit.
|
|
15
|
+
|
|
16
|
+
Run `uv run pytest -q`, `uv run ruff check .`, `uv run ruff format --check .` and
|
|
17
|
+
`uv run ty check` before committing; all four must be clean (tool versions are pinned
|
|
18
|
+
in pyproject.toml). Fix complexity by splitting into named helpers rather than
|
|
19
|
+
suppressing; a `# noqa`/`# ty: ignore` needs a rule code and a reason. Untrusted JSON is
|
|
20
|
+
narrowed with `jsonval`; report documents are `jsonval.Doc`. Add regression tests whenever a
|
|
21
|
+
new false-positive or evidence gap is found. No external model runs are needed.
|
atif_scan-0.7.3/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Hugging Face
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
atif_scan-0.7.3/PKG-INFO
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: atif-scan
|
|
3
|
+
Version: 0.7.3
|
|
4
|
+
Summary: Composable, evidence-scoped behavioral checks over ATIF trajectories
|
|
5
|
+
Project-URL: Repository, https://github.com/huggingface/atif-scan
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Requires-Python: >=3.11
|
|
9
|
+
Requires-Dist: huggingface-hub>=2.0.0
|
|
10
|
+
Requires-Dist: rich>=15.0.0
|
|
11
|
+
Provides-Extra: mcp
|
|
12
|
+
Requires-Dist: mcp<2,>=1.2; extra == 'mcp'
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
# atif-scan
|
|
2
|
+
|
|
3
|
+
Plugin-style detectors and simple rules over ATIF agent trajectories. The analysis core
|
|
4
|
+
is offline, standard-library Python. The CLI adds `huggingface_hub` (only for Hub
|
|
5
|
+
inputs) and `rich` (text output). It never runs anything found in a trace, and never calls a model unless you
|
|
6
|
+
pass `--image-model`.
|
|
7
|
+
|
|
8
|
+
> Findings are **review candidates, not verdicts**. A severity is a review priority,
|
|
9
|
+
> not a probability of cheating. Unknown evidence is never treated as a clean result.
|
|
10
|
+
|
|
11
|
+
## Quick start
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
uv sync
|
|
15
|
+
uv run atif-scan examples/synthetic.json # text on a terminal, JSON when piped
|
|
16
|
+
uv run atif-scan examples/synthetic.json --format json > report.json
|
|
17
|
+
|
|
18
|
+
# Directories and Hub paths expand to every trajectory.json below them:
|
|
19
|
+
uv run atif-scan /external/jobs/run-1/
|
|
20
|
+
uv run atif-scan hf://buckets/my-org/traces/run-1/
|
|
21
|
+
uv run atif-scan harbor://jobs/<job-id>
|
|
22
|
+
uv run atif-scan https://huggingface.co/buckets/my-org/traces/tree/run-1 # pasted web URL
|
|
23
|
+
|
|
24
|
+
# Add a task-specific detector pack, rules and allowances:
|
|
25
|
+
PYTHONPATH=examples uv run atif-scan examples/synthetic.json \
|
|
26
|
+
--task demo-pytest --plugin demo_pack:checks --rules examples/policy.json
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
For a run, the default text view is the **brief**: one screen covering coverage, score
|
|
30
|
+
(with a "flagged successes failed" scenario), review candidates, cost and walltime. A
|
|
31
|
+
single trace gets the per-check **detail** view. Remote inputs are mirrored privately to
|
|
32
|
+
the atif-scan home (`~/.cache/atif-scan`), so rescans only fetch what changed.
|
|
33
|
+
|
|
34
|
+
Then, as needed:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
atif-scan JOB --cite # finding rows with masked trace excerpts
|
|
38
|
+
atif-scan JOB --highlights DIR # excerpts of where the interesting things happened
|
|
39
|
+
atif-scan JOB --browse # private localhost evidence browser + feedback
|
|
40
|
+
atif-scan JOB --questions DIR # review bundle for the flagged successes
|
|
41
|
+
atif-scan hunt --model MODEL --questions DIR # answer it with fast-agent (a provider call)
|
|
42
|
+
atif-scan JOB --answers DIR --brief # answers as annotations, never rescoring
|
|
43
|
+
atif-inspect TRIAL_DIR --step 12 # read one step without parsing ATIF
|
|
44
|
+
atif-scan JOB --image-model MODEL # transcribe images that block a check
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
`atif-scan labels` manages the label store that measures checks and judges. To scan an
|
|
48
|
+
input actually named `labels` or `hunt`, write `./labels` or `./hunt`.
|
|
49
|
+
|
|
50
|
+
## How it works
|
|
51
|
+
|
|
52
|
+
```text
|
|
53
|
+
path / dir / hf:// / harbor:// → sources → loader → Trace → detectors ─┐
|
|
54
|
+
├→ engine → report → JSON | text
|
|
55
|
+
rules, allowances ────┘ (no text snippets)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Components come in two flavours. **Detectors and rules** flag things worth reviewing
|
|
59
|
+
(outside access, test-path reads, odd timestamps). **Allowances** say a finding is
|
|
60
|
+
expected for this trial ("network access is part of this task"), so it is shown as
|
|
61
|
+
`expected` and left out of the score, but not hidden.
|
|
62
|
+
|
|
63
|
+
By default detectors only see **agent-authored** text: messages, reasoning, and tool-call
|
|
64
|
+
arguments (commands, paths, URLs, queries). Prompts, copied context, tool outputs and
|
|
65
|
+
argument payloads (file contents, edits) are left out unless a check says otherwise.
|
|
66
|
+
|
|
67
|
+
## Documentation
|
|
68
|
+
|
|
69
|
+
| Topic | Where |
|
|
70
|
+
|---|---|
|
|
71
|
+
| Inputs, sync, bench-run cohorts, Harbor Hub, manifests, the atif-scan home | [docs/runs.md](docs/runs.md) |
|
|
72
|
+
| Report views (brief, summary, overview, detail), citations, the browser and the static viewer | [docs/reports.md](docs/reports.md) |
|
|
73
|
+
| Built-in detectors, task packs, writing detectors, rules and allowances | [docs/detectors.md](docs/detectors.md) |
|
|
74
|
+
| Review bundles, the questions, blind hunts, answers, `hunt`, `atif-inspect` | [docs/review.md](docs/review.md) |
|
|
75
|
+
| Usage and cost accounting | [docs/usage-accounting.md](docs/usage-accounting.md) |
|
|
76
|
+
| The private findings browser in depth | [docs/private-browser.md](docs/private-browser.md) |
|
|
77
|
+
| Design rules: evidence, unknowns, layering | [docs/design.md](docs/design.md) |
|
|
78
|
+
| Turning judge findings into checks, and measuring both | [docs/improvement-loop.md](docs/improvement-loop.md) |
|
|
79
|
+
| What is safe to share, and what never is | [SECURITY.md](SECURITY.md) |
|
|
80
|
+
|
|
81
|
+
## Development
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
uv run pytest -q && uv run ruff check . && uv run ruff format --check . && uv run ty check
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
All four must be clean. Only synthetic fixtures belong in this repository. Add a
|
|
88
|
+
regression test for every false positive or evidence gap you find. [AGENTS.md](AGENTS.md)
|
|
89
|
+
has the working rules and `tests/test_architecture.py` enforces the package layers.
|
|
90
|
+
|
|
91
|
+
**Golden prompts.** `tests/golden/prompts/` holds the exact text every review question
|
|
92
|
+
puts in front of a judge. After an intended prompt change, bump the question's version
|
|
93
|
+
and regenerate with `UPDATE_GOLDEN=1 uv run pytest tests/test_prompt_golden.py`, then
|
|
94
|
+
review the diff.
|
|
95
|
+
|
|
96
|
+
**Viewer screenshots.** Before a release, `uv run python tools/viewer_shots.py` exports the
|
|
97
|
+
synthetic example, screenshots it with headless Chrome (desktop and phone, light and dark)
|
|
98
|
+
and fails on a blank page, a page that doesn't open at its masthead, or a theme that didn't
|
|
99
|
+
apply. Screenshots go to the atif-scan home, for a person to look at.
|
|
100
|
+
|
|
101
|
+
**Gold masters.** Before changing rules, snapshot scans of real runs and diff after:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
uv run python tools/gold.py snapshot devin-run harbor://jobs/<id> # before
|
|
105
|
+
uv run python tools/gold.py diff devin-run harbor://jobs/<id> # after: per-check ± traces, DQ ±
|
|
106
|
+
uv run python tools/gold.py list
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Snapshots keep allowlisted fields only (input label, task, reward, check statuses, DQ
|
|
110
|
+
list), in the atif-scan home's `gold/` folder (or `$ATIF_SCAN_GOLD_DIR`). Labels are real
|
|
111
|
+
run identifiers, so never commit them.
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# Security and evidence handling
|
|
2
|
+
|
|
3
|
+
- Treat trajectories as untrusted and possibly credential-bearing. Keep traces, reports,
|
|
4
|
+
auth and env files out of Git. Use only synthetic fixtures, and never copy benchmark
|
|
5
|
+
solutions into this repository.
|
|
6
|
+
- Built-ins never execute commands, visit URLs, run models or fetch files.
|
|
7
|
+
- **`--image-model MODEL`** is the scan's one model call, and it is opt-in. It sends
|
|
8
|
+
only the images in trials where an image leaves a check unknown (decoded from the
|
|
9
|
+
trajectory, written to a `0700` temporary folder and deleted afterwards) to MODEL for
|
|
10
|
+
transcription, through `fast-agent go --isolated --no-shell --no-subagents` with
|
|
11
|
+
fixed arguments and no shell (see *fast-agent calls* below). Images may show anything
|
|
12
|
+
on the agent's screen, credentials included, so only use a provider you may send the
|
|
13
|
+
traces to. Transcripts are trace content: they are stored in the atif-scan home's
|
|
14
|
+
`images/` folder (`0700`/`0600`) and never reported (reports carry counts). The loader
|
|
15
|
+
keeps an inline payload's sha256, never the payload.
|
|
16
|
+
- Reports are an explicit field allowlist: no raw text, paths, URLs, argument values or
|
|
17
|
+
exception messages. The one report field with trace text is opt-in `--cite`: bounded,
|
|
18
|
+
secret-masked excerpts of the trace. Masking is best-effort, so treat cited output
|
|
19
|
+
like the trace itself and keep it out of Git and issues. OpenAI account identifiers
|
|
20
|
+
(`observation.account_*`) are masked the same way; a finding is a scrub reminder, and
|
|
21
|
+
its absence does not make a trace safe to publish. Text reports render only that allowlisted document. Manifest IDs,
|
|
22
|
+
check IDs and check titles must be non-sensitive labels.
|
|
23
|
+
- **Review bundles** (`--questions DIR`, formerly `--judge-prompts`) contain trace text
|
|
24
|
+
(bounded, masked excerpts) and local paths. Treat them like the trace itself: keep them
|
|
25
|
+
private and outside Git, in the atif-scan home's `bundles/` folder
|
|
26
|
+
(`~/.cache/atif-scan/bundles/` by default), not the source tree or the expendable
|
|
27
|
+
result cache. Use `umask 077` when generating or answering (directories `0700`, files
|
|
28
|
+
`0600`), and a fresh directory per selection and judge model, so stale questions can't
|
|
29
|
+
mix into a new review. Generating a bundle never sends data anywhere.
|
|
30
|
+
- **Answering** is the user's decision: whoever answers (e.g. `atif-scan hunt`) sends the
|
|
31
|
+
prompts to a model provider. Prompts frame trace text as untrusted data. `hunt` disables
|
|
32
|
+
shell and subagents; the optional `--inspect-tool` grants read-only MCP tools bound to
|
|
33
|
+
one local trajectory, never arbitrary paths or execution. In reports, `--answers` keeps
|
|
34
|
+
only the validated answer, confidence, mechanism and steps, never the free-text reason.
|
|
35
|
+
The judge's reason may quote the trace, so it appears only where trace text already
|
|
36
|
+
does: the static viewer and the private browser, masked as a whole with the trace's
|
|
37
|
+
own secrets (best effort), beside a current valid answer. Highlights and blind review
|
|
38
|
+
exports never carry it.
|
|
39
|
+
- **fast-agent calls** (`--image-model`, `hunt`) run `fast-agent go --isolated`
|
|
40
|
+
(fast-agent 0.10.43+): config, secrets and model aliases are read from the fast-agent
|
|
41
|
+
home, but nothing is written there (no session history, file logs or telemetry) and no
|
|
42
|
+
skills, agent cards, plugins, hooks, or shell, filesystem or subagent tools load. Only
|
|
43
|
+
the MCP server `hunt --inspect-tool` passes is started. An older fast-agent rejects the
|
|
44
|
+
flag: atif-scan then warns once and runs without it (`--no-shell --no-subagents`,
|
|
45
|
+
session history off through the environment), where the home's plugins and hooks may
|
|
46
|
+
still load. Upgrade rather than rely on that.
|
|
47
|
+
- **Companion history.** Full Harbor archives (`--full`) may include Grok compaction
|
|
48
|
+
segments and task artifacts. Companion-history tools inventory only fixed local
|
|
49
|
+
session locations beside the bound trajectory and never follow paths supplied by a
|
|
50
|
+
summary. Numeric file IDs, size limits, symlink rejection and whole-file masking
|
|
51
|
+
precede bounded reads and searches. Markdown remains untrusted data, not instructions
|
|
52
|
+
or validated ATIF reconstruction. Availability does not prove completeness. Only Grok
|
|
53
|
+
Markdown layouts are checked; other formats, including fast-agent JSON snapshots, are
|
|
54
|
+
unchecked, not absent or uncollected.
|
|
55
|
+
- `atif-inspect` prints masked trace text: treat its output like the trace.
|
|
56
|
+
- When a directory or `hf://` prefix is expanded, each input is labelled by its path
|
|
57
|
+
*relative to the root you passed* (e.g. `trial-1/agent/trajectory.json`). The root
|
|
58
|
+
itself never appears. Don't scan roots whose sub-paths are sensitive; use a manifest.
|
|
59
|
+
- Remote inputs are synced by default to the atif-scan home, `~/.cache/atif-scan` (or
|
|
60
|
+
`$ATIF_SCAN_HOME`, `$XDG_CACHE_HOME/atif-scan`; `$ATIF_SCAN_SYNC_DIR` or `--sync-dir` for
|
|
61
|
+
the copies alone). The home also holds the label store and gold snapshots, which name
|
|
62
|
+
real runs: keep all of it private. Those copies are real, possibly
|
|
63
|
+
credential-bearing traces: keep the directory private, never inside a repository, and
|
|
64
|
+
delete it when done (`--no-sync` streams without keeping files). The sync layout is
|
|
65
|
+
confined to that directory (path traversal and symlink destinations are rejected).
|
|
66
|
+
Owned download directories are made `0700` and files `0600`, including existing copies.
|
|
67
|
+
Hugging Face mirrors retain a private `.atif-sync.json` inventory (relative paths and
|
|
68
|
+
content identities, no trace text). Keep it with the copy: it preserves missing inputs
|
|
69
|
+
on offline rescans. Failed sync files are reported, never silently dropped or replaced
|
|
70
|
+
with an old copy; any sync failure returns exit 2.
|
|
71
|
+
- The result cache (`--cache`, default `<sync dir>/results`) stores a subset of the
|
|
72
|
+
allowlisted JSON report (the trace-derived results; run facts are re-read each scan),
|
|
73
|
+
never citations or trace text.
|
|
74
|
+
- Besides trajectories, the scanner reads only small, size-capped run files: reward files
|
|
75
|
+
(`verifier/reward.{json,txt}`, at most 4 KiB), Harbor's `result.json`/`config.json`,
|
|
76
|
+
`trials.jsonl`, and harbor-hf's `run.json` (declared prices) and
|
|
77
|
+
`attempt-costs/*.json` (at most 4 KiB). Only allowlisted numbers, codes and labels are
|
|
78
|
+
extracted from them. `--inspect` reads no file contents.
|
|
79
|
+
- A trial's submitted patch (`<trial>/artifacts/model.patch` beside
|
|
80
|
+
`<trial>/agent/trajectory.json`, as DeepSWE/Pier record it; at most 8 MiB, symlinks
|
|
81
|
+
refused) is read for its `diff --git` file paths, change kinds and added lines. They
|
|
82
|
+
stay in memory for fixed predicates (the DeepSWE pack); reports carry only check IDs,
|
|
83
|
+
counts and trace step locators, never paths, patch lines or patch text. A missing patch
|
|
84
|
+
is unknown, never an empty submission. Patch text is the agent's code: treat it like
|
|
85
|
+
the trace.
|
|
86
|
+
- Harbor Hub jobs are read through the user's own `harbor` CLI: no shell, fixed
|
|
87
|
+
arguments, a validated UUID, stderr withheld. Downloaded trajectories are real traces;
|
|
88
|
+
`--sync-to` puts them where you choose, otherwise they go to a temporary folder that's
|
|
89
|
+
removed afterwards.
|
|
90
|
+
- Network access happens only for `hf://` / huggingface.co inputs you name, via
|
|
91
|
+
`huggingface_hub` and your saved token. Other URLs are refused. Remote error messages
|
|
92
|
+
(which may contain URLs) are withheld; files over the size cap are rejected.
|
|
93
|
+
- Bench-run discovery reads only capped cohort catalogs, receipts and receipt-pinned
|
|
94
|
+
configs. It never imports or executes bench-run code, follows auth/config paths, or
|
|
95
|
+
contacts a provider. Codex diagnostic receipts are recognized by `kind` and a
|
|
96
|
+
top-level `run_id` only; `plan_path` is not opened. Job selection is confined to the
|
|
97
|
+
chosen root's `jobs/` directory; paths escaping the root are rejected. Listing exports
|
|
98
|
+
only run IDs, allowlisted lifecycle statuses, counts and fixed issue codes.
|
|
99
|
+
Replacement receipts (`runs/replacements/*/receipt.json`) and release manifests are
|
|
100
|
+
read the same way: capped, digest-checked against the parent run, confined to the
|
|
101
|
+
root, failing closed when one names the run but doesn't verify. Reports carry only
|
|
102
|
+
their codes (error types, phases), IDs and trial folder names, never operator `reason`
|
|
103
|
+
text.
|
|
104
|
+
- For an errored trial only, fast-agent's `fast-agent-results.json` (beside the
|
|
105
|
+
trajectory, capped) is read for its provider safety details: `provider`, `reason` and
|
|
106
|
+
`category`, each kept only as an identifier-like code. The provider's explanation and
|
|
107
|
+
every message in that file are never reported.
|
|
108
|
+
- Plugins and rule files are **trusted** code/configuration, not a sandbox. A plugin can
|
|
109
|
+
read raw text and do its own I/O. Only load modules you have reviewed. Plugins are
|
|
110
|
+
never discovered automatically. The only code loaded without `--plugin` is this
|
|
111
|
+
package's own bundled packs (`atif_scan.packs`), for runs they recognise; `--packs none`
|
|
112
|
+
disables that.
|
|
113
|
+
- The loader has a size cap, but there is no isolation for JSON depth, regex runtime or
|
|
114
|
+
plugin CPU/filesystem use. Use a restricted worker for hostile inputs.
|
|
115
|
+
- `--browse` is private inspection, not report export: a token-authenticated IPv4
|
|
116
|
+
loopback server with strict Host/Origin checks, fixed assets/routes, CSP, no request
|
|
117
|
+
logging and no-store responses. Never publish or tunnel its port or share its launch
|
|
118
|
+
URL. Whole-field masking precedes paging/search; text is rendered literally and no
|
|
119
|
+
trace commands or URLs are executed. Masking is best effort. Local sources are pinned
|
|
120
|
+
by digest before scanning, checked again before serving, and revalidated on access;
|
|
121
|
+
changed sources require a new scan. Browser inputs are opaque IDs, not client paths.
|
|
122
|
+
Finding-level feedback is separate from reports and trial labels, stored privately in
|
|
123
|
+
`<atif-scan home>/feedback/` (or `--feedback-dir`). Notes are unmasked reviewer text:
|
|
124
|
+
keep them and the append-only journal out of Git. Feedback is bound to source/task,
|
|
125
|
+
trace content and assessment identity; it cannot clear coverage gaps. Latest append
|
|
126
|
+
wins: this is a single-reviewer POSIX tool, not a multi-user adjudication service.
|
|
127
|
+
- `--viewer DIR` is the one output that deliberately contains trace text: a static
|
|
128
|
+
folder (fixed HTML/CSS/JS plus `data.js`) for publishing an individually reviewed
|
|
129
|
+
trajectory. Every field is masked as a whole before export, but masking is best
|
|
130
|
+
effort, so read the export before publishing it; it is not a clearance. The data is
|
|
131
|
+
allowlisted (labels, task, reward, coverage, check IDs/titles/priorities, field
|
|
132
|
+
locators, masked field text and, with `--answers`, the judge's masked reasons) and never includes source paths, citations, feedback
|
|
133
|
+
or raw metadata; media is not exported. By default only non-info findings, all unknown
|
|
134
|
+
findings and their related whole steps (including judge-cited steps) are exported;
|
|
135
|
+
summaries and coverage still describe the full scan. `--viewer-full` includes all
|
|
136
|
+
findings and steps; blind `--review` exports always retain the full chronology.
|
|
137
|
+
Filtering is not a safety clearance. Highlights are offsets proven against the
|
|
138
|
+
masked text, otherwise the whole field is marked as the evidence. Sources are pinned
|
|
139
|
+
by digest before scanning and must still match when exported. The page loads nothing
|
|
140
|
+
outside its folder (CSP, no fonts, images or network), renders trace text literally
|
|
141
|
+
and executes nothing. The folder is created `0700` with `0600` files and must be new
|
|
142
|
+
or empty.
|
|
143
|
+
- `--viewer DIR --review QUESTION` writes a blind review export (same masking, CSP and
|
|
144
|
+
file rules; no findings, scores, judge answers or scanner run sections). Verdicts stay
|
|
145
|
+
in the browser's local storage, keyed by a random export ID, and leave only as a JSON
|
|
146
|
+
file the reviewer downloads. Notes in it are the reviewer's own unmasked text: keep the
|
|
147
|
+
file private. `atif-scan labels import-review` keeps answer, mechanism and reward only,
|
|
148
|
+
never the note, and maps opaque export IDs to trials through the private key.
|
|
149
|
+
- A finding doesn't authorize accusation, disqualification or exclusion. Keep the
|
|
150
|
+
coverage information and get human review.
|
|
151
|
+
|
|
152
|
+
Report security concerns privately to the repository owner. Don't post sensitive
|
|
153
|
+
trajectories in an issue.
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
# Design notes
|
|
2
|
+
|
|
3
|
+
The fine print behind the [README](../README.md). Read this before writing a
|
|
4
|
+
detector or changing the engine.
|
|
5
|
+
|
|
6
|
+
## Modules
|
|
7
|
+
|
|
8
|
+
| Module | Role |
|
|
9
|
+
|---|---|
|
|
10
|
+
| `sources.inputs` | Resolves files, directories, `hf://` paths and Hub URLs to loadable inputs (the only input I/O) |
|
|
11
|
+
| `sources.sync` | Mirrors remote inputs into the private sync folder (inventory, content identities, confinement) |
|
|
12
|
+
| `sources.harbor.runs` | Finds the Harbor files next to each trajectory: reward, trial `result.json` + attempt cost, `exception.txt`, job folders, saved Hub listings and ledgers |
|
|
13
|
+
| `sources.harbor.files` | Parses Harbor's own run files (trial `result.json`, job `config.json`/`result.json`, `trials.jsonl`, harbor-hf `run.json` prices and attempt costs) into allowlisted facts |
|
|
14
|
+
| `data.facts` | The one place that decides which record wins for each per-trial fact (reward, task, error, tokens, cost), and the trace-derived facts that are cached |
|
|
15
|
+
| `data.paths` | The atif-scan home (`$ATIF_SCAN_HOME`) and its folders: `hf/`, `harbor/`, `results/`, `labels/`, `gold/`, `bundles/` |
|
|
16
|
+
| `data.jsonval` | Narrows untrusted JSON values (`as_object`, `count`, `number`, …); wrong shapes become unknown, never zero |
|
|
17
|
+
| `sources.harbor.hub` | `harbor://jobs/<id>`: Hub listing (task/reward/cost) and trajectory downloads via the `harbor` CLI |
|
|
18
|
+
| `sources.harbor.listing` | Hub listing rows: validation and allowlisted run/trial facts, shared by live and saved listings (no I/O) |
|
|
19
|
+
| `sources.layout` | `--inspect`: classifies a listing (Harbor markers, roles, anomalies) without reading traces |
|
|
20
|
+
| `data.model`, `data.loader` | Immutable `Trace → Step → ToolCall / Observation` view of ATIF v1, and the parser that builds it |
|
|
21
|
+
| `data.content`, `data.tools`, `data.pairing`, `data.usage` | What the loader reads: a value's text and media; tool categories and argument channels; call/result pairing; recorded usage and retries |
|
|
22
|
+
| `data.submission` | A trial's submitted git patch (DeepSWE `artifacts/model.patch`): file paths, change kinds and added lines, memory only; read beside the trajectory by `sources.harbor.runs.submission_near` into `Context.submission` |
|
|
23
|
+
| `data.jslit`, `data.shell` | Static readers for Codex code-mode programs and shell commands (never executed); unreadable input is unknown or falls back to text patterns |
|
|
24
|
+
| `checks` | The plugin contract: `CheckSpec`, `Context`, `Detection`, `Status`, `Severity` |
|
|
25
|
+
| `detectors` | Built-in families (`lookup`, `awareness`, `integrity`, `tamper`, `installs`, `side_channel`, …), the shared `vocabulary` packs reuse, the `builtin` registry, and the `RegexDetector` / `SurfaceDetector` / `ObservationDetector` helpers |
|
|
26
|
+
| `rules`, `policy` | Three-valued rule expressions, `Allowance`, and the JSON rule/allow format |
|
|
27
|
+
| `engine` | Dependency ordering, task scope, error isolation, applying allowances |
|
|
28
|
+
| `output.document`, `output.text`, `output.summary`, `output.overview` | The JSON allowlist and filtering; plain-text detail/inspect/citation views; `--summary`; the run overview (accuracy, reruns, finding index, DQ scenario, cost, tokens, walltime), all rendered only from the document |
|
|
29
|
+
| `output.rich`, `output.views` | The rich renderer and its optional entry points |
|
|
30
|
+
| `output.brief`, `output.brief_view` (`words`, `colour`, `run`, `evidence`, `usage`), `output.estimates` | One-screen run integrity report (one renderer per section); token accounting, then cost priced once from given, declared or fitted rates; missing-activity estimates |
|
|
31
|
+
| `cache` | Per-trace result cache keyed by file fingerprint + scanner version + check set + context |
|
|
32
|
+
| `evidence.cite` | Opt-in (`--cite`) masked excerpts with before/after context: the only trace-text output |
|
|
33
|
+
| `evidence.history`, `evidence.extract` | Companion history archives; `atif-inspect` (outline, masked step reads, search) |
|
|
34
|
+
| `review.catalogue`, `review.prompts`, `review.answers` | Follow-up and hunt questions: the catalogue (asks, answer sets, triggers, evidence selection), prompt and schema writing (blind mode shows no findings), answers read back and tallied |
|
|
35
|
+
| `review.labels` | The label store: schema, source precedence, run splits, scanner/Jev evaluation (see improvement-loop.md) |
|
|
36
|
+
| `review.inspect_server` | The read-only MCP server over one trajectory behind `hunt --inspect-tool` (optional `mcp` extra) |
|
|
37
|
+
| `output.bundle` | `--questions`: the review selection (scope, questions, blind mode) written as a question bundle |
|
|
38
|
+
| `cli` | The command line: `args` (parser and option checks), `inputs` (manifests, paths, sync, tasks, plugins), `scan` (cached evaluation, the report document, review bundles), `emit` (brief/overview/summary/detail as text or JSON), `inspect` (`--inspect`), the `labels` and `hunt` commands; `main` (dispatch) and `--submission` in the package |
|
|
39
|
+
|
|
40
|
+
Keep these separate: parsing doesn't know about detectors, detectors don't know about
|
|
41
|
+
rules, and only `output.document.report` decides what gets written out.
|
|
42
|
+
|
|
43
|
+
`tests/test_architecture.py` enforces the layering on every runtime import: no cycles;
|
|
44
|
+
the `data` package (`model`, `loader`, `content`, `tools`, `pairing`, `usage`, `jsonval`,
|
|
45
|
+
`shell`, `jslit`, `credentials`, `facts`, `accounting`, `paths`, `web_*`) imports only itself; `sources` only data and itself; analysis
|
|
46
|
+
(`checks`, `rules`, `policy`, `engine`, `access`, `detectors`, `packs`) only data and
|
|
47
|
+
itself.
|
|
48
|
+
|
|
49
|
+
## Input scope
|
|
50
|
+
|
|
51
|
+
- Accepts `ATIF-v1.x` (or no version). Other major versions are rejected. The loader
|
|
52
|
+
checks the structures it uses, ignores unrelated extensions, and doesn't claim full
|
|
53
|
+
schema validation. Files over 256 MiB are rejected.
|
|
54
|
+
- The `Trace` is an analysis view, not a lossless copy. Top-level metadata and usage
|
|
55
|
+
stats are dropped.
|
|
56
|
+
- Text fields become `Content`. Formats it can't read get `understood=False`, and
|
|
57
|
+
detectors report `unknown` for them instead of `no_match`. Images and other binary
|
|
58
|
+
blocks are excluded from text checks. That doesn't mean they contain nothing relevant.
|
|
59
|
+
An inline payload is kept only as its sha256 (`Content.media_ids`), so the opt-in
|
|
60
|
+
`--image-model` answers (`Context.images`) can be matched to it.
|
|
61
|
+
- Every call's arguments go through `tools.classify`. It walks string leaves together
|
|
62
|
+
with their nearest key and routes each one by key convention and value shape to
|
|
63
|
+
`PAYLOAD`, `URL`, `COMMAND`, `QUERY`, `PATH` or `ARGUMENTS`, in that order of
|
|
64
|
+
precedence (see the tool coverage table in detectors.md). Key lists are generic argument conventions, not
|
|
65
|
+
harness tool names. Extend them with a regression test when a real trace shows a gap.
|
|
66
|
+
- Tool names are *hints* (`tools.TOOLS` → `shell`, `read`, `write`, `search_files`,
|
|
67
|
+
`web_fetch`, `web_search`, `inert`, `other`). They supply:
|
|
68
|
+
- meaning (only a known `web_search` tool counts as a web search);
|
|
69
|
+
- required inputs (a known shell call without a command is `unknown`);
|
|
70
|
+
- `inert`: orchestration tools whose arguments are ignored, e.g. `ToolSearch`, whose
|
|
71
|
+
`query` looks up tools, not the web.
|
|
72
|
+
- `SurfaceDetector` is incomplete when a surface on its channels isn't understood, when
|
|
73
|
+
any call's arguments are unparseable (for tool-input checks), or when its
|
|
74
|
+
`undecidable(surface)` hook says the predicate needs tool semantics it doesn't have.
|
|
75
|
+
An unrecognized tool name alone never reduces coverage. Original `name` and deeply
|
|
76
|
+
immutable `arguments` stay available to plugins (`Surface.tool_name` too).
|
|
77
|
+
- Step metadata (`step_id`, parsed `timestamp`, whether each was recorded), agent-only
|
|
78
|
+
fields on non-agent steps, and `final_metrics.extra.total_tool_use_tokens` are
|
|
79
|
+
retained for integrity checks.
|
|
80
|
+
- `Trace.agent_surfaces()` yields agent prose and classified tool-argument surfaces
|
|
81
|
+
(including `PAYLOAD`, which built-ins don't read).
|
|
82
|
+
`Trace.observation_surfaces()` is a separate, opt-in API for tool outputs.
|
|
83
|
+
- Content-bearing fields are hidden from `repr`. That prevents accidental logging, but
|
|
84
|
+
doesn't stop deliberate logging.
|
|
85
|
+
|
|
86
|
+
## Evidence semantics
|
|
87
|
+
|
|
88
|
+
A detector returns `Detection(status, evidence, complete)`:
|
|
89
|
+
|
|
90
|
+
- `match` means the predicate matched recorded text. It can still be `complete=False`.
|
|
91
|
+
- `no_match` means the detector's whole declared scope was readable and nothing
|
|
92
|
+
matched. It must be `complete=True`.
|
|
93
|
+
- `unknown` and `error` must be `complete=False`.
|
|
94
|
+
|
|
95
|
+
`Detection.of(hits, complete)` builds the usual result: `match` on any hit (deduplicated,
|
|
96
|
+
order kept), else `no_match` when complete, else `unknown`. `Step.results_for(call)` is
|
|
97
|
+
the one call → result link: observations whose `source_call_id` is the call's result key,
|
|
98
|
+
or, when none is and the step has exactly one call, its unlinked observations.
|
|
99
|
+
|
|
100
|
+
Engine-level rules:
|
|
101
|
+
|
|
102
|
+
- A task-scoped check with no task is `unknown`. With a different task it's
|
|
103
|
+
`not_applicable`.
|
|
104
|
+
- A missing trace makes every check `unknown`.
|
|
105
|
+
- `partial` inputs keep their matches (marked incomplete). Their `no_match` results
|
|
106
|
+
become `unknown`. A trace with recording gaps (`Trace.recording_gaps`: compacted
|
|
107
|
+
history, status-only results, claimed but unrecorded actions) is evaluated as partial
|
|
108
|
+
by `Engine.evaluate` itself (`engine.effective_context`), so library callers get the
|
|
109
|
+
same result as the CLI.
|
|
110
|
+
- Exceptions, non-`Detection` return values, and evidence pointing outside the trace
|
|
111
|
+
become `error`. The exception text is never kept.
|
|
112
|
+
|
|
113
|
+
Every negative is limited to what was observed. If no reasoning was recorded, that
|
|
114
|
+
doesn't prove the agent had no benchmark awareness. If no URL was matched, that doesn't
|
|
115
|
+
prove there was no network access.
|
|
116
|
+
|
|
117
|
+
## Tool results and integrity
|
|
118
|
+
|
|
119
|
+
`ObservationDetector` scans recorded tool results. It is `unknown` when the agent
|
|
120
|
+
made calls but no results were recorded. A result is evidence of what the agent
|
|
121
|
+
*received*, not what it authored or used.
|
|
122
|
+
|
|
123
|
+
Integrity checks (`detectors/integrity.py`) go beyond Harbor's validator, which
|
|
124
|
+
checks schema, sequential `step_id`, call links and timestamp *syntax*, but not
|
|
125
|
+
ordering, smearing or telemetry plausibility. Scopes:
|
|
126
|
+
|
|
127
|
+
- Timing checks use recorded timestamps on non-copied steps. They are `unknown` only
|
|
128
|
+
when too few timestamps exist to evaluate. Missing agent-step timestamps are
|
|
129
|
+
reported positively by `integrity.timestamp_missing`.
|
|
130
|
+
- `integrity.step_sequence` is `unknown` when no step carries a `step_id`.
|
|
131
|
+
- `integrity.tool_token_telemetry` concerns *reported* values; not reporting is
|
|
132
|
+
not implausible.
|
|
133
|
+
|
|
134
|
+
## Rules
|
|
135
|
+
|
|
136
|
+
`Ref`, `All`, `AnyOf`, `Not` and `Requires` use strong Kleene logic: `error` and
|
|
137
|
+
`not_applicable` inputs count as unknown. `All(false, unknown)` is false and
|
|
138
|
+
`AnyOf(true, unknown)` is true. `Requires(A, B)` is `All(A, Not(B))`. It reports a
|
|
139
|
+
**violation**, not a passing implication.
|
|
140
|
+
|
|
141
|
+
A matched rule cites the evidence of its matched dependencies. When the consequent is
|
|
142
|
+
false there is nothing to cite, so the dependency's `no_match` is the basis instead.
|
|
143
|
+
|
|
144
|
+
Rules are per-trace co-occurrence checks. They don't encode time order or causation, and
|
|
145
|
+
there are no cross-trial statistics. Write that kind of logic as a Python detector that
|
|
146
|
+
returns the same `Detection` type.
|
|
147
|
+
|
|
148
|
+
## Context checks and rewards
|
|
149
|
+
|
|
150
|
+
A `ContextCheck` (kind `context`) states a fact about the run, e.g. `context.rewarded`.
|
|
151
|
+
It is evaluated before rules, so rules and allowances can reference it. It never counts
|
|
152
|
+
towards `score`/`--fail-on`. An unknown fact doesn't make a report incomplete by
|
|
153
|
+
itself; a rule that depends on it is unknown instead.
|
|
154
|
+
|
|
155
|
+
The Terminal-Bench judge only reviews rewarded trials, and absence rules like "rewarded,
|
|
156
|
+
but no SSH setup" need the reward. `sources` finds it without an extra listing:
|
|
157
|
+
`verifier/reward.json`, then `reward.txt` (Harbor's order), in the trajectory's folder or
|
|
158
|
+
one level up, read only when present in the listing and capped at 4 KiB. Anything that
|
|
159
|
+
isn't a finite number is unknown, never zero.
|
|
160
|
+
|
|
161
|
+
## LLM judges and classifiers
|
|
162
|
+
|
|
163
|
+
Some judgments (e.g. "fabricated an answer after abandoning real work", intent language
|
|
164
|
+
that matched 121 of 441 real traces) need a model. Write them as plugins: a detector that
|
|
165
|
+
returns the usual `Detection` with evidence locators. Keep them out of the core:
|
|
166
|
+
|
|
167
|
+
- Pre-filter with cheap detectors and rules, and judge only candidate surfaces or
|
|
168
|
+
rewarded trials. That's the Terminal-Bench judge's policy too.
|
|
169
|
+
- Network/model use is the plugin's explicit, trusted I/O. Report only the verdict and
|
|
170
|
+
locators, never model text. Map refusals, timeouts and unparseable verdicts to
|
|
171
|
+
`unknown`.
|
|
172
|
+
- Pin the model/prompt in the check `version`.
|
|
173
|
+
|
|
174
|
+
## Allowances
|
|
175
|
+
|
|
176
|
+
An `Allowance` is evaluated after every detector and rule. It has a task scope, a set of
|
|
177
|
+
covered check IDs and an optional `when` expression. It applies only when its status is
|
|
178
|
+
`match`: unconditional within scope, or `when` is true. Then each covered assessment
|
|
179
|
+
that matched gets `expected_by`, and `Assessment.counts` becomes false. That single
|
|
180
|
+
predicate drives `score`, `severity` and `--fail-on`.
|
|
181
|
+
|
|
182
|
+
- An allowance never rewrites a `Detection`. "Expected" is a disposition, not a new status.
|
|
183
|
+
- `unknown` allowances (unknown `when`, missing task, partial trace with an unmet
|
|
184
|
+
condition) don't apply, and they make the report `incomplete`.
|
|
185
|
+
- Rules evaluate raw results, so excusing a fact doesn't silently disarm rules built on it.
|
|
186
|
+
This keeps evaluation single-pass and explicit: cover the rule ID too if intended.
|
|
187
|
+
- Allowances can't be referenced by rules or other allowances.
|
|
188
|
+
|
|
189
|
+
## Report boundary
|
|
190
|
+
|
|
191
|
+
`report.report` builds its output field by field and never serializes trace objects or
|
|
192
|
+
plugin data. Evidence is `(step, channel, call, observation, field, span)`: **array
|
|
193
|
+
positions** (not ATIF step IDs), the index of the classified call argument, and the
|
|
194
|
+
character span of the match when the detector knows it. A `SurfaceDetector` predicate may
|
|
195
|
+
return a `re.Match` or `(start, end)` to supply the span. `cite` turns these into
|
|
196
|
+
excerpts only when asked; renderers never re-open the trace. Check IDs, versions and manifest IDs must be static, non-sensitive
|
|
197
|
+
identifiers. `CheckSpec.title` is a static, non-sensitive label too (at most 60 printable
|
|
198
|
+
characters, one line, no URL); `report.document` lists every check the engine ran in a
|
|
199
|
+
top-level `checks` catalog (`{id: {severity, title}}`) built from `Engine.catalog()` on each
|
|
200
|
+
scan, never from cached trace results. Built-in titles say what was observed, in sentence
|
|
201
|
+
case, at most 48 characters, without severity words or hedging.
|
|
202
|
+
|
|
203
|
+
Built-ins are deterministic: no clock, randomness, network or filesystem access. Bump a
|
|
204
|
+
detector's `version` when its meaning changes. Bump the report `schema_version` for
|
|
205
|
+
incompatible output changes.
|
|
206
|
+
|
|
207
|
+
## Writing and testing a detector
|
|
208
|
+
|
|
209
|
+
- State what the predicate actually shows: text signature, attempted action, observed
|
|
210
|
+
tool result, or verified outcome. Keep these distinct. A `pytest` command is not
|
|
211
|
+
evidence that the tests passed.
|
|
212
|
+
- Return `unknown` when the evidence you need is missing or unreadable, and say where:
|
|
213
|
+
put the positions you could judge in `evidence` and the context you couldn't read in
|
|
214
|
+
`Detection.unread` (a fixed reason code plus its locator: `media`, `unreadable`,
|
|
215
|
+
`compacted`, `web_result_not_recorded`, `prompt_not_recorded`; for a call with no
|
|
216
|
+
result `run_ended`, `result_compacted` or `result_not_recorded`; for a recorded result that
|
|
217
|
+
says it's partial `result_truncated` or `terminal_screen_only` (Terminus 2 captures, see
|
|
218
|
+
`data/terminal.py`); `undecidable` when
|
|
219
|
+
the predicate can't judge without running it; `prompt_names_benchmark` when the prompt
|
|
220
|
+
already says what the evidence says, so a hit can't be told apart from the task). `Detection.of(hits, complete, unread)`
|
|
221
|
+
keeps them only for incomplete results; reports list them as `unread`. If unread context can only *remove* a candidate (priming), a
|
|
222
|
+
negative that holds even without it is a real `no_match`.
|
|
223
|
+
- A trace-wide check with no single location should say what it compared: put its
|
|
224
|
+
figures in `Detection.measure` (identifier names with numbers, None or short codes;
|
|
225
|
+
validated, never text). Reports list them as `measure`; the viewer shows the sum.
|
|
226
|
+
- Never put runtime data into IDs or evidence.
|
|
227
|
+
- Patterns run over megabytes of tool output and minified code. Bound every gap
|
|
228
|
+
between two literals (`\S{0,256}?`, `[^\n]{0,300}`: document the bound), split
|
|
229
|
+
`A.*B` into a search for `A` then for `B` from its end, and never lead with `^\s*`
|
|
230
|
+
under `re.M` (use `^[^\S\n]*`). A literal prefilter (`detectors.text.gated`) is only
|
|
231
|
+
exact when every alternative contains a hint; test that per alternative. Add a timing
|
|
232
|
+
guard for anything that was superlinear.
|
|
233
|
+
|
|
234
|
+
Checklist for tests:
|
|
235
|
+
|
|
236
|
+
- A positive match and a near-miss negative, across tool aliases.
|
|
237
|
+
- System/user text, copied context and tool outputs don't leak into authored-text checks.
|
|
238
|
+
- Ordinary look-alikes: local tests, perf benchmarks, package installs.
|
|
239
|
+
- Missing trace, malformed arguments, missing task, partial trace.
|
|
240
|
+
- Secrets, signed URLs and multiline commands never appear in the report.
|
|
241
|
+
- Unknown dependencies, cycles and `requires` behave as described above.
|
|
242
|
+
|
|
243
|
+
## Known limits of the built-in pack
|
|
244
|
+
|
|
245
|
+
The built-ins are exploratory. Shell comments, heredocs, and program source embedded in
|
|
246
|
+
a command can match. Quoted prose can echo task instructions. Dynamic URLs, hosted
|
|
247
|
+
(native) search and unconventional tool names can be missed. Improve them with versioned
|
|
248
|
+
tests, not by tuning scores after the fact.
|