sad-py 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. sad_py-0.1.0/.github/dependabot.yml +10 -0
  2. sad_py-0.1.0/.github/workflows/ci.yml +57 -0
  3. sad_py-0.1.0/.github/workflows/publish.yml +31 -0
  4. sad_py-0.1.0/.github/workflows/release.yml +34 -0
  5. sad_py-0.1.0/.github/workflows/reliability.yml +27 -0
  6. sad_py-0.1.0/.gitignore +13 -0
  7. sad_py-0.1.0/.pre-commit-config.yaml +13 -0
  8. sad_py-0.1.0/CHANGELOG.md +19 -0
  9. sad_py-0.1.0/CONTRIBUTING.md +29 -0
  10. sad_py-0.1.0/LICENSE +21 -0
  11. sad_py-0.1.0/PKG-INFO +122 -0
  12. sad_py-0.1.0/README.md +92 -0
  13. sad_py-0.1.0/SECURITY.md +9 -0
  14. sad_py-0.1.0/docs/architecture.md +101 -0
  15. sad_py-0.1.0/docs/integrations.md +64 -0
  16. sad_py-0.1.0/docs/releasing.md +9 -0
  17. sad_py-0.1.0/docs/reliability.md +43 -0
  18. sad_py-0.1.0/docs/result.schema.json +327 -0
  19. sad_py-0.1.0/examples/app.ts +8 -0
  20. sad_py-0.1.0/examples/burp_worker.py +31 -0
  21. sad_py-0.1.0/examples/history.json +6 -0
  22. sad_py-0.1.0/examples/index.html +6 -0
  23. sad_py-0.1.0/pyproject.toml +46 -0
  24. sad_py-0.1.0/scripts/benchmark.py +43 -0
  25. sad_py-0.1.0/scripts/generate_schema.py +74 -0
  26. sad_py-0.1.0/scripts/reliability_report.py +100 -0
  27. sad_py-0.1.0/src/sad/__init__.py +24 -0
  28. sad_py-0.1.0/src/sad/__main__.py +3 -0
  29. sad_py-0.1.0/src/sad/analyzer.py +255 -0
  30. sad_py-0.1.0/src/sad/cli.py +74 -0
  31. sad_py-0.1.0/src/sad/crawler/__init__.py +192 -0
  32. sad_py-0.1.0/src/sad/html/__init__.py +94 -0
  33. sad_py-0.1.0/src/sad/ingest/__init__.py +205 -0
  34. sad_py-0.1.0/src/sad/javascript/__init__.py +3 -0
  35. sad_py-0.1.0/src/sad/javascript/engine.py +1264 -0
  36. sad_py-0.1.0/src/sad/javascript/parser.py +45 -0
  37. sad_py-0.1.0/src/sad/javascript/resources.py +75 -0
  38. sad_py-0.1.0/src/sad/javascript/sourcemaps.py +114 -0
  39. sad_py-0.1.0/src/sad/javascript/values.py +249 -0
  40. sad_py-0.1.0/src/sad/models.py +125 -0
  41. sad_py-0.1.0/src/sad/normalization/__init__.py +114 -0
  42. sad_py-0.1.0/src/sad/output/__init__.py +46 -0
  43. sad_py-0.1.0/src/sad/protocols/__init__.py +50 -0
  44. sad_py-0.1.0/src/sad/py.typed +0 -0
  45. sad_py-0.1.0/tests/__init__.py +0 -0
  46. sad_py-0.1.0/tests/conftest.py +29 -0
  47. sad_py-0.1.0/tests/corpus.py +210 -0
  48. sad_py-0.1.0/tests/fixtures/adapted/axios.js +6 -0
  49. sad_py-0.1.0/tests/fixtures/adapted/connect.ts +7 -0
  50. sad_py-0.1.0/tests/fixtures/adapted/openapi.ts +21 -0
  51. sad_py-0.1.0/tests/fixtures/adapted/trpc.ts +6 -0
  52. sad_py-0.1.0/tests/fixtures/bundles/README.md +1 -0
  53. sad_py-0.1.0/tests/fixtures/bundles/parcel.js +2 -0
  54. sad_py-0.1.0/tests/fixtures/bundles/rollup.js +2 -0
  55. sad_py-0.1.0/tests/fixtures/bundles/vite-chunk.js +1 -0
  56. sad_py-0.1.0/tests/fixtures/bundles/vite.js +3 -0
  57. sad_py-0.1.0/tests/fixtures/bundles/webpack.js +2 -0
  58. sad_py-0.1.0/tests/fixtures/research.json +6 -0
  59. sad_py-0.1.0/tests/fixtures/upstream/OpenAPITools/LICENSE +202 -0
  60. sad_py-0.1.0/tests/fixtures/upstream/OpenAPITools/runtime.ts +572 -0
  61. sad_py-0.1.0/tests/fixtures/upstream/README.md +12 -0
  62. sad_py-0.1.0/tests/fixtures/upstream/axios/LICENSE +7 -0
  63. sad_py-0.1.0/tests/fixtures/upstream/axios/README.md +2892 -0
  64. sad_py-0.1.0/tests/fixtures/upstream/connectrpc/App.tsx +115 -0
  65. sad_py-0.1.0/tests/fixtures/upstream/connectrpc/LICENSE +201 -0
  66. sad_py-0.1.0/tests/fixtures/upstream/connectrpc/eliza_pb.ts +191 -0
  67. sad_py-0.1.0/tests/fixtures/upstream/manifest.json +48 -0
  68. sad_py-0.1.0/tests/fixtures/upstream/trpc/App.tsx +11 -0
  69. sad_py-0.1.0/tests/fixtures/upstream/trpc/Greeting.tsx +8 -0
  70. sad_py-0.1.0/tests/fixtures/upstream/trpc/LICENSE +21 -0
  71. sad_py-0.1.0/tests/fixtures/upstream/trpc/trpc.ts +21 -0
  72. sad_py-0.1.0/tests/test_analysis_budgets.py +100 -0
  73. sad_py-0.1.0/tests/test_bundles.py +25 -0
  74. sad_py-0.1.0/tests/test_collection_boundaries.py +256 -0
  75. sad_py-0.1.0/tests/test_controlflow_edges.py +140 -0
  76. sad_py-0.1.0/tests/test_corpus.py +27 -0
  77. sad_py-0.1.0/tests/test_crawler_cli.py +87 -0
  78. sad_py-0.1.0/tests/test_engine.py +279 -0
  79. sad_py-0.1.0/tests/test_ingest.py +160 -0
  80. sad_py-0.1.0/tests/test_properties.py +104 -0
  81. sad_py-0.1.0/tests/test_reliability_regressions.py +100 -0
  82. sad_py-0.1.0/tests/test_robustness.py +200 -0
  83. sad_py-0.1.0/tests/test_sourcemaps.py +189 -0
  84. sad_py-0.1.0/tests/test_stress.py +28 -0
  85. sad_py-0.1.0/tests/test_upstream.py +57 -0
@@ -0,0 +1,10 @@
1
+ version: 2
2
+ updates:
3
+ - package-ecosystem: pip
4
+ directory: /
5
+ schedule:
6
+ interval: weekly
7
+ - package-ecosystem: github-actions
8
+ directory: /
9
+ schedule:
10
+ interval: weekly
@@ -0,0 +1,57 @@
1
+ name: CI
2
+ on: [push, pull_request, workflow_call]
3
+ permissions:
4
+ contents: read
5
+ jobs:
6
+ test:
7
+ runs-on: ubuntu-latest
8
+ strategy:
9
+ fail-fast: false
10
+ matrix:
11
+ python: ['3.11', '3.12', '3.13', '3.14']
12
+ steps:
13
+ - uses: actions/checkout@v6
14
+ with:
15
+ persist-credentials: false
16
+ - uses: actions/setup-python@v6
17
+ with:
18
+ python-version: ${{ matrix.python }}
19
+ - run: python -m pip install -e '.[dev]'
20
+ - run: ruff check src tests examples scripts
21
+ - run: ruff format --check src tests examples scripts
22
+ - run: mypy src/sad
23
+ - run: pytest -m 'not stress' --cov=sad --cov-branch --cov-report=term-missing --cov-report=json:artifacts/coverage.json --cov-report=xml:artifacts/coverage.xml --junitxml=artifacts/tests.xml
24
+ - run: python scripts/reliability_report.py
25
+ - uses: actions/upload-artifact@v4
26
+ if: always()
27
+ with:
28
+ name: reliability-python-${{ matrix.python }}
29
+ path: artifacts/
30
+ - run: python -m build
31
+ - run: python -m twine check dist/*
32
+ - name: Install and smoke-test wheel outside checkout
33
+ shell: bash
34
+ run: |
35
+ python -m venv /tmp/sad-wheel
36
+ /tmp/sad-wheel/bin/pip install dist/*.whl
37
+ cd /tmp
38
+ /tmp/sad-wheel/bin/sad --version
39
+ /tmp/sad-wheel/bin/python -c 'from sad import Analyzer; assert Analyzer().analyze_js("fetch(\"/smoke\")").endpoints[0].url == "/smoke"'
40
+ platform-smoke:
41
+ runs-on: ${{ matrix.os }}
42
+ strategy:
43
+ matrix:
44
+ os: [windows-latest, macos-latest]
45
+ python: ['3.11', '3.14']
46
+ steps:
47
+ - uses: actions/checkout@v6
48
+ with:
49
+ persist-credentials: false
50
+ - uses: actions/setup-python@v6
51
+ with:
52
+ python-version: ${{ matrix.python }}
53
+ - run: python -m pip install -e '.[dev]'
54
+ - run: pytest tests/test_ingest.py tests/test_crawler_cli.py tests/test_collection_boundaries.py tests/test_properties.py
55
+ - run: python -m build
56
+ - run: python -m pip install --force-reinstall --no-deps --find-links dist sad-py
57
+ - run: sad --version
@@ -0,0 +1,31 @@
1
+ name: Publish to PyPI
2
+ on:
3
+ release:
4
+ types: [published]
5
+ permissions:
6
+ contents: read
7
+ jobs:
8
+ checks:
9
+ uses: ./.github/workflows/ci.yml
10
+ publish:
11
+ needs: checks
12
+ runs-on: ubuntu-latest
13
+ environment: pypi
14
+ permissions:
15
+ id-token: write
16
+ steps:
17
+ - uses: actions/checkout@v6
18
+ with:
19
+ persist-credentials: false
20
+ - uses: actions/setup-python@v6
21
+ with:
22
+ python-version: '3.12'
23
+ - run: python -m pip install build twine
24
+ - name: Validate release version and distributions
25
+ env:
26
+ RELEASE_TAG: ${{ github.event.release.tag_name }}
27
+ run: |
28
+ python -c 'import os,tomllib; assert os.environ["RELEASE_TAG"] == "v" + tomllib.load(open("pyproject.toml","rb"))["project"]["version"]'
29
+ python -m build
30
+ python -m twine check dist/*
31
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,34 @@
1
+ name: Draft release
2
+ on:
3
+ push:
4
+ tags: ['v*']
5
+ permissions:
6
+ contents: read
7
+ jobs:
8
+ checks:
9
+ uses: ./.github/workflows/ci.yml
10
+ draft:
11
+ needs: checks
12
+ runs-on: ubuntu-latest
13
+ permissions:
14
+ contents: write
15
+ steps:
16
+ - uses: actions/checkout@v6
17
+ with:
18
+ persist-credentials: false
19
+ - uses: actions/setup-python@v6
20
+ with:
21
+ python-version: '3.12'
22
+ - run: python -m pip install build twine
23
+ - name: Verify tag and build
24
+ env:
25
+ RELEASE_TAG: ${{ github.ref_name }}
26
+ run: |
27
+ python -c 'import os,tomllib; assert os.environ["RELEASE_TAG"] == "v" + tomllib.load(open("pyproject.toml","rb"))["project"]["version"]'
28
+ python -m build
29
+ python -m twine check dist/*
30
+ - name: Create draft with distributions
31
+ env:
32
+ GH_TOKEN: ${{ github.token }}
33
+ RELEASE_TAG: ${{ github.ref_name }}
34
+ run: gh release create "$RELEASE_TAG" dist/* --draft --verify-tag --generate-notes
@@ -0,0 +1,27 @@
1
+ name: Extended reliability
2
+ on:
3
+ schedule:
4
+ - cron: '17 3 * * 1'
5
+ workflow_dispatch:
6
+ permissions:
7
+ contents: read
8
+ jobs:
9
+ extended:
10
+ runs-on: ubuntu-latest
11
+ timeout-minutes: 20
12
+ steps:
13
+ - uses: actions/checkout@v6
14
+ with:
15
+ persist-credentials: false
16
+ - uses: actions/setup-python@v6
17
+ with:
18
+ python-version: '3.14'
19
+ - run: python -m pip install -e '.[dev]'
20
+ - run: pytest --cov=sad --cov-branch --cov-report=json:artifacts/coverage.json --junitxml=artifacts/tests.xml
21
+ - run: python scripts/reliability_report.py
22
+ - run: python scripts/benchmark.py --operations 10000
23
+ - uses: actions/upload-artifact@v4
24
+ if: always()
25
+ with:
26
+ name: extended-reliability
27
+ path: artifacts/
@@ -0,0 +1,13 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.py[cod]
4
+ .pytest_cache/
5
+ .mypy_cache/
6
+ .ruff_cache/
7
+ .coverage
8
+ htmlcov/
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ artifacts/
13
+ .hypothesis/
@@ -0,0 +1,13 @@
1
+ repos:
2
+ - repo: local
3
+ hooks:
4
+ - id: ruff-check
5
+ name: ruff check
6
+ entry: .venv/bin/ruff check --fix
7
+ language: system
8
+ types: [python]
9
+ - id: ruff-format
10
+ name: ruff format
11
+ entry: .venv/bin/ruff format
12
+ language: system
13
+ types: [python]
@@ -0,0 +1,19 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 — Unreleased
4
+
5
+ - Library-first Analyzer, public dataclasses and versioned JSON model.
6
+ - Tree-sitter JavaScript, TypeScript, JSX and TSX parsing with bounded AST caching.
7
+ - Symbolic strings, aliases, objects, destructuring, conditional value joins, modules, wrapper chains, closures and basic class clients.
8
+ - HTTP and protocol adapters, HTML forms, programmatic forms, source-map embedded sources and literal lazy imports.
9
+ - Offline Burp XML, normalized JSON, HAR, raw HTTP and directory imports; captured responses feed static analysis.
10
+ - Conservative static/observed correlation with original observations retained.
11
+ - Bounded scoped crawler, CLI and JSON/JSONL/CSV/table serialization.
12
+ - Structured incompleteness issues, isolated branch heaps, return unions, bounded loops/recursive summaries, live module bindings, path mappings and bound calls.
13
+ - Static bundle module tables, indexed/VLQ source maps, client state tracking and Connect wire descriptors.
14
+ - An offline labeled detection corpus, pinned upstream examples, property/fuzz tests, separate line/branch gates and cross-platform CI artifacts.
15
+ - Documented endpoint origins and captured-response discovery in the README.
16
+ - Canonical seed redirects, fair source shares of the global analysis budget, retained partial exports, configurable CLI analysis steps and explicit partial-result status.
17
+ - CI, typed package, wheel/sdist validation, draft releases and trusted PyPI publishing workflow.
18
+
19
+ This is an initial implementation with the coverage limits in docs/architecture.md. No public release has been made.
@@ -0,0 +1,29 @@
1
+ # Contributing
2
+
3
+ Use Python 3.11 or newer and a virtual environment:
4
+
5
+ ```sh
6
+ python -m venv .venv
7
+ . .venv/bin/activate
8
+ pip install -e '.[dev]'
9
+ ruff format src tests
10
+ ruff check src tests
11
+ mypy src/sad
12
+ pytest -m "not stress" --cov=sad --cov-branch --cov-report=term-missing --cov-report=json:artifacts/coverage.json
13
+ python scripts/reliability_report.py
14
+ pytest -m stress
15
+ python scripts/benchmark.py
16
+ python -m build
17
+ python -m twine check dist/*
18
+ ```
19
+
20
+ Add a minimal, synthetic fixture for each new behavior, plus a false-positive regression when adding sink recognition. Include expected methods, symbolic URLs, provenance and diagnostics. Never commit customer captures, cookies, authorization headers, secrets, or identifying production source maps.
21
+
22
+ Keep I/O outside the JavaScript value engine. Network sinks require resolved provenance; matching a method name alone is insufficient. Unknown values must survive as symbolic values or explicit nulls. All new syntax handling must honor the analysis budgets and never execute target code.
23
+
24
+ Public model changes need a changelog entry, schema update, and compatibility assessment. Before 1.0, incompatible changes require a minor version bump; additive compatible fields may ship in patch releases. Keep `Analyzer`, input adapters, and serialization independent of CLI argument parsing.
25
+
26
+ Run the complete CI matrix before release. Performance changes should include a representative synthetic bundle measurement and preserve deterministic outputs. See docs/architecture.md for limitations that need contributors.
27
+
28
+
29
+ Normal tests run offline with real socket connections and child-process execution blocked. Do not add xfail to supported corpus cases or reduce independently enforced 95% line/90% branch gates. Keep research cases outside supported recall totals. Upstream fixtures require pinned commits, content hashes, retained licenses and independently reviewed expected operations.
sad_py-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 SAD contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
sad_py-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,122 @@
1
+ Metadata-Version: 2.5
2
+ Name: sad-py
3
+ Version: 0.1.0
4
+ Summary: Static, evidence-backed discovery of frontend API operations
5
+ Author: SAD contributors
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Keywords: api,burp,javascript,security,static-analysis
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Typing :: Typed
12
+ Requires-Python: >=3.11
13
+ Requires-Dist: defusedxml<0.8,>=0.7
14
+ Requires-Dist: httpx<0.29,>=0.28
15
+ Requires-Dist: tree-sitter-javascript<0.26,>=0.23
16
+ Requires-Dist: tree-sitter-typescript<0.24,>=0.23
17
+ Requires-Dist: tree-sitter<0.26,>=0.25
18
+ Provides-Extra: dev
19
+ Requires-Dist: build>=1.2; extra == 'dev'
20
+ Requires-Dist: hypothesis>=6; extra == 'dev'
21
+ Requires-Dist: jsonschema>=4.23; extra == 'dev'
22
+ Requires-Dist: mypy>=1.15; extra == 'dev'
23
+ Requires-Dist: pytest-cov>=6; extra == 'dev'
24
+ Requires-Dist: pytest-timeout>=2.3; extra == 'dev'
25
+ Requires-Dist: pytest>=8; extra == 'dev'
26
+ Requires-Dist: ruff>=0.11; extra == 'dev'
27
+ Requires-Dist: twine>=6; extra == 'dev'
28
+ Requires-Dist: types-defusedxml; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # SAD: Statistic API Discovery
32
+
33
+ SAD is a Python library and CLI for evidence-backed static discovery of frontend API operations, including operations absent from captured traffic. Intended for authorized application analysis and security testing. Target JavaScript is never executed.
34
+
35
+ **Status:** early release, with a bounded abstract interpreter and explicit coverage limits; not a claim of complete JavaScript semantics or exhaustive discovery.
36
+
37
+ ## Install
38
+
39
+ Requires Python 3.11+. From this repository:
40
+
41
+ ```sh
42
+ pip install .
43
+ # Contributors
44
+ pip install -e '.[dev]'
45
+ ```
46
+
47
+ The PyPI distribution name is `sad-py`; the Python import and executable are `sad`. Once published, install it with `pip install sad-py`. No release has been published by this repository setup.
48
+
49
+ ## Python
50
+
51
+ ```python
52
+ from sad import Analyzer
53
+ from sad.ingest import BurpImporter
54
+
55
+ analyzer = Analyzer()
56
+ result = analyzer.analyze_url("https://example.com")
57
+ for endpoint in result.endpoints:
58
+ print(endpoint.method, endpoint.url, endpoint.evidence)
59
+
60
+ result = analyzer.analyze_dataset(BurpImporter.from_xml("history.xml"))
61
+ result = analyzer.analyze_js('fetch(`/users/${id}`, {method: "DELETE"})')
62
+ result = analyzer.analyze_path("./assets")
63
+ ```
64
+
65
+ ## CLI
66
+
67
+ ```sh
68
+ sad crawl https://example.com --depth 2 --max-pages 30
69
+ sad analyze ./assets --format json
70
+ sad analyze-js app.js --base-url https://example.com
71
+ sad import-burp history.xml --format jsonl
72
+ sad import-burp requests.json --format csv
73
+ sad import-burp traffic.har
74
+ sad import-burp request.http --base-url https://example.com
75
+ ```
76
+
77
+ The crawler follows canonical redirects of the starting URL between a host and its `www` variant, and from HTTP to HTTPS on standard ports. It adds the redirected origin to the collection scope so its pages and scripts can be collected. Other origins require explicit scope configuration through `--allow-origin` or `CrawlConfig.allowed_origins`; exclusions and redirect limits still apply. Discovering an API never causes it to be requested.
78
+
79
+ ## Analysis budgets and partial results
80
+
81
+ `--depth` and `--max-pages` control collection. They do not increase the JavaScript analysis budget. Every command accepts `--max-steps` (default: 200,000), corresponding to `Analyzer(max_steps=...)`:
82
+
83
+ ```sh
84
+ sad crawl https://www.google.com/ --depth 2 --max-pages 30 --max-steps 2000000 --format json --strict
85
+ ```
86
+
87
+ The global step budget is divided among supplied JavaScript sources so one expensive script cannot prevent independent sources from being analyzed. When a source exhausts its share, completed discoveries and export bindings are retained, a located `analysis-budget` issue is emitted and other sources continue. Raising the budget allows more exploration, but does not guarantee completeness or remove other limits.
88
+
89
+ `Argument combinations exceeded 16` reports widening of too many possible argument combinations. SAD keeps explored alternatives and a symbolic remainder and sets `analysis_incomplete=true`; it does not claim that every concrete combination was analyzed. Loops, recursion and missing inputs can also cause partial results. The CLI explicitly marks incomplete output, and `--strict` exits with status 1 when diagnostics occur.
90
+
91
+ `high` confidence applies to an individual detected operation, not to coverage of the whole application. An HTML form can produce an endpoint even if JavaScript analysis is incomplete; inspect its `evidence` in JSON to distinguish form declarations from JavaScript sinks.
92
+
93
+ ## Burp history: observed and previously unseen APIs
94
+
95
+ SAD imports the captured requests **and analyzes captured response bodies**. HTML responses contribute inline scripts and forms; JavaScript responses feed the same static analysis engine used for source files. This can discover operations that were never invoked during the captured session. Assets absent from the capture are not downloaded by an offline import.
96
+
97
+ For example, a captured `app.js` response might contain:
98
+
99
+ ```javascript
100
+ fetch("/api/users", {method: "POST"});
101
+ fetch(`/api/users/${id}`, {method: "DELETE"});
102
+ fetch("/api/admin/export");
103
+ ```
104
+
105
+ If the history contains `POST /api/users`, SAD correlates that request with the first static discovery. If it also contains `DELETE /api/users/123`, SAD can correlate it with the symbolic `/api/users/{id}`. The export operation remains a static discovery even if no export request appears in the history. Original observations and contributing evidence are retained.
106
+
107
+ ## Endpoint origin states
108
+
109
+ The `origin` field describes **where the evidence came from**:
110
+
111
+ | State | Meaning | Example |
112
+ | --- | --- | --- |
113
+ | `static-analysis` | Discovered by reading source code or an HTML declaration, including code inside captured responses. No matching request was supplied. | A captured script calls `fetch("/api/admin/export")`, but that URL is absent from the history. |
114
+ | `observed` | A request exists in the supplied traffic, with no matching static discovery. SAD did not make that request. | Burp recorded `GET /api/profile`, but the supplied code does not expose a corresponding call. |
115
+ | `observed+static` | An observed request was correlated with a static discovery. The logical endpoint retains both evidence sources. | Code declares `DELETE /api/users/{id}` and Burp recorded `DELETE /api/users/123`. |
116
+ | `inferred` | Reserved for future discoveries based on indirect heuristics that cannot yet be traced to a resolved network sink or an observed request. **The current engine does not assign this state automatically.** | A future adapter might suggest a route from incomplete framework metadata. |
117
+
118
+ `origin` and `confidence` answer different questions: origin identifies the evidence source; confidence expresses certainty in the mapping. An unresolved URL such as `/api/users/{id}` remains `static-analysis` with symbolic values and appropriate confidence. Resolving a wrapper to `fetch` is also static analysis; the word “inference” does not automatically imply `origin="inferred"`.
119
+
120
+ These states never cause SAD to invoke an endpoint. SAD never executes target JavaScript. Live crawling only collects source resources; importing files or captured traffic is offline.
121
+
122
+ See [measured reliability and limitations](docs/reliability.md), [architecture and coverage](docs/architecture.md), [integration contract](docs/integrations.md), and [release checklist](docs/releasing.md). JSON, JSONL, CSV and table outputs share the same library models. Diagnostics indicate malformed inputs and exhausted budgets; incomplete results are retained.
sad_py-0.1.0/README.md ADDED
@@ -0,0 +1,92 @@
1
+ # SAD: Statistic API Discovery
2
+
3
+ SAD is a Python library and CLI for evidence-backed static discovery of frontend API operations, including operations absent from captured traffic. Intended for authorized application analysis and security testing. Target JavaScript is never executed.
4
+
5
+ **Status:** early release, with a bounded abstract interpreter and explicit coverage limits; not a claim of complete JavaScript semantics or exhaustive discovery.
6
+
7
+ ## Install
8
+
9
+ Requires Python 3.11+. From this repository:
10
+
11
+ ```sh
12
+ pip install .
13
+ # Contributors
14
+ pip install -e '.[dev]'
15
+ ```
16
+
17
+ The PyPI distribution name is `sad-py`; the Python import and executable are `sad`. Once published, install it with `pip install sad-py`. No release has been published by this repository setup.
18
+
19
+ ## Python
20
+
21
+ ```python
22
+ from sad import Analyzer
23
+ from sad.ingest import BurpImporter
24
+
25
+ analyzer = Analyzer()
26
+ result = analyzer.analyze_url("https://example.com")
27
+ for endpoint in result.endpoints:
28
+ print(endpoint.method, endpoint.url, endpoint.evidence)
29
+
30
+ result = analyzer.analyze_dataset(BurpImporter.from_xml("history.xml"))
31
+ result = analyzer.analyze_js('fetch(`/users/${id}`, {method: "DELETE"})')
32
+ result = analyzer.analyze_path("./assets")
33
+ ```
34
+
35
+ ## CLI
36
+
37
+ ```sh
38
+ sad crawl https://example.com --depth 2 --max-pages 30
39
+ sad analyze ./assets --format json
40
+ sad analyze-js app.js --base-url https://example.com
41
+ sad import-burp history.xml --format jsonl
42
+ sad import-burp requests.json --format csv
43
+ sad import-burp traffic.har
44
+ sad import-burp request.http --base-url https://example.com
45
+ ```
46
+
47
+ The crawler follows canonical redirects of the starting URL between a host and its `www` variant, and from HTTP to HTTPS on standard ports. It adds the redirected origin to the collection scope so its pages and scripts can be collected. Other origins require explicit scope configuration through `--allow-origin` or `CrawlConfig.allowed_origins`; exclusions and redirect limits still apply. Discovering an API never causes it to be requested.
48
+
49
+ ## Analysis budgets and partial results
50
+
51
+ `--depth` and `--max-pages` control collection. They do not increase the JavaScript analysis budget. Every command accepts `--max-steps` (default: 200,000), corresponding to `Analyzer(max_steps=...)`:
52
+
53
+ ```sh
54
+ sad crawl https://www.google.com/ --depth 2 --max-pages 30 --max-steps 2000000 --format json --strict
55
+ ```
56
+
57
+ The global step budget is divided among supplied JavaScript sources so one expensive script cannot prevent independent sources from being analyzed. When a source exhausts its share, completed discoveries and export bindings are retained, a located `analysis-budget` issue is emitted and other sources continue. Raising the budget allows more exploration, but does not guarantee completeness or remove other limits.
58
+
59
+ `Argument combinations exceeded 16` reports widening of too many possible argument combinations. SAD keeps explored alternatives and a symbolic remainder and sets `analysis_incomplete=true`; it does not claim that every concrete combination was analyzed. Loops, recursion and missing inputs can also cause partial results. The CLI explicitly marks incomplete output, and `--strict` exits with status 1 when diagnostics occur.
60
+
61
+ `high` confidence applies to an individual detected operation, not to coverage of the whole application. An HTML form can produce an endpoint even if JavaScript analysis is incomplete; inspect its `evidence` in JSON to distinguish form declarations from JavaScript sinks.
62
+
63
+ ## Burp history: observed and previously unseen APIs
64
+
65
+ SAD imports the captured requests **and analyzes captured response bodies**. HTML responses contribute inline scripts and forms; JavaScript responses feed the same static analysis engine used for source files. This can discover operations that were never invoked during the captured session. Assets absent from the capture are not downloaded by an offline import.
66
+
67
+ For example, a captured `app.js` response might contain:
68
+
69
+ ```javascript
70
+ fetch("/api/users", {method: "POST"});
71
+ fetch(`/api/users/${id}`, {method: "DELETE"});
72
+ fetch("/api/admin/export");
73
+ ```
74
+
75
+ If the history contains `POST /api/users`, SAD correlates that request with the first static discovery. If it also contains `DELETE /api/users/123`, SAD can correlate it with the symbolic `/api/users/{id}`. The export operation remains a static discovery even if no export request appears in the history. Original observations and contributing evidence are retained.
76
+
77
+ ## Endpoint origin states
78
+
79
+ The `origin` field describes **where the evidence came from**:
80
+
81
+ | State | Meaning | Example |
82
+ | --- | --- | --- |
83
+ | `static-analysis` | Discovered by reading source code or an HTML declaration, including code inside captured responses. No matching request was supplied. | A captured script calls `fetch("/api/admin/export")`, but that URL is absent from the history. |
84
+ | `observed` | A request exists in the supplied traffic, with no matching static discovery. SAD did not make that request. | Burp recorded `GET /api/profile`, but the supplied code does not expose a corresponding call. |
85
+ | `observed+static` | An observed request was correlated with a static discovery. The logical endpoint retains both evidence sources. | Code declares `DELETE /api/users/{id}` and Burp recorded `DELETE /api/users/123`. |
86
+ | `inferred` | Reserved for future discoveries based on indirect heuristics that cannot yet be traced to a resolved network sink or an observed request. **The current engine does not assign this state automatically.** | A future adapter might suggest a route from incomplete framework metadata. |
87
+
88
+ `origin` and `confidence` answer different questions: origin identifies the evidence source; confidence expresses certainty in the mapping. An unresolved URL such as `/api/users/{id}` remains `static-analysis` with symbolic values and appropriate confidence. Resolving a wrapper to `fetch` is also static analysis; the word “inference” does not automatically imply `origin="inferred"`.
89
+
90
+ These states never cause SAD to invoke an endpoint. SAD never executes target JavaScript. Live crawling only collects source resources; importing files or captured traffic is offline.
91
+
92
+ See [measured reliability and limitations](docs/reliability.md), [architecture and coverage](docs/architecture.md), [integration contract](docs/integrations.md), and [release checklist](docs/releasing.md). JSON, JSONL, CSV and table outputs share the same library models. Diagnostics indicate malformed inputs and exhausted budgets; incomplete results are retained.
@@ -0,0 +1,9 @@
1
+ # Security policy
2
+
3
+ SAD analyzes untrusted code without executing it. Report vulnerabilities through the repository's GitHub **Security → Report a vulnerability** facility once private reporting is enabled. If unavailable, open an issue requesting a private contact without exploit details or sensitive data. There is no staffed response SLA yet. Security fixes target the latest release.
4
+
5
+ The AST engine has byte, step, recursion and cache limits. XML imports use defusedxml. Source-map source names are virtual identifiers and never authorize local reads. Crawler redirects are scope-checked, TLS verification stays enabled, and ambient proxy credentials are disabled. Imported captures and local analysis do not trigger network access.
6
+
7
+ The crawler intentionally supports internal targets for authorized assessments. Its same-origin policy is not an SSRF defense for an internet-facing service: DNS rebinding and access to explicitly seeded private hosts are not blocked. Integrators accepting untrusted crawl seeds must enforce network egress policy outside SAD. Run hostile workloads in a resource-limited worker; native parsers and bounded interpretation are not a process isolation boundary.
8
+
9
+ Output preserves request headers and bodies, which can include credentials and personal information. Protect output as sensitive assessment data. Raw HTTP import expects decoded, dechunked bodies; it does not transparently decompress captures.
@@ -0,0 +1,101 @@
1
+ # Architecture and coverage
2
+
3
+ SAD is a bounded static analyzer, not a JavaScript runtime. Its success measure is evidence-backed coverage of potentially invokable operations. Results are candidates, not proof of reachability, successful authorization, or server existence.
4
+
5
+ ```mermaid
6
+ flowchart LR
7
+ A[Files / captured traffic / scoped crawler] --> B[Dataset]
8
+ B --> C[HTML and source-map expansion]
9
+ C --> D[Cached JS/TS ASTs]
10
+ D --> E[Modules and abstract value propagation]
11
+ E --> F[Resolved network sinks]
12
+ F --> G[Normalization and correlation]
13
+ B --> G
14
+ G --> H[Python / JSON / JSONL / CSV / CLI]
15
+ ```
16
+
17
+ ## Module responsibilities
18
+
19
+ | Module | Responsibility |
20
+ | --- | --- |
21
+ | `models` | Public input, evidence, endpoint and result dataclasses |
22
+ | `ingest` | Capture adapters and bounded input reads |
23
+ | `html` | HTMLParser-based scripts, links and forms |
24
+ | `javascript/parser` | Tree-sitter grammars and locked content-addressed LRU |
25
+ | `javascript/values` | Unknown, function closure, client, bound member and value unions |
26
+ | `javascript/engine` | Module graph, environments, calls, bounded interpretation and sink adapters |
27
+ | `javascript/resources` | AST import discovery and source-map embedded sources |
28
+ | `protocols` | Extensible package/global sink registry |
29
+ | `crawler` | Scoped resource acquisition; never invokes discovered APIs |
30
+ | `normalization` | Canonicalization, exact/symbolic correlation and deduplication |
31
+ | `output`, `cli` | Frontends over the same result objects |
32
+
33
+ The engine loads available modules on demand, links imports, resolves constants/objects/member accesses, and propagates abstract arguments through calls. Calls record caller/callee edges. Uncalled function bodies are explored with symbolic parameters to discover hidden APIs. This can emit both a broad wrapper candidate and a specialized caller result. Branches are explored independently and simple scalar assignments are joined into bounded alternatives. Aliased known clients remain sinks; an arbitrary object method called `get` does not.
34
+
35
+ ## Current coverage
36
+
37
+ | Input or pattern | Implemented behavior | Limits |
38
+ | --- | --- | --- |
39
+ | JS, TS, JSX, TSX, minified code | Tree-sitter parsing and error recovery | Unsupported syntax is traversed for nested sinks; not full semantics |
40
+ | Strings and configuration | Constants, concatenation, templates, objects, spread, explicit environment values | No arbitrary built-in execution or computed string algorithms |
41
+ | Calls | Function declarations, arrows, aliases, destructuring, nested wrappers, returned closures, basic classes | Basic inheritance, receivers and `bind`/`call`/`apply` are supported; decorators, prototype mutation and a full async scheduling model remain incomplete |
42
+ | Branches | Isolated abstract branch heaps, return unions, aliases, bounded loops and recursive summaries | Up to 16 alternatives, 64 loop iterations and eight recursive-summary rounds; widening and recursion emit issues. Exception paths are conservative. |
43
+ | Modules | Relative ESM/CommonJS, default/named exports, re-exports, namespace imports, literal dynamic imports | Cyclic binding cells and explicitly supplied path mappings are supported; package exports and automatic tsconfig loading remain incomplete |
44
+ | fetch, XHR, beacon | URLs, methods, request objects, headers and bodies | XHR open/send/header mutations are linked; response semantics remain incomplete |
45
+ | Axios, jQuery | Calls, HTTP verbs, factory base URLs and headers | Factory/default configuration and request overrides supported; interceptors remain incomplete |
46
+ | ky, got, ofetch, unfetch, request, superagent, wretch | Imported client conventions, common factories, verb calls; basic wretch chains | Fluent response/body modifiers and advanced per-library options incomplete |
47
+ | Angular | Imported HttpClient parameter annotation recognized | Framework DI/decorators and constructor property promotion incomplete |
48
+ | Custom/React/Vue/Svelte/generated clients | Follow their source to known sinks; explicit package registration | Opaque external implementations need adapters; no framework-name guessing |
49
+ | GraphQL/Apollo/urql/graphql-request | Configured URLs and common query/request calls; JSON query payload detection | Operation parsing, Apollo link composition, Relay artifacts/subscriptions incomplete |
50
+ | WebSocket, Socket.IO, SSE | Connection targets from constructors/imported clients | Event channels, reconnection configuration, Socket.IO transport path not modeled |
51
+ | JSON-RPC, SOAP | Structured JSON-RPC bodies and SOAP content-type/header evidence | No WSDL/service discovery |
52
+ | gRPC-Web | `GrpcWebClientBase` RPC URL calls and content-type evidence | Protobuf reflection and opaque generated modules incomplete |
53
+ | Connect-Web, tRPC | Known transport/client factories and conventional method/procedure paths | Literal Connect descriptors supply wire names; opaque protobuf encodings, batching and advanced link chains remain incomplete |
54
+ | Forms | Declarative forms, controls, basic `createElement('form')` action/method/submit | No DOM tree execution, event-driven mutation, form-owner lookup or shadow DOM |
55
+ | Bundles/chunks | Parse ordinary JS in bundles, inspect uncalled factories, follow literal imports | Static numeric Webpack/Parcel module tables and literal ESM chunks are supported; arbitrary loader execution and computed manifests remain unsupported |
56
+ | Source maps | External maps supplied/collected, base64 inline maps, embedded sources with original names/lines | VLQ original locations and inline indexed sections are supported; external indexed sections must be supplied, and source names never authorize filesystem reads |
57
+
58
+ ## Confidence and evidence
59
+
60
+ Every endpoint has AST/form/capture evidence and a location when available. Calls through wrappers include their call chain; configured bases include factory evidence. Source-map embedded sources use `map#section-N:source-N:original-name` provenance.
61
+
62
+ - **High:** a resolved known network sink and concrete URL; observed traffic and HTML declarations also qualify. A high score does not prove runtime reachability.
63
+ - **Medium:** symbolic URL segments, or conventional tRPC/Connect paths whose wire representation may differ.
64
+ - **Low:** a URL beginning with an unresolved symbolic base.
65
+
66
+ Unknown metadata uses JSON null, and symbolic strings use `{id}` or `{dynamic}`. Unknown methods are null, never silently asserted as GET. GET is used only for APIs with that default. `origin` is `static-analysis`, `observed`, `observed+static`, or reserved `inferred`; current protocol heuristics use static-analysis with medium confidence.
67
+
68
+ ## Correlation
69
+
70
+ Canonicalization lowercases scheme/host, removes fragments and default ports, and preserves path case, encoded delimiters and query order. Correlation requires matching protocol, origin and method, preferring exact paths over symbolic paths. Queries are retained in parameters and observations. Multiple different numeric observed paths with a shared structure can be grouped under `{id}`; a single numeric path is kept concrete. This is a heuristic, not a claim of server routing semantics. Original URLs, headers, bodies, status and evidence remain attached.
71
+
72
+ A logical endpoint's top-level request metadata is a representative, not a complete union of every possible static payload. Evidence retains the contributing calls; raw traffic variants remain in observations and merged static configurations in request_variants. Multiple GraphQL operations sharing a transport URL are not yet modeled as separate operations.
73
+
74
+ ## Budgets, determinism and concurrency
75
+
76
+ Default AST cache: 128 entries, keyed by grammar and SHA-256 content, with an 8 MB per-resource limit. Default interpreter budgets: 200,000 steps, call depth 20, 16 branch alternatives, 64 loop iterations and eight recursive-summary rounds. Reuse an Analyzer to avoid reparsing unchanged content; value evaluation remains per analysis. The cache lock protects parsing, but parallel parsing is not enabled. For parallel analyses, use separate Analyzer instances per worker rather than modifying a registry concurrently.
77
+
78
+ The step limit is a global bound, divided into source shares after parsing. Each source's top-level code and function exploration consume its share; imported code consumes the defining source's share. Exhausting a source retains its partial exports and operations and emits an `analysis-budget` issue with that source's location. Independent sources continue. Unused shares are not reassigned, and a budget too small to visit every source still stops at the global bound. The CLI exposes `--max-steps`; `--depth` and `--max-pages` govern collection rather than abstract interpretation. Increasing steps does not change widening, recursion or loop limits.
79
+
80
+ Crawling defaults to depth 2, 30 pages, 300 resources, 8 MB per response, 64 MB aggregate decoded bytes, a 10-second request timeout, and five redirects. Diagnostics identify truncation and unavailable imports. Inputs are analyzed in deterministic order and results are sorted. This is not a whole-project incremental data-flow database.
81
+
82
+ Crawler resources preserve their document URL for relative fetch semantics. Offline captured JS uses its asset URL when neither `Resource.document_url` nor `base_url` supplies document context; provide that context when it matters. An explicit base applies across the analysis. Local source-map names are never read from disk automatically.
83
+
84
+ ## Next substantial milestones
85
+
86
+ 1. A complete control-flow graph and stronger exception/prototype semantics beyond the bounded structured analysis.
87
+ 2. Declarative, independently testable sink adapters and complete generated protocol descriptors.
88
+ 3. Computed bundle/chunk manifests, complete client mutation tracking and opaque generated protocol descriptors.
89
+ 4. Broader real-world corpora, persistent incremental summaries and safely parallel parsing.
90
+ 5. Native Burp Montoya companion, per-operation GraphQL/RPC identities, OpenAPI export, diffs and issue generation.
91
+
92
+ These are outstanding engineering work, not advertised as implemented behavior.
93
+
94
+
95
+ ## Structured incompleteness
96
+
97
+ `AnalysisResult.issues` contains structured codes, reasons and source locations. `analysis_incomplete` is true when parsing recovered errors, imports or wire descriptors are absent, URL/method components remain symbolic, or a resource/value/loop/recursion budget was reached. Existing string diagnostics remain available. No issues does not establish exhaustive JavaScript coverage.
98
+
99
+ `Analyzer(path_mappings={"@api/*": "/lib/*"})` resolves aliases exclusively against supplied resources. `Resource.document_url` supplies document context without any network access. JSON schema 1.0 keeps the new result fields optional for validation of older outputs; new SAD results always include them.
100
+
101
+ The [reliability report](reliability.md) documents the labeled corpus, measured code coverage, property tests and retained research cases.