dorksmith 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dorksmith-0.1.0/.gitignore +55 -0
- dorksmith-0.1.0/LICENSE +21 -0
- dorksmith-0.1.0/PKG-INFO +128 -0
- dorksmith-0.1.0/README.md +78 -0
- dorksmith-0.1.0/pyproject.toml +50 -0
- dorksmith-0.1.0/scripts/sync_catalogs.py +40 -0
- dorksmith-0.1.0/src/dorksmith/__init__.py +56 -0
- dorksmith-0.1.0/src/dorksmith/analyzer.py +151 -0
- dorksmith-0.1.0/src/dorksmith/catalogs.py +73 -0
- dorksmith-0.1.0/src/dorksmith/cli.py +124 -0
- dorksmith-0.1.0/src/dorksmith/data/filetypes.json +68 -0
- dorksmith-0.1.0/src/dorksmith/data/intents.json +712 -0
- dorksmith-0.1.0/src/dorksmith/data/operators.google.json +590 -0
- dorksmith-0.1.0/src/dorksmith/data/platforms.json +118 -0
- dorksmith-0.1.0/src/dorksmith/errors.py +15 -0
- dorksmith-0.1.0/src/dorksmith/generator.py +334 -0
- dorksmith-0.1.0/src/dorksmith/handles.py +135 -0
- dorksmith-0.1.0/src/dorksmith/infer.py +41 -0
- dorksmith-0.1.0/src/dorksmith/normalizer.py +213 -0
- dorksmith-0.1.0/src/dorksmith/placeholders.py +34 -0
- dorksmith-0.1.0/src/dorksmith/py.typed +0 -0
- dorksmith-0.1.0/src/dorksmith/quoting.py +82 -0
- dorksmith-0.1.0/src/dorksmith/ranker.py +131 -0
- dorksmith-0.1.0/src/dorksmith/resolver.py +222 -0
- dorksmith-0.1.0/src/dorksmith/validator.py +184 -0
- dorksmith-0.1.0/tests/test_golden.py +39 -0
- dorksmith-0.1.0/tests/test_unit.py +153 -0
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Spec (kept local)
|
|
2
|
+
SPECS.md
|
|
3
|
+
specs.md
|
|
4
|
+
|
|
5
|
+
# .NET
|
|
6
|
+
bin/
|
|
7
|
+
obj/
|
|
8
|
+
*.user
|
|
9
|
+
*.suo
|
|
10
|
+
.vs/
|
|
11
|
+
TestResults/
|
|
12
|
+
*.trx
|
|
13
|
+
|
|
14
|
+
# State / data
|
|
15
|
+
state/
|
|
16
|
+
*.db
|
|
17
|
+
*.db-journal
|
|
18
|
+
*.db-wal
|
|
19
|
+
*.db-shm
|
|
20
|
+
|
|
21
|
+
# Node (test-only, if ever added)
|
|
22
|
+
node_modules/
|
|
23
|
+
playwright-report/
|
|
24
|
+
test-results/
|
|
25
|
+
|
|
26
|
+
# OS
|
|
27
|
+
.DS_Store
|
|
28
|
+
Thumbs.db
|
|
29
|
+
desktop.ini
|
|
30
|
+
|
|
31
|
+
# Env / secrets
|
|
32
|
+
.env
|
|
33
|
+
.env.*
|
|
34
|
+
!.env.example
|
|
35
|
+
|
|
36
|
+
# Tooling caches
|
|
37
|
+
graft/
|
|
38
|
+
.claude/
|
|
39
|
+
|
|
40
|
+
# Playwright
|
|
41
|
+
web/tests/node_modules/
|
|
42
|
+
web/tests/playwright-report/
|
|
43
|
+
web/tests/test-results/
|
|
44
|
+
|
|
45
|
+
# Packages
|
|
46
|
+
packages/dorksmith-js/node_modules/
|
|
47
|
+
packages/dorksmith-js/dist/
|
|
48
|
+
packages/dorksmith-py/dist/
|
|
49
|
+
packages/dorksmith-py/build/
|
|
50
|
+
packages/dorksmith-py/*.egg-info/
|
|
51
|
+
packages/dorksmith-py/src/*.egg-info/
|
|
52
|
+
__pycache__/
|
|
53
|
+
*.pyc
|
|
54
|
+
.pytest_cache/
|
|
55
|
+
.venv/
|
dorksmith-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 aelena
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
dorksmith-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dorksmith
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Deterministic search-dork generator and OSINT/SOCMINT query engine. Catalog-driven, pure Python, no dependencies.
|
|
5
|
+
Project-URL: Homepage, https://github.com/aelena/dorksmith
|
|
6
|
+
Project-URL: Repository, https://github.com/aelena/dorksmith
|
|
7
|
+
Project-URL: Issues, https://github.com/aelena/dorksmith/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/aelena/dorksmith/commits/main/packages/dorksmith-py
|
|
9
|
+
Author: aelena
|
|
10
|
+
License: MIT License
|
|
11
|
+
|
|
12
|
+
Copyright (c) 2026 aelena
|
|
13
|
+
|
|
14
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
15
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
16
|
+
in the Software without restriction, including without limitation the rights
|
|
17
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
18
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
19
|
+
furnished to do so, subject to the following conditions:
|
|
20
|
+
|
|
21
|
+
The above copyright notice and this permission notice shall be included in all
|
|
22
|
+
copies or substantial portions of the Software.
|
|
23
|
+
|
|
24
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
25
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
26
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
27
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
28
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
29
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
30
|
+
SOFTWARE.
|
|
31
|
+
License-File: LICENSE
|
|
32
|
+
Keywords: deterministic,dork,google-dorks,osint,search-operators,security,socmint
|
|
33
|
+
Classifier: Development Status :: 4 - Beta
|
|
34
|
+
Classifier: Intended Audience :: Information Technology
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Operating System :: OS Independent
|
|
37
|
+
Classifier: Programming Language :: Python :: 3
|
|
38
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
39
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
40
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
41
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
42
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
43
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
44
|
+
Classifier: Topic :: Security
|
|
45
|
+
Classifier: Typing :: Typed
|
|
46
|
+
Requires-Python: >=3.10
|
|
47
|
+
Provides-Extra: test
|
|
48
|
+
Requires-Dist: pytest>=8; extra == 'test'
|
|
49
|
+
Description-Content-Type: text/markdown
|
|
50
|
+
|
|
51
|
+
# dorksmith (PyPI)
|
|
52
|
+
|
|
53
|
+
Deterministic search-dork generator and OSINT/SOCMINT query engine as a pure-Python package with no dependencies (Python ≥ 3.10). Same engine and catalogs as the [Dorksmith](https://github.com/aelena/dorksmith) web app and the npm package, conformance-tested against the same golden fixtures.
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
pip install dorksmith
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from dorksmith import generate, expand_handle, validate_query, search_url
|
|
61
|
+
|
|
62
|
+
result = generate("example.com", "domain", "public-documents",
|
|
63
|
+
options={"fileTypes": ["pdf", "docx"], "excludeTerms": ["jobs"], "maxVariants": 3})
|
|
64
|
+
for v in result.variants:
|
|
65
|
+
print(f"{v.label:<14} {v.query}")
|
|
66
|
+
# Balanced site:example.com (filetype:pdf OR filetype:docx) -jobs
|
|
67
|
+
# Precise site:example.com filetype:pdf -jobs
|
|
68
|
+
# Title-focused site:example.com (filetype:pdf OR filetype:docx) (intitle:report OR intitle:policy OR intitle:presentation OR intitle:"annual report") -jobs
|
|
69
|
+
|
|
70
|
+
search_url(result.variants[0].query)
|
|
71
|
+
# 'https://www.google.com/search?q=site%3Aexample.com%20%28filetype%3Apdf%20OR%20filetype%3Adocx%29%20-jobs'
|
|
72
|
+
|
|
73
|
+
handle = expand_handle("@alice42", categories=["developer"], max_platforms=3)
|
|
74
|
+
handle.profiles[0].url, handle.profiles[0].status
|
|
75
|
+
# ('https://github.com/alice42', 'not-checked')
|
|
76
|
+
|
|
77
|
+
validate_query("cache:example.com site:example.com or filetype:.pdf").warnings
|
|
78
|
+
# ["Lower-case 'or' is treated as an ordinary word; ...", "filetype: values take no leading dot ...", "cache: is deprecated: ..."]
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Results are dataclasses; `.to_dict()` gives the same camelCase shape as the HTTP API (`POST /api/v1/dorks/generate`), minus `requestId` and `rateLimit`.
|
|
82
|
+
|
|
83
|
+
## API
|
|
84
|
+
|
|
85
|
+
| Function | Purpose |
|
|
86
|
+
|---|---|
|
|
87
|
+
| `generate(input, input_type, intent, *, engine="google", options=None, catalogs=None, limits=DEFAULT_LIMITS)` | Ranked variants with explanation, operators, warnings and rank reason. Raises `InputValidationError` (`.field`, `.unprocessable`). |
|
|
88
|
+
| `validate_query(query, engine="google", catalogs=None)` | Operators used with support state and warnings. Never rewrites the query. |
|
|
89
|
+
| `expand_handle(username, *, categories=None, platform_ids=None, max_platforms=None, catalogs=None)` | Profile URLs (`status` always `not-checked`) and site-scoped queries. |
|
|
90
|
+
| `infer_input_type(text, known_extensions=None)` | Deterministic input-type suggestion. |
|
|
91
|
+
| `search_url(query, engine="google")` | Search-engine URL, percent-encoded locally. |
|
|
92
|
+
| `validate_catalogs(catalogs)` | Structural/cross-reference validation, same rules as the API's readiness check. |
|
|
93
|
+
| `bundled_catalogs()`, `load_catalogs(path)`, `catalog_version()` | Embedded or custom catalogs. |
|
|
94
|
+
| lower-level: `normalize_text`, `try_normalize_domain`, `quote`, `safe_term`, `analyze_query`, `expand_template`, … | Building blocks for custom pipelines. |
|
|
95
|
+
|
|
96
|
+
`options` keys match the HTTP API: `fileTypes`, `excludeTerms`, `after`, `before`, `site`, `maxVariants`, `organization`, `location`, `role`, `displayName`.
|
|
97
|
+
|
|
98
|
+
### Bring your own catalogs
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from dorksmith import load_catalogs, validate_catalogs, generate
|
|
102
|
+
catalogs = load_catalogs("path/to/data") # operators.google.json, intents.json, platforms.json, filetypes.json
|
|
103
|
+
assert validate_catalogs(catalogs) == []
|
|
104
|
+
generate("x", "keyword", "general-discovery", catalogs=catalogs)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## CLI
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
dorksmith generate example.com --intent exposed-config-files --max 3
|
|
111
|
+
dorksmith generate "Alice Smith" --intent person-organization --organization "Example Corp"
|
|
112
|
+
dorksmith handle alice42 --categories developer --json
|
|
113
|
+
dorksmith validate 'cache:example.com site:example.com'
|
|
114
|
+
dorksmith intents
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Exit codes: `0` ok, `2` invalid input (or a query with syntax errors for `validate`), `3` valid request that cannot be generated.
|
|
118
|
+
|
|
119
|
+
## Development
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
python -m pip install -e ".[test]"
|
|
123
|
+
python scripts/sync_catalogs.py # copies ../../data/*.json into src/dorksmith/data
|
|
124
|
+
python -m pytest -q # unit tests + the 151 golden fixtures from the monorepo
|
|
125
|
+
python -m build # sdist + wheel
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Publishing is done by the `py-package` GitHub workflow when a `py-v*` tag is pushed, using PyPI trusted publishing (see the repository README).
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# dorksmith (PyPI)
|
|
2
|
+
|
|
3
|
+
Deterministic search-dork generator and OSINT/SOCMINT query engine as a pure-Python package with no dependencies (Python ≥ 3.10). Same engine and catalogs as the [Dorksmith](https://github.com/aelena/dorksmith) web app and the npm package, conformance-tested against the same golden fixtures.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install dorksmith
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from dorksmith import generate, expand_handle, validate_query, search_url
|
|
11
|
+
|
|
12
|
+
result = generate("example.com", "domain", "public-documents",
|
|
13
|
+
options={"fileTypes": ["pdf", "docx"], "excludeTerms": ["jobs"], "maxVariants": 3})
|
|
14
|
+
for v in result.variants:
|
|
15
|
+
print(f"{v.label:<14} {v.query}")
|
|
16
|
+
# Balanced site:example.com (filetype:pdf OR filetype:docx) -jobs
|
|
17
|
+
# Precise site:example.com filetype:pdf -jobs
|
|
18
|
+
# Title-focused site:example.com (filetype:pdf OR filetype:docx) (intitle:report OR intitle:policy OR intitle:presentation OR intitle:"annual report") -jobs
|
|
19
|
+
|
|
20
|
+
search_url(result.variants[0].query)
|
|
21
|
+
# 'https://www.google.com/search?q=site%3Aexample.com%20%28filetype%3Apdf%20OR%20filetype%3Adocx%29%20-jobs'
|
|
22
|
+
|
|
23
|
+
handle = expand_handle("@alice42", categories=["developer"], max_platforms=3)
|
|
24
|
+
handle.profiles[0].url, handle.profiles[0].status
|
|
25
|
+
# ('https://github.com/alice42', 'not-checked')
|
|
26
|
+
|
|
27
|
+
validate_query("cache:example.com site:example.com or filetype:.pdf").warnings
|
|
28
|
+
# ["Lower-case 'or' is treated as an ordinary word; ...", "filetype: values take no leading dot ...", "cache: is deprecated: ..."]
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Results are dataclasses; `.to_dict()` gives the same camelCase shape as the HTTP API (`POST /api/v1/dorks/generate`), minus `requestId` and `rateLimit`.
|
|
32
|
+
|
|
33
|
+
## API
|
|
34
|
+
|
|
35
|
+
| Function | Purpose |
|
|
36
|
+
|---|---|
|
|
37
|
+
| `generate(input, input_type, intent, *, engine="google", options=None, catalogs=None, limits=DEFAULT_LIMITS)` | Ranked variants with explanation, operators, warnings and rank reason. Raises `InputValidationError` (`.field`, `.unprocessable`). |
|
|
38
|
+
| `validate_query(query, engine="google", catalogs=None)` | Operators used with support state and warnings. Never rewrites the query. |
|
|
39
|
+
| `expand_handle(username, *, categories=None, platform_ids=None, max_platforms=None, catalogs=None)` | Profile URLs (`status` always `not-checked`) and site-scoped queries. |
|
|
40
|
+
| `infer_input_type(text, known_extensions=None)` | Deterministic input-type suggestion. |
|
|
41
|
+
| `search_url(query, engine="google")` | Search-engine URL, percent-encoded locally. |
|
|
42
|
+
| `validate_catalogs(catalogs)` | Structural/cross-reference validation, same rules as the API's readiness check. |
|
|
43
|
+
| `bundled_catalogs()`, `load_catalogs(path)`, `catalog_version()` | Embedded or custom catalogs. |
|
|
44
|
+
| lower-level: `normalize_text`, `try_normalize_domain`, `quote`, `safe_term`, `analyze_query`, `expand_template`, … | Building blocks for custom pipelines. |
|
|
45
|
+
|
|
46
|
+
`options` keys match the HTTP API: `fileTypes`, `excludeTerms`, `after`, `before`, `site`, `maxVariants`, `organization`, `location`, `role`, `displayName`.
|
|
47
|
+
|
|
48
|
+
### Bring your own catalogs
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from dorksmith import load_catalogs, validate_catalogs, generate
|
|
52
|
+
catalogs = load_catalogs("path/to/data") # operators.google.json, intents.json, platforms.json, filetypes.json
|
|
53
|
+
assert validate_catalogs(catalogs) == []
|
|
54
|
+
generate("x", "keyword", "general-discovery", catalogs=catalogs)
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## CLI
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
dorksmith generate example.com --intent exposed-config-files --max 3
|
|
61
|
+
dorksmith generate "Alice Smith" --intent person-organization --organization "Example Corp"
|
|
62
|
+
dorksmith handle alice42 --categories developer --json
|
|
63
|
+
dorksmith validate 'cache:example.com site:example.com'
|
|
64
|
+
dorksmith intents
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Exit codes: `0` ok, `2` invalid input (or a query with syntax errors for `validate`), `3` valid request that cannot be generated.
|
|
68
|
+
|
|
69
|
+
## Development
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
python -m pip install -e ".[test]"
|
|
73
|
+
python scripts/sync_catalogs.py # copies ../../data/*.json into src/dorksmith/data
|
|
74
|
+
python -m pytest -q # unit tests + the 151 golden fixtures from the monorepo
|
|
75
|
+
python -m build # sdist + wheel
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Publishing is done by the `py-package` GitHub workflow when a `py-v*` tag is pushed, using PyPI trusted publishing (see the repository README).
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.25"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dorksmith"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Deterministic search-dork generator and OSINT/SOCMINT query engine. Catalog-driven, pure Python, no dependencies."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { file = "LICENSE" }
|
|
11
|
+
authors = [{ name = "aelena" }]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
keywords = ["osint", "socmint", "google-dorks", "search-operators", "dork", "security", "deterministic"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Information Technology",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Programming Language :: Python :: 3.13",
|
|
25
|
+
"Topic :: Security",
|
|
26
|
+
"Topic :: Internet :: WWW/HTTP :: Indexing/Search",
|
|
27
|
+
"Typing :: Typed",
|
|
28
|
+
]
|
|
29
|
+
dependencies = []
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
test = ["pytest>=8"]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://github.com/aelena/dorksmith"
|
|
36
|
+
Repository = "https://github.com/aelena/dorksmith"
|
|
37
|
+
Issues = "https://github.com/aelena/dorksmith/issues"
|
|
38
|
+
Changelog = "https://github.com/aelena/dorksmith/commits/main/packages/dorksmith-py"
|
|
39
|
+
|
|
40
|
+
[project.scripts]
|
|
41
|
+
dorksmith = "dorksmith.cli:main"
|
|
42
|
+
|
|
43
|
+
[tool.hatch.build.targets.wheel]
|
|
44
|
+
packages = ["src/dorksmith"]
|
|
45
|
+
|
|
46
|
+
[tool.hatch.build.targets.sdist]
|
|
47
|
+
include = ["src/dorksmith", "tests", "scripts", "README.md", "LICENSE", "pyproject.toml"]
|
|
48
|
+
|
|
49
|
+
[tool.pytest.ini_options]
|
|
50
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Copy the repository catalogs (../../data/*.json) into src/dorksmith/data.
|
|
3
|
+
|
|
4
|
+
`--check` exits non-zero when the embedded copies differ (used in CI).
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
HERE = Path(__file__).resolve().parent
|
|
12
|
+
PKG = HERE.parent
|
|
13
|
+
SOURCE = PKG.parent.parent / "data"
|
|
14
|
+
TARGET = PKG / "src" / "dorksmith" / "data"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def main() -> int:
|
|
18
|
+
check = "--check" in sys.argv
|
|
19
|
+
if not SOURCE.is_dir():
|
|
20
|
+
print(f"no repository data directory at {SOURCE}; keeping embedded catalogs")
|
|
21
|
+
return 0
|
|
22
|
+
TARGET.mkdir(parents=True, exist_ok=True)
|
|
23
|
+
changed = False
|
|
24
|
+
files = sorted(SOURCE.glob("*.json"))
|
|
25
|
+
for src in files:
|
|
26
|
+
dst = TARGET / src.name
|
|
27
|
+
data = src.read_bytes()
|
|
28
|
+
if not dst.exists() or dst.read_bytes() != data:
|
|
29
|
+
changed = True
|
|
30
|
+
if not check:
|
|
31
|
+
dst.write_bytes(data)
|
|
32
|
+
if check and changed:
|
|
33
|
+
print("embedded catalogs are out of date; run `python scripts/sync_catalogs.py`", file=sys.stderr)
|
|
34
|
+
return 1
|
|
35
|
+
print("embedded catalogs are up to date" if check else f"synced {len(files)} catalog(s)")
|
|
36
|
+
return 0
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""dorksmith — deterministic search-dork generator and OSINT/SOCMINT query engine.
|
|
2
|
+
|
|
3
|
+
from dorksmith import generate, expand_handle, validate_query
|
|
4
|
+
result = generate("example.com", "domain", "public-documents", options={"fileTypes": ["pdf"]})
|
|
5
|
+
for v in result.variants:
|
|
6
|
+
print(v.label, v.query)
|
|
7
|
+
|
|
8
|
+
The engine is pure: same request + same catalogs = same output. Pass ``catalogs=load_catalogs(path)``
|
|
9
|
+
to run with edited catalogs; the bundled ones are used otherwise.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from urllib.parse import quote as _url_quote
|
|
14
|
+
|
|
15
|
+
from .analyzer import GOOGLE_WORD_LIMIT, QueryAnalysis, analyze_query
|
|
16
|
+
from .catalogs import Catalogs, bundled_catalog_files, bundled_catalogs, load_catalogs
|
|
17
|
+
from .errors import InputValidationError
|
|
18
|
+
from .generator import (
|
|
19
|
+
DEFAULT_LIMITS, INPUT_TYPES, GenerateResult, Limits, OperatorUse, ValidateQueryResult, Variant, build_context,
|
|
20
|
+
expand_template, generate, validate_query,
|
|
21
|
+
)
|
|
22
|
+
from .handles import HANDLE_NOTICE, NOT_CHECKED, HandleExpandResult, HandleQuery, ProfileCandidate, escape_data_string, expand_handle
|
|
23
|
+
from .infer import infer_input_type
|
|
24
|
+
from .normalizer import (
|
|
25
|
+
canonicalize, domain_label, normalize_text, try_normalize_domain, try_normalize_email, try_normalize_url,
|
|
26
|
+
try_normalize_username, try_parse_filename, try_parse_iso_date,
|
|
27
|
+
)
|
|
28
|
+
from .placeholders import ALL_PLACEHOLDERS, OPTIONAL_BY_DEFAULT, placeholders_in
|
|
29
|
+
from .quoting import exclusion, looks_like_syntax, operator_value, or_group, quote, safe_term, safe_terms, unquote
|
|
30
|
+
from .validator import validate_catalogs
|
|
31
|
+
|
|
32
|
+
__version__ = "0.1.0"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def catalog_version(catalogs: Catalogs | None = None) -> str:
|
|
36
|
+
"""Version string of the catalogs in use (from intents.json)."""
|
|
37
|
+
return (catalogs or bundled_catalogs()).catalog_version
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def search_url(query: str, engine: str = "google", catalogs: Catalogs | None = None) -> str:
|
|
41
|
+
"""Search-engine URL for a query, built locally with percent-encoding."""
|
|
42
|
+
c = catalogs or bundled_catalogs()
|
|
43
|
+
template = c.operators.get(engine, {}).get("searchUrlTemplate", "https://www.google.com/search?q={query}")
|
|
44
|
+
return template.replace("{query}", _url_quote(query, safe="-_.!~*'()"))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
__all__ = [
|
|
48
|
+
"__version__", "ALL_PLACEHOLDERS", "Catalogs", "DEFAULT_LIMITS", "GOOGLE_WORD_LIMIT", "GenerateResult", "HANDLE_NOTICE",
|
|
49
|
+
"HandleExpandResult", "HandleQuery", "INPUT_TYPES", "InputValidationError", "Limits", "NOT_CHECKED", "OPTIONAL_BY_DEFAULT",
|
|
50
|
+
"OperatorUse", "ProfileCandidate", "QueryAnalysis", "ValidateQueryResult", "Variant", "analyze_query", "build_context",
|
|
51
|
+
"bundled_catalog_files", "bundled_catalogs", "canonicalize", "catalog_version", "domain_label", "escape_data_string",
|
|
52
|
+
"exclusion", "expand_handle", "expand_template", "generate", "infer_input_type", "load_catalogs", "looks_like_syntax",
|
|
53
|
+
"normalize_text", "operator_value", "or_group", "placeholders_in", "quote", "safe_term", "safe_terms", "search_url",
|
|
54
|
+
"try_normalize_domain", "try_normalize_email", "try_normalize_url", "try_normalize_username", "try_parse_filename",
|
|
55
|
+
"try_parse_iso_date", "unquote", "validate_catalogs", "validate_query",
|
|
56
|
+
]
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""Quote-aware operator scanner with plain-language warnings."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from .catalogs import Json
|
|
8
|
+
from .normalizer import is_letter
|
|
9
|
+
|
|
10
|
+
GOOGLE_WORD_LIMIT = 32
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class QueryAnalysis:
|
|
15
|
+
operators: list[str]
|
|
16
|
+
warnings: list[str]
|
|
17
|
+
has_errors: bool
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _tokenize(query: str) -> list[tuple[str, bool]]:
|
|
21
|
+
tokens: list[tuple[str, bool]] = []
|
|
22
|
+
i, n = 0, len(query)
|
|
23
|
+
while i < n:
|
|
24
|
+
if query[i].isspace():
|
|
25
|
+
i += 1
|
|
26
|
+
continue
|
|
27
|
+
if query[i] == '"':
|
|
28
|
+
end = query.find('"', i + 1)
|
|
29
|
+
if end < 0:
|
|
30
|
+
end = n - 1
|
|
31
|
+
tokens.append((query[i + 1: max(i + 1, end)], True))
|
|
32
|
+
i = end + 1
|
|
33
|
+
continue
|
|
34
|
+
start = i
|
|
35
|
+
while i < n and not query[i].isspace():
|
|
36
|
+
if query[i] == '"':
|
|
37
|
+
end = query.find('"', i + 1)
|
|
38
|
+
i = n if end < 0 else end + 1
|
|
39
|
+
continue
|
|
40
|
+
i += 1
|
|
41
|
+
tokens.append((query[start:i], False))
|
|
42
|
+
return tokens
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def analyze_query(query: str, catalog: Json, by_token: dict[str, Json]) -> QueryAnalysis:
|
|
46
|
+
operators: list[str] = []
|
|
47
|
+
warnings: list[str] = []
|
|
48
|
+
has_errors = False
|
|
49
|
+
|
|
50
|
+
def use(token: str) -> None:
|
|
51
|
+
if token not in operators:
|
|
52
|
+
operators.append(token)
|
|
53
|
+
|
|
54
|
+
depth = 0
|
|
55
|
+
tokens = _tokenize(query)
|
|
56
|
+
quote_count = query.count('"')
|
|
57
|
+
if quote_count > 0:
|
|
58
|
+
use('"')
|
|
59
|
+
if quote_count % 2 == 1:
|
|
60
|
+
warnings.append("Unbalanced double quotes: the last phrase will not be treated as exact.")
|
|
61
|
+
has_errors = True
|
|
62
|
+
|
|
63
|
+
for text, quoted in tokens:
|
|
64
|
+
if quoted:
|
|
65
|
+
if "*" in text:
|
|
66
|
+
use("*")
|
|
67
|
+
continue
|
|
68
|
+
t = text
|
|
69
|
+
if not t:
|
|
70
|
+
continue
|
|
71
|
+
if t == "OR":
|
|
72
|
+
use("OR"); continue
|
|
73
|
+
if t == "AND":
|
|
74
|
+
use("AND"); continue
|
|
75
|
+
if t == "|":
|
|
76
|
+
use("|"); continue
|
|
77
|
+
if t in ("or", "and"):
|
|
78
|
+
warnings.append(f"Lower-case '{t}' is treated as an ordinary word; use upper-case {t.upper()} for a boolean operator.")
|
|
79
|
+
continue
|
|
80
|
+
|
|
81
|
+
body = t
|
|
82
|
+
while body and body[0] == "(":
|
|
83
|
+
depth += 1; use("("); body = body[1:]
|
|
84
|
+
closing = 0
|
|
85
|
+
while body and body[-1] == ")":
|
|
86
|
+
closing += 1; body = body[:-1]
|
|
87
|
+
depth -= closing
|
|
88
|
+
|
|
89
|
+
if body.startswith("AROUND("):
|
|
90
|
+
use("AROUND("); continue
|
|
91
|
+
if len(body) > 1 and body[0] == "-":
|
|
92
|
+
use("-"); body = body[1:]
|
|
93
|
+
elif len(body) > 1 and body[0] == "+":
|
|
94
|
+
use("+"); body = body[1:]
|
|
95
|
+
elif len(body) > 1 and body[0] == "~":
|
|
96
|
+
use("~"); body = body[1:]
|
|
97
|
+
if len(body) > 1 and body[0] == "#":
|
|
98
|
+
use("#"); continue
|
|
99
|
+
if len(body) > 1 and body[0] == "@":
|
|
100
|
+
use("@"); continue
|
|
101
|
+
if body == "*":
|
|
102
|
+
use("*"); continue
|
|
103
|
+
if ".." in body and any(ch.isdigit() for ch in body):
|
|
104
|
+
use(".."); continue
|
|
105
|
+
if len(body) > 1 and body[0] == "$" and all(ch.isdigit() or ch in ".," for ch in body[1:]):
|
|
106
|
+
use("$"); continue
|
|
107
|
+
|
|
108
|
+
colon = body.find(":")
|
|
109
|
+
if colon > 0 and all(is_letter(ch) for ch in body[:colon]):
|
|
110
|
+
prefix = body[: colon + 1].lower()
|
|
111
|
+
value = body[colon + 1:]
|
|
112
|
+
op = by_token.get(prefix)
|
|
113
|
+
if op:
|
|
114
|
+
use(prefix)
|
|
115
|
+
if prefix == "filetype:" and value.startswith("."):
|
|
116
|
+
warnings.append("filetype: values take no leading dot (filetype:pdf).")
|
|
117
|
+
if prefix == "site:" and "://" in value:
|
|
118
|
+
warnings.append("site: values take no scheme (site:example.com).")
|
|
119
|
+
if prefix in ("before:", "after:") and len(value) not in (4, 10):
|
|
120
|
+
warnings.append(f"{prefix} expects YYYY-MM-DD or YYYY.")
|
|
121
|
+
if op.get("takesValue") and not value:
|
|
122
|
+
warnings.append(f"{prefix} has no value.")
|
|
123
|
+
has_errors = True
|
|
124
|
+
elif len(body[:colon]) <= 15 and "://" not in body:
|
|
125
|
+
warnings.append(f"'{prefix}' is not a known {catalog.get('engineName', '')} operator and will be searched as plain text.")
|
|
126
|
+
|
|
127
|
+
if depth != 0:
|
|
128
|
+
warnings.append("Unbalanced parentheses.")
|
|
129
|
+
has_errors = True
|
|
130
|
+
|
|
131
|
+
for token in operators:
|
|
132
|
+
op = by_token.get(token)
|
|
133
|
+
if not op:
|
|
134
|
+
continue
|
|
135
|
+
support = op.get("support")
|
|
136
|
+
caveats = op.get("caveats") or []
|
|
137
|
+
if support == "deprecated":
|
|
138
|
+
warnings.append(f"{op['token']} is deprecated: {caveats[0] if caveats else 'it no longer works.'}")
|
|
139
|
+
elif support == "unreliable":
|
|
140
|
+
warnings.append(f"{op['token']} is unreliable: {caveats[0] if caveats else 'results are inconsistent.'}")
|
|
141
|
+
elif support == "unknown":
|
|
142
|
+
warnings.append(f"{op['token']} has unknown support status.")
|
|
143
|
+
|
|
144
|
+
word_count = sum(1 for text, _ in tokens if text)
|
|
145
|
+
if word_count > GOOGLE_WORD_LIMIT:
|
|
146
|
+
warnings.append(f"Query has {word_count} terms; Google ignores everything after the first {GOOGLE_WORD_LIMIT}.")
|
|
147
|
+
|
|
148
|
+
return QueryAnalysis(operators, warnings, has_errors)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
_ = re # keep re available for callers that extend the analyzer
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Catalog loading. Catalogs are plain dicts mirroring the JSON files in the repository's data/ directory."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from importlib import resources
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
Json = dict[str, Any]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class Catalogs:
|
|
15
|
+
"""The four catalogs the engine consumes. ``operators`` is keyed by engine id."""
|
|
16
|
+
|
|
17
|
+
operators: dict[str, Json]
|
|
18
|
+
intents: Json
|
|
19
|
+
platforms: Json
|
|
20
|
+
file_types: Json
|
|
21
|
+
_operator_index: dict[str, dict[str, Json]] = field(default_factory=dict, repr=False, compare=False)
|
|
22
|
+
|
|
23
|
+
@property
|
|
24
|
+
def catalog_version(self) -> str:
|
|
25
|
+
return str(self.intents.get("catalogVersion", ""))
|
|
26
|
+
|
|
27
|
+
def operator_index(self, engine: str) -> dict[str, Json]:
|
|
28
|
+
"""token -> operator definition for an engine (cached)."""
|
|
29
|
+
if engine not in self._operator_index:
|
|
30
|
+
index: dict[str, Json] = {}
|
|
31
|
+
for op in self.operators[engine].get("operators", []):
|
|
32
|
+
index.setdefault(op["token"], op)
|
|
33
|
+
self._operator_index[engine] = index
|
|
34
|
+
return self._operator_index[engine]
|
|
35
|
+
|
|
36
|
+
def intent(self, intent_id: str) -> Json | None:
|
|
37
|
+
for i in self.intents.get("intents", []):
|
|
38
|
+
if i.get("id") == intent_id:
|
|
39
|
+
return i
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _assemble(files: dict[str, Json]) -> Catalogs:
|
|
44
|
+
operators = {v["engine"]: v for k, v in files.items() if k.startswith("operators.")}
|
|
45
|
+
return Catalogs(operators=operators, intents=files["intents.json"], platforms=files["platforms.json"], file_types=files["filetypes.json"])
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def load_catalogs(directory: str | Path) -> Catalogs:
|
|
49
|
+
"""Load catalogs from a directory containing operators.<engine>.json, intents.json, platforms.json, filetypes.json."""
|
|
50
|
+
d = Path(directory)
|
|
51
|
+
files = {p.name: json.loads(p.read_text(encoding="utf-8")) for p in sorted(d.glob("*.json"))}
|
|
52
|
+
return _assemble(files)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def bundled_catalog_files() -> dict[str, Json]:
|
|
56
|
+
"""The raw catalog files embedded in the package, keyed by file name."""
|
|
57
|
+
root = resources.files("dorksmith") / "data"
|
|
58
|
+
out: dict[str, Json] = {}
|
|
59
|
+
for entry in sorted(root.iterdir(), key=lambda e: e.name):
|
|
60
|
+
if entry.name.endswith(".json"):
|
|
61
|
+
out[entry.name] = json.loads(entry.read_text(encoding="utf-8"))
|
|
62
|
+
return out
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
_bundled: Catalogs | None = None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def bundled_catalogs() -> Catalogs:
|
|
69
|
+
"""The catalogs embedded in this package (loaded once)."""
|
|
70
|
+
global _bundled
|
|
71
|
+
if _bundled is None:
|
|
72
|
+
_bundled = _assemble(bundled_catalog_files())
|
|
73
|
+
return _bundled
|