dorksmith 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,55 @@
1
+ # Spec (kept local)
2
+ SPECS.md
3
+ specs.md
4
+
5
+ # .NET
6
+ bin/
7
+ obj/
8
+ *.user
9
+ *.suo
10
+ .vs/
11
+ TestResults/
12
+ *.trx
13
+
14
+ # State / data
15
+ state/
16
+ *.db
17
+ *.db-journal
18
+ *.db-wal
19
+ *.db-shm
20
+
21
+ # Node (test-only, if ever added)
22
+ node_modules/
23
+ playwright-report/
24
+ test-results/
25
+
26
+ # OS
27
+ .DS_Store
28
+ Thumbs.db
29
+ desktop.ini
30
+
31
+ # Env / secrets
32
+ .env
33
+ .env.*
34
+ !.env.example
35
+
36
+ # Tooling caches
37
+ graft/
38
+ .claude/
39
+
40
+ # Playwright
41
+ web/tests/node_modules/
42
+ web/tests/playwright-report/
43
+ web/tests/test-results/
44
+
45
+ # Packages
46
+ packages/dorksmith-js/node_modules/
47
+ packages/dorksmith-js/dist/
48
+ packages/dorksmith-py/dist/
49
+ packages/dorksmith-py/build/
50
+ packages/dorksmith-py/*.egg-info/
51
+ packages/dorksmith-py/src/*.egg-info/
52
+ __pycache__/
53
+ *.pyc
54
+ .pytest_cache/
55
+ .venv/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 aelena
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,128 @@
1
+ Metadata-Version: 2.5
2
+ Name: dorksmith
3
+ Version: 0.1.0
4
+ Summary: Deterministic search-dork generator and OSINT/SOCMINT query engine. Catalog-driven, pure Python, no dependencies.
5
+ Project-URL: Homepage, https://github.com/aelena/dorksmith
6
+ Project-URL: Repository, https://github.com/aelena/dorksmith
7
+ Project-URL: Issues, https://github.com/aelena/dorksmith/issues
8
+ Project-URL: Changelog, https://github.com/aelena/dorksmith/commits/main/packages/dorksmith-py
9
+ Author: aelena
10
+ License: MIT License
11
+
12
+ Copyright (c) 2026 aelena
13
+
14
+ Permission is hereby granted, free of charge, to any person obtaining a copy
15
+ of this software and associated documentation files (the "Software"), to deal
16
+ in the Software without restriction, including without limitation the rights
17
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
18
+ copies of the Software, and to permit persons to whom the Software is
19
+ furnished to do so, subject to the following conditions:
20
+
21
+ The above copyright notice and this permission notice shall be included in all
22
+ copies or substantial portions of the Software.
23
+
24
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
25
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
26
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
27
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
28
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
29
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
30
+ SOFTWARE.
31
+ License-File: LICENSE
32
+ Keywords: deterministic,dork,google-dorks,osint,search-operators,security,socmint
33
+ Classifier: Development Status :: 4 - Beta
34
+ Classifier: Intended Audience :: Information Technology
35
+ Classifier: License :: OSI Approved :: MIT License
36
+ Classifier: Operating System :: OS Independent
37
+ Classifier: Programming Language :: Python :: 3
38
+ Classifier: Programming Language :: Python :: 3 :: Only
39
+ Classifier: Programming Language :: Python :: 3.10
40
+ Classifier: Programming Language :: Python :: 3.11
41
+ Classifier: Programming Language :: Python :: 3.12
42
+ Classifier: Programming Language :: Python :: 3.13
43
+ Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
44
+ Classifier: Topic :: Security
45
+ Classifier: Typing :: Typed
46
+ Requires-Python: >=3.10
47
+ Provides-Extra: test
48
+ Requires-Dist: pytest>=8; extra == 'test'
49
+ Description-Content-Type: text/markdown
50
+
51
+ # dorksmith (PyPI)
52
+
53
+ Deterministic search-dork generator and OSINT/SOCMINT query engine as a pure-Python package with no dependencies (Python ≥ 3.10). Same engine and catalogs as the [Dorksmith](https://github.com/aelena/dorksmith) web app and the npm package, conformance-tested against the same golden fixtures.
54
+
55
+ ```bash
56
+ pip install dorksmith
57
+ ```
58
+
59
+ ```python
60
+ from dorksmith import generate, expand_handle, validate_query, search_url
61
+
62
+ result = generate("example.com", "domain", "public-documents",
63
+ options={"fileTypes": ["pdf", "docx"], "excludeTerms": ["jobs"], "maxVariants": 3})
64
+ for v in result.variants:
65
+ print(f"{v.label:<14} {v.query}")
66
+ # Balanced site:example.com (filetype:pdf OR filetype:docx) -jobs
67
+ # Precise site:example.com filetype:pdf -jobs
68
+ # Title-focused site:example.com (filetype:pdf OR filetype:docx) (intitle:report OR intitle:policy OR intitle:presentation OR intitle:"annual report") -jobs
69
+
70
+ search_url(result.variants[0].query)
71
+ # 'https://www.google.com/search?q=site%3Aexample.com%20%28filetype%3Apdf%20OR%20filetype%3Adocx%29%20-jobs'
72
+
73
+ handle = expand_handle("@alice42", categories=["developer"], max_platforms=3)
74
+ handle.profiles[0].url, handle.profiles[0].status
75
+ # ('https://github.com/alice42', 'not-checked')
76
+
77
+ validate_query("cache:example.com site:example.com or filetype:.pdf").warnings
78
+ # ["Lower-case 'or' is treated as an ordinary word; ...", "filetype: values take no leading dot ...", "cache: is deprecated: ..."]
79
+ ```
80
+
81
+ Results are dataclasses; `.to_dict()` gives the same camelCase shape as the HTTP API (`POST /api/v1/dorks/generate`), minus `requestId` and `rateLimit`.
82
+
83
+ ## API
84
+
85
+ | Function | Purpose |
86
+ |---|---|
87
+ | `generate(input, input_type, intent, *, engine="google", options=None, catalogs=None, limits=DEFAULT_LIMITS)` | Ranked variants with explanation, operators, warnings and rank reason. Raises `InputValidationError` (`.field`, `.unprocessable`). |
88
+ | `validate_query(query, engine="google", catalogs=None)` | Operators used with support state and warnings. Never rewrites the query. |
89
+ | `expand_handle(username, *, categories=None, platform_ids=None, max_platforms=None, catalogs=None)` | Profile URLs (`status` always `not-checked`) and site-scoped queries. |
90
+ | `infer_input_type(text, known_extensions=None)` | Deterministic input-type suggestion. |
91
+ | `search_url(query, engine="google")` | Search-engine URL, percent-encoded locally. |
92
+ | `validate_catalogs(catalogs)` | Structural/cross-reference validation, same rules as the API's readiness check. |
93
+ | `bundled_catalogs()`, `load_catalogs(path)`, `catalog_version()` | Embedded or custom catalogs. |
94
+ | lower-level: `normalize_text`, `try_normalize_domain`, `quote`, `safe_term`, `analyze_query`, `expand_template`, … | Building blocks for custom pipelines. |
95
+
96
+ `options` keys match the HTTP API: `fileTypes`, `excludeTerms`, `after`, `before`, `site`, `maxVariants`, `organization`, `location`, `role`, `displayName`.
97
+
98
+ ### Bring your own catalogs
99
+
100
+ ```python
101
+ from dorksmith import load_catalogs, validate_catalogs, generate
102
+ catalogs = load_catalogs("path/to/data") # operators.google.json, intents.json, platforms.json, filetypes.json
103
+ assert validate_catalogs(catalogs) == []
104
+ generate("x", "keyword", "general-discovery", catalogs=catalogs)
105
+ ```
106
+
107
+ ## CLI
108
+
109
+ ```bash
110
+ dorksmith generate example.com --intent exposed-config-files --max 3
111
+ dorksmith generate "Alice Smith" --intent person-organization --organization "Example Corp"
112
+ dorksmith handle alice42 --categories developer --json
113
+ dorksmith validate 'cache:example.com site:example.com'
114
+ dorksmith intents
115
+ ```
116
+
117
+ Exit codes: `0` ok, `2` invalid input (or a query with syntax errors for `validate`), `3` valid request that cannot be generated.
118
+
119
+ ## Development
120
+
121
+ ```bash
122
+ python -m pip install -e ".[test]"
123
+ python scripts/sync_catalogs.py # copies ../../data/*.json into src/dorksmith/data
124
+ python -m pytest -q # unit tests + the 151 golden fixtures from the monorepo
125
+ python -m build # sdist + wheel
126
+ ```
127
+
128
+ Publishing is done by the `py-package` GitHub workflow when a `py-v*` tag is pushed, using PyPI trusted publishing (see the repository README).
@@ -0,0 +1,78 @@
1
+ # dorksmith (PyPI)
2
+
3
+ Deterministic search-dork generator and OSINT/SOCMINT query engine as a pure-Python package with no dependencies (Python ≥ 3.10). Same engine and catalogs as the [Dorksmith](https://github.com/aelena/dorksmith) web app and the npm package, conformance-tested against the same golden fixtures.
4
+
5
+ ```bash
6
+ pip install dorksmith
7
+ ```
8
+
9
+ ```python
10
+ from dorksmith import generate, expand_handle, validate_query, search_url
11
+
12
+ result = generate("example.com", "domain", "public-documents",
13
+ options={"fileTypes": ["pdf", "docx"], "excludeTerms": ["jobs"], "maxVariants": 3})
14
+ for v in result.variants:
15
+ print(f"{v.label:<14} {v.query}")
16
+ # Balanced site:example.com (filetype:pdf OR filetype:docx) -jobs
17
+ # Precise site:example.com filetype:pdf -jobs
18
+ # Title-focused site:example.com (filetype:pdf OR filetype:docx) (intitle:report OR intitle:policy OR intitle:presentation OR intitle:"annual report") -jobs
19
+
20
+ search_url(result.variants[0].query)
21
+ # 'https://www.google.com/search?q=site%3Aexample.com%20%28filetype%3Apdf%20OR%20filetype%3Adocx%29%20-jobs'
22
+
23
+ handle = expand_handle("@alice42", categories=["developer"], max_platforms=3)
24
+ handle.profiles[0].url, handle.profiles[0].status
25
+ # ('https://github.com/alice42', 'not-checked')
26
+
27
+ validate_query("cache:example.com site:example.com or filetype:.pdf").warnings
28
+ # ["Lower-case 'or' is treated as an ordinary word; ...", "filetype: values take no leading dot ...", "cache: is deprecated: ..."]
29
+ ```
30
+
31
+ Results are dataclasses; `.to_dict()` gives the same camelCase shape as the HTTP API (`POST /api/v1/dorks/generate`), minus `requestId` and `rateLimit`.
32
+
33
+ ## API
34
+
35
+ | Function | Purpose |
36
+ |---|---|
37
+ | `generate(input, input_type, intent, *, engine="google", options=None, catalogs=None, limits=DEFAULT_LIMITS)` | Ranked variants with explanation, operators, warnings and rank reason. Raises `InputValidationError` (`.field`, `.unprocessable`). |
38
+ | `validate_query(query, engine="google", catalogs=None)` | Operators used with support state and warnings. Never rewrites the query. |
39
+ | `expand_handle(username, *, categories=None, platform_ids=None, max_platforms=None, catalogs=None)` | Profile URLs (`status` always `not-checked`) and site-scoped queries. |
40
+ | `infer_input_type(text, known_extensions=None)` | Deterministic input-type suggestion. |
41
+ | `search_url(query, engine="google")` | Search-engine URL, percent-encoded locally. |
42
+ | `validate_catalogs(catalogs)` | Structural/cross-reference validation, same rules as the API's readiness check. |
43
+ | `bundled_catalogs()`, `load_catalogs(path)`, `catalog_version()` | Embedded or custom catalogs. |
44
+ | lower-level: `normalize_text`, `try_normalize_domain`, `quote`, `safe_term`, `analyze_query`, `expand_template`, … | Building blocks for custom pipelines. |
45
+
46
+ `options` keys match the HTTP API: `fileTypes`, `excludeTerms`, `after`, `before`, `site`, `maxVariants`, `organization`, `location`, `role`, `displayName`.
47
+
48
+ ### Bring your own catalogs
49
+
50
+ ```python
51
+ from dorksmith import load_catalogs, validate_catalogs, generate
52
+ catalogs = load_catalogs("path/to/data") # operators.google.json, intents.json, platforms.json, filetypes.json
53
+ assert validate_catalogs(catalogs) == []
54
+ generate("x", "keyword", "general-discovery", catalogs=catalogs)
55
+ ```
56
+
57
+ ## CLI
58
+
59
+ ```bash
60
+ dorksmith generate example.com --intent exposed-config-files --max 3
61
+ dorksmith generate "Alice Smith" --intent person-organization --organization "Example Corp"
62
+ dorksmith handle alice42 --categories developer --json
63
+ dorksmith validate 'cache:example.com site:example.com'
64
+ dorksmith intents
65
+ ```
66
+
67
+ Exit codes: `0` ok, `2` invalid input (or a query with syntax errors for `validate`), `3` valid request that cannot be generated.
68
+
69
+ ## Development
70
+
71
+ ```bash
72
+ python -m pip install -e ".[test]"
73
+ python scripts/sync_catalogs.py # copies ../../data/*.json into src/dorksmith/data
74
+ python -m pytest -q # unit tests + the 151 golden fixtures from the monorepo
75
+ python -m build # sdist + wheel
76
+ ```
77
+
78
+ Publishing is done by the `py-package` GitHub workflow when a `py-v*` tag is pushed, using PyPI trusted publishing (see the repository README).
@@ -0,0 +1,50 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.25"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "dorksmith"
7
+ version = "0.1.0"
8
+ description = "Deterministic search-dork generator and OSINT/SOCMINT query engine. Catalog-driven, pure Python, no dependencies."
9
+ readme = "README.md"
10
+ license = { file = "LICENSE" }
11
+ authors = [{ name = "aelena" }]
12
+ requires-python = ">=3.10"
13
+ keywords = ["osint", "socmint", "google-dorks", "search-operators", "dork", "security", "deterministic"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Information Technology",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Operating System :: OS Independent",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3 :: Only",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Programming Language :: Python :: 3.13",
25
+ "Topic :: Security",
26
+ "Topic :: Internet :: WWW/HTTP :: Indexing/Search",
27
+ "Typing :: Typed",
28
+ ]
29
+ dependencies = []
30
+
31
+ [project.optional-dependencies]
32
+ test = ["pytest>=8"]
33
+
34
+ [project.urls]
35
+ Homepage = "https://github.com/aelena/dorksmith"
36
+ Repository = "https://github.com/aelena/dorksmith"
37
+ Issues = "https://github.com/aelena/dorksmith/issues"
38
+ Changelog = "https://github.com/aelena/dorksmith/commits/main/packages/dorksmith-py"
39
+
40
+ [project.scripts]
41
+ dorksmith = "dorksmith.cli:main"
42
+
43
+ [tool.hatch.build.targets.wheel]
44
+ packages = ["src/dorksmith"]
45
+
46
+ [tool.hatch.build.targets.sdist]
47
+ include = ["src/dorksmith", "tests", "scripts", "README.md", "LICENSE", "pyproject.toml"]
48
+
49
+ [tool.pytest.ini_options]
50
+ testpaths = ["tests"]
@@ -0,0 +1,40 @@
1
+ #!/usr/bin/env python3
2
+ """Copy the repository catalogs (../../data/*.json) into src/dorksmith/data.
3
+
4
+ `--check` exits non-zero when the embedded copies differ (used in CI).
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ HERE = Path(__file__).resolve().parent
12
+ PKG = HERE.parent
13
+ SOURCE = PKG.parent.parent / "data"
14
+ TARGET = PKG / "src" / "dorksmith" / "data"
15
+
16
+
17
+ def main() -> int:
18
+ check = "--check" in sys.argv
19
+ if not SOURCE.is_dir():
20
+ print(f"no repository data directory at {SOURCE}; keeping embedded catalogs")
21
+ return 0
22
+ TARGET.mkdir(parents=True, exist_ok=True)
23
+ changed = False
24
+ files = sorted(SOURCE.glob("*.json"))
25
+ for src in files:
26
+ dst = TARGET / src.name
27
+ data = src.read_bytes()
28
+ if not dst.exists() or dst.read_bytes() != data:
29
+ changed = True
30
+ if not check:
31
+ dst.write_bytes(data)
32
+ if check and changed:
33
+ print("embedded catalogs are out of date; run `python scripts/sync_catalogs.py`", file=sys.stderr)
34
+ return 1
35
+ print("embedded catalogs are up to date" if check else f"synced {len(files)} catalog(s)")
36
+ return 0
37
+
38
+
39
+ if __name__ == "__main__":
40
+ raise SystemExit(main())
@@ -0,0 +1,56 @@
1
+ """dorksmith — deterministic search-dork generator and OSINT/SOCMINT query engine.
2
+
3
+ from dorksmith import generate, expand_handle, validate_query
4
+ result = generate("example.com", "domain", "public-documents", options={"fileTypes": ["pdf"]})
5
+ for v in result.variants:
6
+ print(v.label, v.query)
7
+
8
+ The engine is pure: same request + same catalogs = same output. Pass ``catalogs=load_catalogs(path)``
9
+ to run with edited catalogs; the bundled ones are used otherwise.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from urllib.parse import quote as _url_quote
14
+
15
+ from .analyzer import GOOGLE_WORD_LIMIT, QueryAnalysis, analyze_query
16
+ from .catalogs import Catalogs, bundled_catalog_files, bundled_catalogs, load_catalogs
17
+ from .errors import InputValidationError
18
+ from .generator import (
19
+ DEFAULT_LIMITS, INPUT_TYPES, GenerateResult, Limits, OperatorUse, ValidateQueryResult, Variant, build_context,
20
+ expand_template, generate, validate_query,
21
+ )
22
+ from .handles import HANDLE_NOTICE, NOT_CHECKED, HandleExpandResult, HandleQuery, ProfileCandidate, escape_data_string, expand_handle
23
+ from .infer import infer_input_type
24
+ from .normalizer import (
25
+ canonicalize, domain_label, normalize_text, try_normalize_domain, try_normalize_email, try_normalize_url,
26
+ try_normalize_username, try_parse_filename, try_parse_iso_date,
27
+ )
28
+ from .placeholders import ALL_PLACEHOLDERS, OPTIONAL_BY_DEFAULT, placeholders_in
29
+ from .quoting import exclusion, looks_like_syntax, operator_value, or_group, quote, safe_term, safe_terms, unquote
30
+ from .validator import validate_catalogs
31
+
32
+ __version__ = "0.1.0"
33
+
34
+
35
+ def catalog_version(catalogs: Catalogs | None = None) -> str:
36
+ """Version string of the catalogs in use (from intents.json)."""
37
+ return (catalogs or bundled_catalogs()).catalog_version
38
+
39
+
40
+ def search_url(query: str, engine: str = "google", catalogs: Catalogs | None = None) -> str:
41
+ """Search-engine URL for a query, built locally with percent-encoding."""
42
+ c = catalogs or bundled_catalogs()
43
+ template = c.operators.get(engine, {}).get("searchUrlTemplate", "https://www.google.com/search?q={query}")
44
+ return template.replace("{query}", _url_quote(query, safe="-_.!~*'()"))
45
+
46
+
47
+ __all__ = [
48
+ "__version__", "ALL_PLACEHOLDERS", "Catalogs", "DEFAULT_LIMITS", "GOOGLE_WORD_LIMIT", "GenerateResult", "HANDLE_NOTICE",
49
+ "HandleExpandResult", "HandleQuery", "INPUT_TYPES", "InputValidationError", "Limits", "NOT_CHECKED", "OPTIONAL_BY_DEFAULT",
50
+ "OperatorUse", "ProfileCandidate", "QueryAnalysis", "ValidateQueryResult", "Variant", "analyze_query", "build_context",
51
+ "bundled_catalog_files", "bundled_catalogs", "canonicalize", "catalog_version", "domain_label", "escape_data_string",
52
+ "exclusion", "expand_handle", "expand_template", "generate", "infer_input_type", "load_catalogs", "looks_like_syntax",
53
+ "normalize_text", "operator_value", "or_group", "placeholders_in", "quote", "safe_term", "safe_terms", "search_url",
54
+ "try_normalize_domain", "try_normalize_email", "try_normalize_url", "try_normalize_username", "try_parse_filename",
55
+ "try_parse_iso_date", "unquote", "validate_catalogs", "validate_query",
56
+ ]
@@ -0,0 +1,151 @@
1
+ """Quote-aware operator scanner with plain-language warnings."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ from dataclasses import dataclass
6
+
7
+ from .catalogs import Json
8
+ from .normalizer import is_letter
9
+
10
+ GOOGLE_WORD_LIMIT = 32
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class QueryAnalysis:
15
+ operators: list[str]
16
+ warnings: list[str]
17
+ has_errors: bool
18
+
19
+
20
+ def _tokenize(query: str) -> list[tuple[str, bool]]:
21
+ tokens: list[tuple[str, bool]] = []
22
+ i, n = 0, len(query)
23
+ while i < n:
24
+ if query[i].isspace():
25
+ i += 1
26
+ continue
27
+ if query[i] == '"':
28
+ end = query.find('"', i + 1)
29
+ if end < 0:
30
+ end = n - 1
31
+ tokens.append((query[i + 1: max(i + 1, end)], True))
32
+ i = end + 1
33
+ continue
34
+ start = i
35
+ while i < n and not query[i].isspace():
36
+ if query[i] == '"':
37
+ end = query.find('"', i + 1)
38
+ i = n if end < 0 else end + 1
39
+ continue
40
+ i += 1
41
+ tokens.append((query[start:i], False))
42
+ return tokens
43
+
44
+
45
+ def analyze_query(query: str, catalog: Json, by_token: dict[str, Json]) -> QueryAnalysis:
46
+ operators: list[str] = []
47
+ warnings: list[str] = []
48
+ has_errors = False
49
+
50
+ def use(token: str) -> None:
51
+ if token not in operators:
52
+ operators.append(token)
53
+
54
+ depth = 0
55
+ tokens = _tokenize(query)
56
+ quote_count = query.count('"')
57
+ if quote_count > 0:
58
+ use('"')
59
+ if quote_count % 2 == 1:
60
+ warnings.append("Unbalanced double quotes: the last phrase will not be treated as exact.")
61
+ has_errors = True
62
+
63
+ for text, quoted in tokens:
64
+ if quoted:
65
+ if "*" in text:
66
+ use("*")
67
+ continue
68
+ t = text
69
+ if not t:
70
+ continue
71
+ if t == "OR":
72
+ use("OR"); continue
73
+ if t == "AND":
74
+ use("AND"); continue
75
+ if t == "|":
76
+ use("|"); continue
77
+ if t in ("or", "and"):
78
+ warnings.append(f"Lower-case '{t}' is treated as an ordinary word; use upper-case {t.upper()} for a boolean operator.")
79
+ continue
80
+
81
+ body = t
82
+ while body and body[0] == "(":
83
+ depth += 1; use("("); body = body[1:]
84
+ closing = 0
85
+ while body and body[-1] == ")":
86
+ closing += 1; body = body[:-1]
87
+ depth -= closing
88
+
89
+ if body.startswith("AROUND("):
90
+ use("AROUND("); continue
91
+ if len(body) > 1 and body[0] == "-":
92
+ use("-"); body = body[1:]
93
+ elif len(body) > 1 and body[0] == "+":
94
+ use("+"); body = body[1:]
95
+ elif len(body) > 1 and body[0] == "~":
96
+ use("~"); body = body[1:]
97
+ if len(body) > 1 and body[0] == "#":
98
+ use("#"); continue
99
+ if len(body) > 1 and body[0] == "@":
100
+ use("@"); continue
101
+ if body == "*":
102
+ use("*"); continue
103
+ if ".." in body and any(ch.isdigit() for ch in body):
104
+ use(".."); continue
105
+ if len(body) > 1 and body[0] == "$" and all(ch.isdigit() or ch in ".," for ch in body[1:]):
106
+ use("$"); continue
107
+
108
+ colon = body.find(":")
109
+ if colon > 0 and all(is_letter(ch) for ch in body[:colon]):
110
+ prefix = body[: colon + 1].lower()
111
+ value = body[colon + 1:]
112
+ op = by_token.get(prefix)
113
+ if op:
114
+ use(prefix)
115
+ if prefix == "filetype:" and value.startswith("."):
116
+ warnings.append("filetype: values take no leading dot (filetype:pdf).")
117
+ if prefix == "site:" and "://" in value:
118
+ warnings.append("site: values take no scheme (site:example.com).")
119
+ if prefix in ("before:", "after:") and len(value) not in (4, 10):
120
+ warnings.append(f"{prefix} expects YYYY-MM-DD or YYYY.")
121
+ if op.get("takesValue") and not value:
122
+ warnings.append(f"{prefix} has no value.")
123
+ has_errors = True
124
+ elif len(body[:colon]) <= 15 and "://" not in body:
125
+ warnings.append(f"'{prefix}' is not a known {catalog.get('engineName', '')} operator and will be searched as plain text.")
126
+
127
+ if depth != 0:
128
+ warnings.append("Unbalanced parentheses.")
129
+ has_errors = True
130
+
131
+ for token in operators:
132
+ op = by_token.get(token)
133
+ if not op:
134
+ continue
135
+ support = op.get("support")
136
+ caveats = op.get("caveats") or []
137
+ if support == "deprecated":
138
+ warnings.append(f"{op['token']} is deprecated: {caveats[0] if caveats else 'it no longer works.'}")
139
+ elif support == "unreliable":
140
+ warnings.append(f"{op['token']} is unreliable: {caveats[0] if caveats else 'results are inconsistent.'}")
141
+ elif support == "unknown":
142
+ warnings.append(f"{op['token']} has unknown support status.")
143
+
144
+ word_count = sum(1 for text, _ in tokens if text)
145
+ if word_count > GOOGLE_WORD_LIMIT:
146
+ warnings.append(f"Query has {word_count} terms; Google ignores everything after the first {GOOGLE_WORD_LIMIT}.")
147
+
148
+ return QueryAnalysis(operators, warnings, has_errors)
149
+
150
+
151
+ _ = re # keep re available for callers that extend the analyzer
@@ -0,0 +1,73 @@
1
+ """Catalog loading. Catalogs are plain dicts mirroring the JSON files in the repository's data/ directory."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ from dataclasses import dataclass, field
6
+ from importlib import resources
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ Json = dict[str, Any]
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class Catalogs:
15
+ """The four catalogs the engine consumes. ``operators`` is keyed by engine id."""
16
+
17
+ operators: dict[str, Json]
18
+ intents: Json
19
+ platforms: Json
20
+ file_types: Json
21
+ _operator_index: dict[str, dict[str, Json]] = field(default_factory=dict, repr=False, compare=False)
22
+
23
+ @property
24
+ def catalog_version(self) -> str:
25
+ return str(self.intents.get("catalogVersion", ""))
26
+
27
+ def operator_index(self, engine: str) -> dict[str, Json]:
28
+ """token -> operator definition for an engine (cached)."""
29
+ if engine not in self._operator_index:
30
+ index: dict[str, Json] = {}
31
+ for op in self.operators[engine].get("operators", []):
32
+ index.setdefault(op["token"], op)
33
+ self._operator_index[engine] = index
34
+ return self._operator_index[engine]
35
+
36
+ def intent(self, intent_id: str) -> Json | None:
37
+ for i in self.intents.get("intents", []):
38
+ if i.get("id") == intent_id:
39
+ return i
40
+ return None
41
+
42
+
43
+ def _assemble(files: dict[str, Json]) -> Catalogs:
44
+ operators = {v["engine"]: v for k, v in files.items() if k.startswith("operators.")}
45
+ return Catalogs(operators=operators, intents=files["intents.json"], platforms=files["platforms.json"], file_types=files["filetypes.json"])
46
+
47
+
48
+ def load_catalogs(directory: str | Path) -> Catalogs:
49
+ """Load catalogs from a directory containing operators.<engine>.json, intents.json, platforms.json, filetypes.json."""
50
+ d = Path(directory)
51
+ files = {p.name: json.loads(p.read_text(encoding="utf-8")) for p in sorted(d.glob("*.json"))}
52
+ return _assemble(files)
53
+
54
+
55
+ def bundled_catalog_files() -> dict[str, Json]:
56
+ """The raw catalog files embedded in the package, keyed by file name."""
57
+ root = resources.files("dorksmith") / "data"
58
+ out: dict[str, Json] = {}
59
+ for entry in sorted(root.iterdir(), key=lambda e: e.name):
60
+ if entry.name.endswith(".json"):
61
+ out[entry.name] = json.loads(entry.read_text(encoding="utf-8"))
62
+ return out
63
+
64
+
65
+ _bundled: Catalogs | None = None
66
+
67
+
68
+ def bundled_catalogs() -> Catalogs:
69
+ """The catalogs embedded in this package (loaded once)."""
70
+ global _bundled
71
+ if _bundled is None:
72
+ _bundled = _assemble(bundled_catalog_files())
73
+ return _bundled