dorksmith 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dorksmith/__init__.py +56 -0
- dorksmith/analyzer.py +151 -0
- dorksmith/catalogs.py +73 -0
- dorksmith/cli.py +124 -0
- dorksmith/data/filetypes.json +68 -0
- dorksmith/data/intents.json +712 -0
- dorksmith/data/operators.google.json +590 -0
- dorksmith/data/platforms.json +118 -0
- dorksmith/errors.py +15 -0
- dorksmith/generator.py +334 -0
- dorksmith/handles.py +135 -0
- dorksmith/infer.py +41 -0
- dorksmith/normalizer.py +213 -0
- dorksmith/placeholders.py +34 -0
- dorksmith/py.typed +0 -0
- dorksmith/quoting.py +82 -0
- dorksmith/ranker.py +131 -0
- dorksmith/resolver.py +222 -0
- dorksmith/validator.py +184 -0
- dorksmith-0.1.0.dist-info/METADATA +128 -0
- dorksmith-0.1.0.dist-info/RECORD +24 -0
- dorksmith-0.1.0.dist-info/WHEEL +4 -0
- dorksmith-0.1.0.dist-info/entry_points.txt +2 -0
- dorksmith-0.1.0.dist-info/licenses/LICENSE +21 -0
dorksmith/__init__.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""dorksmith — deterministic search-dork generator and OSINT/SOCMINT query engine.
|
|
2
|
+
|
|
3
|
+
from dorksmith import generate, expand_handle, validate_query
|
|
4
|
+
result = generate("example.com", "domain", "public-documents", options={"fileTypes": ["pdf"]})
|
|
5
|
+
for v in result.variants:
|
|
6
|
+
print(v.label, v.query)
|
|
7
|
+
|
|
8
|
+
The engine is pure: same request + same catalogs = same output. Pass ``catalogs=load_catalogs(path)``
|
|
9
|
+
to run with edited catalogs; the bundled ones are used otherwise.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from urllib.parse import quote as _url_quote
|
|
14
|
+
|
|
15
|
+
from .analyzer import GOOGLE_WORD_LIMIT, QueryAnalysis, analyze_query
|
|
16
|
+
from .catalogs import Catalogs, bundled_catalog_files, bundled_catalogs, load_catalogs
|
|
17
|
+
from .errors import InputValidationError
|
|
18
|
+
from .generator import (
|
|
19
|
+
DEFAULT_LIMITS, INPUT_TYPES, GenerateResult, Limits, OperatorUse, ValidateQueryResult, Variant, build_context,
|
|
20
|
+
expand_template, generate, validate_query,
|
|
21
|
+
)
|
|
22
|
+
from .handles import HANDLE_NOTICE, NOT_CHECKED, HandleExpandResult, HandleQuery, ProfileCandidate, escape_data_string, expand_handle
|
|
23
|
+
from .infer import infer_input_type
|
|
24
|
+
from .normalizer import (
|
|
25
|
+
canonicalize, domain_label, normalize_text, try_normalize_domain, try_normalize_email, try_normalize_url,
|
|
26
|
+
try_normalize_username, try_parse_filename, try_parse_iso_date,
|
|
27
|
+
)
|
|
28
|
+
from .placeholders import ALL_PLACEHOLDERS, OPTIONAL_BY_DEFAULT, placeholders_in
|
|
29
|
+
from .quoting import exclusion, looks_like_syntax, operator_value, or_group, quote, safe_term, safe_terms, unquote
|
|
30
|
+
from .validator import validate_catalogs
|
|
31
|
+
|
|
32
|
+
__version__ = "0.1.0"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def catalog_version(catalogs: Catalogs | None = None) -> str:
|
|
36
|
+
"""Version string of the catalogs in use (from intents.json)."""
|
|
37
|
+
return (catalogs or bundled_catalogs()).catalog_version
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def search_url(query: str, engine: str = "google", catalogs: Catalogs | None = None) -> str:
|
|
41
|
+
"""Search-engine URL for a query, built locally with percent-encoding."""
|
|
42
|
+
c = catalogs or bundled_catalogs()
|
|
43
|
+
template = c.operators.get(engine, {}).get("searchUrlTemplate", "https://www.google.com/search?q={query}")
|
|
44
|
+
return template.replace("{query}", _url_quote(query, safe="-_.!~*'()"))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
__all__ = [
|
|
48
|
+
"__version__", "ALL_PLACEHOLDERS", "Catalogs", "DEFAULT_LIMITS", "GOOGLE_WORD_LIMIT", "GenerateResult", "HANDLE_NOTICE",
|
|
49
|
+
"HandleExpandResult", "HandleQuery", "INPUT_TYPES", "InputValidationError", "Limits", "NOT_CHECKED", "OPTIONAL_BY_DEFAULT",
|
|
50
|
+
"OperatorUse", "ProfileCandidate", "QueryAnalysis", "ValidateQueryResult", "Variant", "analyze_query", "build_context",
|
|
51
|
+
"bundled_catalog_files", "bundled_catalogs", "canonicalize", "catalog_version", "domain_label", "escape_data_string",
|
|
52
|
+
"exclusion", "expand_handle", "expand_template", "generate", "infer_input_type", "load_catalogs", "looks_like_syntax",
|
|
53
|
+
"normalize_text", "operator_value", "or_group", "placeholders_in", "quote", "safe_term", "safe_terms", "search_url",
|
|
54
|
+
"try_normalize_domain", "try_normalize_email", "try_normalize_url", "try_normalize_username", "try_parse_filename",
|
|
55
|
+
"try_parse_iso_date", "unquote", "validate_catalogs", "validate_query",
|
|
56
|
+
]
|
dorksmith/analyzer.py
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""Quote-aware operator scanner with plain-language warnings."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from .catalogs import Json
|
|
8
|
+
from .normalizer import is_letter
|
|
9
|
+
|
|
10
|
+
GOOGLE_WORD_LIMIT = 32
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class QueryAnalysis:
|
|
15
|
+
operators: list[str]
|
|
16
|
+
warnings: list[str]
|
|
17
|
+
has_errors: bool
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _tokenize(query: str) -> list[tuple[str, bool]]:
|
|
21
|
+
tokens: list[tuple[str, bool]] = []
|
|
22
|
+
i, n = 0, len(query)
|
|
23
|
+
while i < n:
|
|
24
|
+
if query[i].isspace():
|
|
25
|
+
i += 1
|
|
26
|
+
continue
|
|
27
|
+
if query[i] == '"':
|
|
28
|
+
end = query.find('"', i + 1)
|
|
29
|
+
if end < 0:
|
|
30
|
+
end = n - 1
|
|
31
|
+
tokens.append((query[i + 1: max(i + 1, end)], True))
|
|
32
|
+
i = end + 1
|
|
33
|
+
continue
|
|
34
|
+
start = i
|
|
35
|
+
while i < n and not query[i].isspace():
|
|
36
|
+
if query[i] == '"':
|
|
37
|
+
end = query.find('"', i + 1)
|
|
38
|
+
i = n if end < 0 else end + 1
|
|
39
|
+
continue
|
|
40
|
+
i += 1
|
|
41
|
+
tokens.append((query[start:i], False))
|
|
42
|
+
return tokens
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def analyze_query(query: str, catalog: Json, by_token: dict[str, Json]) -> QueryAnalysis:
|
|
46
|
+
operators: list[str] = []
|
|
47
|
+
warnings: list[str] = []
|
|
48
|
+
has_errors = False
|
|
49
|
+
|
|
50
|
+
def use(token: str) -> None:
|
|
51
|
+
if token not in operators:
|
|
52
|
+
operators.append(token)
|
|
53
|
+
|
|
54
|
+
depth = 0
|
|
55
|
+
tokens = _tokenize(query)
|
|
56
|
+
quote_count = query.count('"')
|
|
57
|
+
if quote_count > 0:
|
|
58
|
+
use('"')
|
|
59
|
+
if quote_count % 2 == 1:
|
|
60
|
+
warnings.append("Unbalanced double quotes: the last phrase will not be treated as exact.")
|
|
61
|
+
has_errors = True
|
|
62
|
+
|
|
63
|
+
for text, quoted in tokens:
|
|
64
|
+
if quoted:
|
|
65
|
+
if "*" in text:
|
|
66
|
+
use("*")
|
|
67
|
+
continue
|
|
68
|
+
t = text
|
|
69
|
+
if not t:
|
|
70
|
+
continue
|
|
71
|
+
if t == "OR":
|
|
72
|
+
use("OR"); continue
|
|
73
|
+
if t == "AND":
|
|
74
|
+
use("AND"); continue
|
|
75
|
+
if t == "|":
|
|
76
|
+
use("|"); continue
|
|
77
|
+
if t in ("or", "and"):
|
|
78
|
+
warnings.append(f"Lower-case '{t}' is treated as an ordinary word; use upper-case {t.upper()} for a boolean operator.")
|
|
79
|
+
continue
|
|
80
|
+
|
|
81
|
+
body = t
|
|
82
|
+
while body and body[0] == "(":
|
|
83
|
+
depth += 1; use("("); body = body[1:]
|
|
84
|
+
closing = 0
|
|
85
|
+
while body and body[-1] == ")":
|
|
86
|
+
closing += 1; body = body[:-1]
|
|
87
|
+
depth -= closing
|
|
88
|
+
|
|
89
|
+
if body.startswith("AROUND("):
|
|
90
|
+
use("AROUND("); continue
|
|
91
|
+
if len(body) > 1 and body[0] == "-":
|
|
92
|
+
use("-"); body = body[1:]
|
|
93
|
+
elif len(body) > 1 and body[0] == "+":
|
|
94
|
+
use("+"); body = body[1:]
|
|
95
|
+
elif len(body) > 1 and body[0] == "~":
|
|
96
|
+
use("~"); body = body[1:]
|
|
97
|
+
if len(body) > 1 and body[0] == "#":
|
|
98
|
+
use("#"); continue
|
|
99
|
+
if len(body) > 1 and body[0] == "@":
|
|
100
|
+
use("@"); continue
|
|
101
|
+
if body == "*":
|
|
102
|
+
use("*"); continue
|
|
103
|
+
if ".." in body and any(ch.isdigit() for ch in body):
|
|
104
|
+
use(".."); continue
|
|
105
|
+
if len(body) > 1 and body[0] == "$" and all(ch.isdigit() or ch in ".," for ch in body[1:]):
|
|
106
|
+
use("$"); continue
|
|
107
|
+
|
|
108
|
+
colon = body.find(":")
|
|
109
|
+
if colon > 0 and all(is_letter(ch) for ch in body[:colon]):
|
|
110
|
+
prefix = body[: colon + 1].lower()
|
|
111
|
+
value = body[colon + 1:]
|
|
112
|
+
op = by_token.get(prefix)
|
|
113
|
+
if op:
|
|
114
|
+
use(prefix)
|
|
115
|
+
if prefix == "filetype:" and value.startswith("."):
|
|
116
|
+
warnings.append("filetype: values take no leading dot (filetype:pdf).")
|
|
117
|
+
if prefix == "site:" and "://" in value:
|
|
118
|
+
warnings.append("site: values take no scheme (site:example.com).")
|
|
119
|
+
if prefix in ("before:", "after:") and len(value) not in (4, 10):
|
|
120
|
+
warnings.append(f"{prefix} expects YYYY-MM-DD or YYYY.")
|
|
121
|
+
if op.get("takesValue") and not value:
|
|
122
|
+
warnings.append(f"{prefix} has no value.")
|
|
123
|
+
has_errors = True
|
|
124
|
+
elif len(body[:colon]) <= 15 and "://" not in body:
|
|
125
|
+
warnings.append(f"'{prefix}' is not a known {catalog.get('engineName', '')} operator and will be searched as plain text.")
|
|
126
|
+
|
|
127
|
+
if depth != 0:
|
|
128
|
+
warnings.append("Unbalanced parentheses.")
|
|
129
|
+
has_errors = True
|
|
130
|
+
|
|
131
|
+
for token in operators:
|
|
132
|
+
op = by_token.get(token)
|
|
133
|
+
if not op:
|
|
134
|
+
continue
|
|
135
|
+
support = op.get("support")
|
|
136
|
+
caveats = op.get("caveats") or []
|
|
137
|
+
if support == "deprecated":
|
|
138
|
+
warnings.append(f"{op['token']} is deprecated: {caveats[0] if caveats else 'it no longer works.'}")
|
|
139
|
+
elif support == "unreliable":
|
|
140
|
+
warnings.append(f"{op['token']} is unreliable: {caveats[0] if caveats else 'results are inconsistent.'}")
|
|
141
|
+
elif support == "unknown":
|
|
142
|
+
warnings.append(f"{op['token']} has unknown support status.")
|
|
143
|
+
|
|
144
|
+
word_count = sum(1 for text, _ in tokens if text)
|
|
145
|
+
if word_count > GOOGLE_WORD_LIMIT:
|
|
146
|
+
warnings.append(f"Query has {word_count} terms; Google ignores everything after the first {GOOGLE_WORD_LIMIT}.")
|
|
147
|
+
|
|
148
|
+
return QueryAnalysis(operators, warnings, has_errors)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
_ = re # keep re available for callers that extend the analyzer
|
dorksmith/catalogs.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Catalog loading. Catalogs are plain dicts mirroring the JSON files in the repository's data/ directory."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from importlib import resources
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
Json = dict[str, Any]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class Catalogs:
|
|
15
|
+
"""The four catalogs the engine consumes. ``operators`` is keyed by engine id."""
|
|
16
|
+
|
|
17
|
+
operators: dict[str, Json]
|
|
18
|
+
intents: Json
|
|
19
|
+
platforms: Json
|
|
20
|
+
file_types: Json
|
|
21
|
+
_operator_index: dict[str, dict[str, Json]] = field(default_factory=dict, repr=False, compare=False)
|
|
22
|
+
|
|
23
|
+
@property
|
|
24
|
+
def catalog_version(self) -> str:
|
|
25
|
+
return str(self.intents.get("catalogVersion", ""))
|
|
26
|
+
|
|
27
|
+
def operator_index(self, engine: str) -> dict[str, Json]:
|
|
28
|
+
"""token -> operator definition for an engine (cached)."""
|
|
29
|
+
if engine not in self._operator_index:
|
|
30
|
+
index: dict[str, Json] = {}
|
|
31
|
+
for op in self.operators[engine].get("operators", []):
|
|
32
|
+
index.setdefault(op["token"], op)
|
|
33
|
+
self._operator_index[engine] = index
|
|
34
|
+
return self._operator_index[engine]
|
|
35
|
+
|
|
36
|
+
def intent(self, intent_id: str) -> Json | None:
|
|
37
|
+
for i in self.intents.get("intents", []):
|
|
38
|
+
if i.get("id") == intent_id:
|
|
39
|
+
return i
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _assemble(files: dict[str, Json]) -> Catalogs:
|
|
44
|
+
operators = {v["engine"]: v for k, v in files.items() if k.startswith("operators.")}
|
|
45
|
+
return Catalogs(operators=operators, intents=files["intents.json"], platforms=files["platforms.json"], file_types=files["filetypes.json"])
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def load_catalogs(directory: str | Path) -> Catalogs:
|
|
49
|
+
"""Load catalogs from a directory containing operators.<engine>.json, intents.json, platforms.json, filetypes.json."""
|
|
50
|
+
d = Path(directory)
|
|
51
|
+
files = {p.name: json.loads(p.read_text(encoding="utf-8")) for p in sorted(d.glob("*.json"))}
|
|
52
|
+
return _assemble(files)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def bundled_catalog_files() -> dict[str, Json]:
|
|
56
|
+
"""The raw catalog files embedded in the package, keyed by file name."""
|
|
57
|
+
root = resources.files("dorksmith") / "data"
|
|
58
|
+
out: dict[str, Json] = {}
|
|
59
|
+
for entry in sorted(root.iterdir(), key=lambda e: e.name):
|
|
60
|
+
if entry.name.endswith(".json"):
|
|
61
|
+
out[entry.name] = json.loads(entry.read_text(encoding="utf-8"))
|
|
62
|
+
return out
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
_bundled: Catalogs | None = None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def bundled_catalogs() -> Catalogs:
|
|
69
|
+
"""The catalogs embedded in this package (loaded once)."""
|
|
70
|
+
global _bundled
|
|
71
|
+
if _bundled is None:
|
|
72
|
+
_bundled = _assemble(bundled_catalog_files())
|
|
73
|
+
return _bundled
|
dorksmith/cli.py
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Command line: dorksmith generate|handle|validate|intents|operators|platforms|filetypes."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
from . import __version__, bundled_catalogs, catalog_version, expand_handle, generate, infer_input_type, search_url, validate_query
|
|
9
|
+
from .errors import InputValidationError
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _csv(value: str | None) -> list[str]:
|
|
13
|
+
return [v.strip() for v in value.split(",") if v.strip()] if value else []
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _parser() -> argparse.ArgumentParser:
|
|
17
|
+
p = argparse.ArgumentParser(prog="dorksmith", description=f"Deterministic search-dork generator (catalog {catalog_version()}, package {__version__}).")
|
|
18
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
19
|
+
|
|
20
|
+
g = sub.add_parser("generate", help="generate ranked query variants")
|
|
21
|
+
g.add_argument("input", nargs="+", help="target text (domain, name, handle, email, phrase, ...)")
|
|
22
|
+
g.add_argument("--type", dest="input_type", help="input type; inferred when omitted")
|
|
23
|
+
g.add_argument("--intent", help="intent id; first compatible intent when omitted")
|
|
24
|
+
g.add_argument("--filetypes", help="comma-separated extensions")
|
|
25
|
+
g.add_argument("--exclude", help="comma-separated excluded terms")
|
|
26
|
+
g.add_argument("--after"); g.add_argument("--before"); g.add_argument("--site")
|
|
27
|
+
g.add_argument("--max", type=int, dest="max_variants")
|
|
28
|
+
g.add_argument("--organization"); g.add_argument("--location"); g.add_argument("--role"); g.add_argument("--display-name", dest="display_name")
|
|
29
|
+
g.add_argument("--urls", action="store_true", help="print a search URL under each variant")
|
|
30
|
+
g.add_argument("--json", action="store_true")
|
|
31
|
+
|
|
32
|
+
h = sub.add_parser("handle", help="expand a handle into profile URLs and queries")
|
|
33
|
+
h.add_argument("username")
|
|
34
|
+
h.add_argument("--categories"); h.add_argument("--platforms"); h.add_argument("--max", type=int, dest="max_platforms")
|
|
35
|
+
h.add_argument("--json", action="store_true")
|
|
36
|
+
|
|
37
|
+
v = sub.add_parser("validate", help="analyse a hand-written query")
|
|
38
|
+
v.add_argument("query", nargs="+")
|
|
39
|
+
v.add_argument("--json", action="store_true")
|
|
40
|
+
|
|
41
|
+
for name in ("intents", "operators", "platforms", "filetypes"):
|
|
42
|
+
sub.add_parser(name, help=f"list the {name} catalog").add_argument("--json", action="store_true")
|
|
43
|
+
return p
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def main(argv: list[str] | None = None) -> int:
|
|
47
|
+
args = _parser().parse_args(argv)
|
|
48
|
+
catalogs = bundled_catalogs()
|
|
49
|
+
try:
|
|
50
|
+
if args.command == "generate":
|
|
51
|
+
text = " ".join(args.input)
|
|
52
|
+
known = {e["ext"] for e in catalogs.file_types["extensions"]}
|
|
53
|
+
input_type = args.input_type or infer_input_type(text, known) or "keyword"
|
|
54
|
+
intent = args.intent or next(i["id"] for i in catalogs.intents["intents"] if input_type in i["compatibleInputTypes"])
|
|
55
|
+
r = generate(text, input_type, intent, options={
|
|
56
|
+
"fileTypes": _csv(args.filetypes), "excludeTerms": _csv(args.exclude), "after": args.after, "before": args.before, "site": args.site,
|
|
57
|
+
"maxVariants": args.max_variants, "organization": args.organization, "location": args.location, "role": args.role, "displayName": args.display_name,
|
|
58
|
+
})
|
|
59
|
+
if args.json:
|
|
60
|
+
print(json.dumps(r.to_dict(), indent=2, ensure_ascii=False))
|
|
61
|
+
return 0
|
|
62
|
+
print(f"# {r.input_type} \"{r.normalized_input}\" · intent {r.intent} · catalog {r.catalog_version}")
|
|
63
|
+
for var in r.variants:
|
|
64
|
+
print(f"\n[{var.label}] {var.id}\n {var.query}\n ↳ {var.explanation}")
|
|
65
|
+
for w in var.warnings:
|
|
66
|
+
print(f" ⚠ {w}")
|
|
67
|
+
if args.urls:
|
|
68
|
+
print(f" {search_url(var.query)}")
|
|
69
|
+
elif args.command == "handle":
|
|
70
|
+
r = expand_handle(args.username, categories=_csv(args.categories), platform_ids=_csv(args.platforms), max_platforms=args.max_platforms)
|
|
71
|
+
if args.json:
|
|
72
|
+
print(json.dumps(r.to_dict(), indent=2, ensure_ascii=False))
|
|
73
|
+
return 0
|
|
74
|
+
print(f"# {r.normalized_username} · {len(r.profiles)} platforms · {r.notice}")
|
|
75
|
+
for p in r.profiles:
|
|
76
|
+
print(f"{p.platform_name:<28} {p.url or '(no reliable URL)'}")
|
|
77
|
+
if p.caveat:
|
|
78
|
+
print(f"{'':<29}⚠ {p.caveat}")
|
|
79
|
+
print("\n# queries")
|
|
80
|
+
for q in r.queries:
|
|
81
|
+
print(f" {q.query}")
|
|
82
|
+
elif args.command == "validate":
|
|
83
|
+
r = validate_query(" ".join(args.query))
|
|
84
|
+
if args.json:
|
|
85
|
+
print(json.dumps(r.to_dict(), indent=2, ensure_ascii=False))
|
|
86
|
+
return 0
|
|
87
|
+
for o in r.operators:
|
|
88
|
+
print(f"{o.token:<12} {o.support:<11} {o.name}")
|
|
89
|
+
for w in r.warnings:
|
|
90
|
+
print(f"⚠ {w}")
|
|
91
|
+
return 2 if r.has_errors else 0
|
|
92
|
+
elif args.command == "intents":
|
|
93
|
+
rows = catalogs.intents["intents"]
|
|
94
|
+
if args.json:
|
|
95
|
+
print(json.dumps([{k: i[k] for k in ("id", "label", "group", "compatibleInputTypes", "safety")} for i in rows], indent=2))
|
|
96
|
+
else:
|
|
97
|
+
for i in rows:
|
|
98
|
+
print(f"{i['id']:<28} {','.join(i['compatibleInputTypes']):<52} {i['label']}")
|
|
99
|
+
elif args.command == "operators":
|
|
100
|
+
if args.json:
|
|
101
|
+
print(json.dumps(catalogs.operators, indent=2))
|
|
102
|
+
else:
|
|
103
|
+
for o in next(iter(catalogs.operators.values()))["operators"]:
|
|
104
|
+
print(f"{o['token']:<14} {o['support']:<11} {'generates' if o.get('generate') else ' '} {o['name']}")
|
|
105
|
+
elif args.command == "platforms":
|
|
106
|
+
if args.json:
|
|
107
|
+
print(json.dumps(catalogs.platforms, indent=2))
|
|
108
|
+
else:
|
|
109
|
+
for p in catalogs.platforms["platforms"]:
|
|
110
|
+
print(f"{p['id']:<18} {p['category']:<20} {'' if p.get('enabled', True) else '(disabled) '}{p['profileUrlTemplate']}")
|
|
111
|
+
elif args.command == "filetypes":
|
|
112
|
+
if args.json:
|
|
113
|
+
print(json.dumps(catalogs.file_types, indent=2))
|
|
114
|
+
else:
|
|
115
|
+
for e in catalogs.file_types["extensions"]:
|
|
116
|
+
print(f"{e['ext']:<12} {e['group']:<11} {e['label']}")
|
|
117
|
+
return 0
|
|
118
|
+
except InputValidationError as e:
|
|
119
|
+
print(f"error: {e.message}" + (f" ({e.field})" if e.field else ""), file=sys.stderr)
|
|
120
|
+
return 3 if e.unprocessable else 2
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
if __name__ == "__main__": # pragma: no cover
|
|
124
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"catalogVersion": "2026-09-24",
|
|
4
|
+
"groups": [
|
|
5
|
+
{ "id": "documents", "label": "Office documents", "extensions": ["pdf", "docx", "doc", "pptx", "ppt", "xlsx", "xls", "odt", "odp", "ods", "rtf", "txt", "csv"] },
|
|
6
|
+
{ "id": "data", "label": "Data & configuration", "extensions": ["json", "xml", "yaml", "yml", "csv", "sql", "ini", "cfg", "conf", "env", "toml", "properties"] },
|
|
7
|
+
{ "id": "logs", "label": "Logs & diagnostics", "extensions": ["log", "txt", "out", "err", "dmp", "trace"] },
|
|
8
|
+
{ "id": "archives", "label": "Archives & backups", "extensions": ["zip", "tar", "gz", "tgz", "7z", "rar", "bak", "old", "sql"] },
|
|
9
|
+
{ "id": "code", "label": "Source & build", "extensions": ["js", "map", "py", "php", "java", "cs", "sh", "ps1", "rb", "go"] },
|
|
10
|
+
{ "id": "media", "label": "Media", "extensions": ["jpg", "jpeg", "png", "gif", "svg", "mp4", "mp3"] }
|
|
11
|
+
],
|
|
12
|
+
"extensions": [
|
|
13
|
+
{ "ext": "pdf", "label": "PDF document", "group": "documents", "weight": 100 },
|
|
14
|
+
{ "ext": "docx", "label": "Word document", "group": "documents", "weight": 95 },
|
|
15
|
+
{ "ext": "doc", "label": "Word 97-2003 document", "group": "documents", "weight": 70 },
|
|
16
|
+
{ "ext": "pptx", "label": "PowerPoint presentation","group": "documents", "weight": 90 },
|
|
17
|
+
{ "ext": "ppt", "label": "PowerPoint 97-2003", "group": "documents", "weight": 60 },
|
|
18
|
+
{ "ext": "xlsx", "label": "Excel workbook", "group": "documents", "weight": 92 },
|
|
19
|
+
{ "ext": "xls", "label": "Excel 97-2003 workbook", "group": "documents", "weight": 65 },
|
|
20
|
+
{ "ext": "odt", "label": "OpenDocument text", "group": "documents", "weight": 40 },
|
|
21
|
+
{ "ext": "odp", "label": "OpenDocument presentation", "group": "documents", "weight": 35 },
|
|
22
|
+
{ "ext": "ods", "label": "OpenDocument spreadsheet", "group": "documents", "weight": 35 },
|
|
23
|
+
{ "ext": "rtf", "label": "Rich text", "group": "documents", "weight": 30 },
|
|
24
|
+
{ "ext": "txt", "label": "Plain text", "group": "documents", "weight": 80 },
|
|
25
|
+
{ "ext": "csv", "label": "CSV data", "group": "data", "weight": 85 },
|
|
26
|
+
{ "ext": "json", "label": "JSON data", "group": "data", "weight": 88, "defensive": true },
|
|
27
|
+
{ "ext": "xml", "label": "XML data", "group": "data", "weight": 82, "defensive": true },
|
|
28
|
+
{ "ext": "yaml", "label": "YAML configuration", "group": "data", "weight": 75, "defensive": true },
|
|
29
|
+
{ "ext": "yml", "label": "YAML configuration", "group": "data", "weight": 74, "defensive": true },
|
|
30
|
+
{ "ext": "sql", "label": "SQL dump / script", "group": "data", "weight": 78, "defensive": true },
|
|
31
|
+
{ "ext": "ini", "label": "INI configuration", "group": "data", "weight": 60, "defensive": true },
|
|
32
|
+
{ "ext": "cfg", "label": "Configuration file", "group": "data", "weight": 58, "defensive": true },
|
|
33
|
+
{ "ext": "conf", "label": "Configuration file", "group": "data", "weight": 62, "defensive": true },
|
|
34
|
+
{ "ext": "env", "label": "Environment file", "group": "data", "weight": 70, "defensive": true },
|
|
35
|
+
{ "ext": "toml", "label": "TOML configuration", "group": "data", "weight": 45, "defensive": true },
|
|
36
|
+
{ "ext": "properties", "label": "Java properties", "group": "data", "weight": 50, "defensive": true },
|
|
37
|
+
{ "ext": "log", "label": "Log file", "group": "logs", "weight": 84, "defensive": true },
|
|
38
|
+
{ "ext": "out", "label": "Process output", "group": "logs", "weight": 30, "defensive": true },
|
|
39
|
+
{ "ext": "err", "label": "Error output", "group": "logs", "weight": 30, "defensive": true },
|
|
40
|
+
{ "ext": "dmp", "label": "Memory dump", "group": "logs", "weight": 35, "defensive": true },
|
|
41
|
+
{ "ext": "trace","label": "Trace file", "group": "logs", "weight": 28, "defensive": true },
|
|
42
|
+
{ "ext": "zip", "label": "ZIP archive", "group": "archives", "weight": 76, "defensive": true },
|
|
43
|
+
{ "ext": "tar", "label": "TAR archive", "group": "archives", "weight": 55, "defensive": true },
|
|
44
|
+
{ "ext": "gz", "label": "Gzip archive", "group": "archives", "weight": 56, "defensive": true },
|
|
45
|
+
{ "ext": "tgz", "label": "Gzipped TAR archive", "group": "archives", "weight": 50, "defensive": true },
|
|
46
|
+
{ "ext": "7z", "label": "7-Zip archive", "group": "archives", "weight": 48, "defensive": true },
|
|
47
|
+
{ "ext": "rar", "label": "RAR archive", "group": "archives", "weight": 46, "defensive": true },
|
|
48
|
+
{ "ext": "bak", "label": "Backup file", "group": "archives", "weight": 72, "defensive": true },
|
|
49
|
+
{ "ext": "old", "label": "Old / renamed file", "group": "archives", "weight": 52, "defensive": true },
|
|
50
|
+
{ "ext": "js", "label": "JavaScript", "group": "code", "weight": 60 },
|
|
51
|
+
{ "ext": "map", "label": "Source map", "group": "code", "weight": 66, "defensive": true },
|
|
52
|
+
{ "ext": "py", "label": "Python source", "group": "code", "weight": 50 },
|
|
53
|
+
{ "ext": "php", "label": "PHP source", "group": "code", "weight": 48 },
|
|
54
|
+
{ "ext": "java", "label": "Java source", "group": "code", "weight": 40 },
|
|
55
|
+
{ "ext": "cs", "label": "C# source", "group": "code", "weight": 40 },
|
|
56
|
+
{ "ext": "sh", "label": "Shell script", "group": "code", "weight": 54, "defensive": true },
|
|
57
|
+
{ "ext": "ps1", "label": "PowerShell script", "group": "code", "weight": 52, "defensive": true },
|
|
58
|
+
{ "ext": "rb", "label": "Ruby source", "group": "code", "weight": 30 },
|
|
59
|
+
{ "ext": "go", "label": "Go source", "group": "code", "weight": 30 },
|
|
60
|
+
{ "ext": "jpg", "label": "JPEG image", "group": "media", "weight": 40 },
|
|
61
|
+
{ "ext": "jpeg", "label": "JPEG image", "group": "media", "weight": 30 },
|
|
62
|
+
{ "ext": "png", "label": "PNG image", "group": "media", "weight": 40 },
|
|
63
|
+
{ "ext": "gif", "label": "GIF image", "group": "media", "weight": 20 },
|
|
64
|
+
{ "ext": "svg", "label": "SVG image", "group": "media", "weight": 25 },
|
|
65
|
+
{ "ext": "mp4", "label": "MP4 video", "group": "media", "weight": 25 },
|
|
66
|
+
{ "ext": "mp3", "label": "MP3 audio", "group": "media", "weight": 20 }
|
|
67
|
+
]
|
|
68
|
+
}
|