ictrp-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/LICENSE +37 -0
- package/README.md +208 -0
- package/README_ZH.md +189 -0
- package/dist/cli/setup-cli.d.ts +14 -0
- package/dist/cli/setup-cli.js +230 -0
- package/dist/cli/setup-cli.js.map +1 -0
- package/dist/index.d.ts +18 -0
- package/dist/index.js +477 -0
- package/dist/index.js.map +1 -0
- package/dist/runtime/bootstrap.d.ts +99 -0
- package/dist/runtime/bootstrap.js +350 -0
- package/dist/runtime/bootstrap.js.map +1 -0
- package/dist/runtime/env-probe.d.ts +108 -0
- package/dist/runtime/env-probe.js +479 -0
- package/dist/runtime/env-probe.js.map +1 -0
- package/dist/runtime/sidecar-client.d.ts +50 -0
- package/dist/runtime/sidecar-client.js +120 -0
- package/dist/runtime/sidecar-client.js.map +1 -0
- package/dist/runtime/supervisor.d.ts +47 -0
- package/dist/runtime/supervisor.js +248 -0
- package/dist/runtime/supervisor.js.map +1 -0
- package/package.json +60 -0
- package/sidecar/ictrp_sidecar.py +602 -0
- package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
- package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
- package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
- package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
- package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
- package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
- package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
- package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
- package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
- package/sidecar/vendor/ictrp_mcp/server.py +368 -0
- package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
- package/sidecar/vendor/pyproject.toml +25 -0
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Error taxonomy.
|
|
2
|
+
|
|
3
|
+
The central design constraint: a failure must never be representable as a
|
|
4
|
+
successful empty result. `NO_RESULTS` is reserved for exactly one situation -- a
|
|
5
|
+
structurally valid CSV that arrived intact and contained zero data rows.
|
|
6
|
+
|
|
7
|
+
Anything else that prevents us from returning records raises, so that callers
|
|
8
|
+
receive an explicit failure rather than an empty list.
|
|
9
|
+
|
|
10
|
+
See docs/MEASUREMENTS.md for the measured behaviour this encodes.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from enum import Enum
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ErrorCode(str, Enum):
|
|
20
|
+
"""Stable, machine-readable failure codes."""
|
|
21
|
+
|
|
22
|
+
#: The one legitimate zero. Only produced by a validated CSV with 0 data rows.
|
|
23
|
+
NO_RESULTS = "NO_RESULTS"
|
|
24
|
+
|
|
25
|
+
#: Upstream refused or intercepted the request (WAF, 403/405/429/503, block page).
|
|
26
|
+
UPSTREAM_BLOCKED = "UPSTREAM_BLOCKED"
|
|
27
|
+
|
|
28
|
+
#: The portal responded, but not in the shape we depend on: a redirect to
|
|
29
|
+
#: NoAccess.aspx, a non-CSV content type where CSV was requested, a changed
|
|
30
|
+
#: header, or a missing required column.
|
|
31
|
+
UPSTREAM_CONTRACT_DRIFT = "UPSTREAM_CONTRACT_DRIFT"
|
|
32
|
+
|
|
33
|
+
#: Transport-level failure: timeout, connection reset, 5xx.
|
|
34
|
+
UPSTREAM_ERROR = "UPSTREAM_ERROR"
|
|
35
|
+
|
|
36
|
+
#: The caller supplied something we cannot accept.
|
|
37
|
+
INVALID_ARGUMENT = "INVALID_ARGUMENT"
|
|
38
|
+
|
|
39
|
+
#: A referenced result-set id is unknown or expired.
|
|
40
|
+
CACHE_MISS = "CACHE_MISS"
|
|
41
|
+
|
|
42
|
+
#: Session state could not be established (no hidden inputs on the form page).
|
|
43
|
+
SESSION_FAILED = "SESSION_FAILED"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
#: Codes for which retrying the identical request may plausibly succeed.
|
|
47
|
+
RETRYABLE: frozenset[ErrorCode] = frozenset(
|
|
48
|
+
{ErrorCode.UPSTREAM_ERROR, ErrorCode.SESSION_FAILED}
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
#: Codes that mean the upstream service did not give us data, as opposed to
|
|
52
|
+
#: having told us there is none.
|
|
53
|
+
FAILURE_CODES: frozenset[ErrorCode] = frozenset(
|
|
54
|
+
{
|
|
55
|
+
ErrorCode.UPSTREAM_BLOCKED,
|
|
56
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
57
|
+
ErrorCode.UPSTREAM_ERROR,
|
|
58
|
+
ErrorCode.SESSION_FAILED,
|
|
59
|
+
}
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class IctrpError(Exception):
|
|
64
|
+
"""A failure carrying enough upstream context to be diagnosable.
|
|
65
|
+
|
|
66
|
+
`body_excerpt` is stripped of HTML and truncated; it is for diagnosis, never
|
|
67
|
+
for parsing.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
def __init__(
|
|
71
|
+
self,
|
|
72
|
+
code: ErrorCode,
|
|
73
|
+
message: str,
|
|
74
|
+
*,
|
|
75
|
+
upstream_status: int | None = None,
|
|
76
|
+
upstream_content_type: str | None = None,
|
|
77
|
+
upstream_body_excerpt: str | None = None,
|
|
78
|
+
hint: str | None = None,
|
|
79
|
+
detail: str | None = None,
|
|
80
|
+
) -> None:
|
|
81
|
+
super().__init__(message)
|
|
82
|
+
self.code = code
|
|
83
|
+
self.message = message
|
|
84
|
+
self.upstream_status = upstream_status
|
|
85
|
+
self.upstream_content_type = upstream_content_type
|
|
86
|
+
self.upstream_body_excerpt = upstream_body_excerpt
|
|
87
|
+
self.hint = hint
|
|
88
|
+
self.detail = detail
|
|
89
|
+
|
|
90
|
+
@property
|
|
91
|
+
def retryable(self) -> bool:
|
|
92
|
+
return self.code in RETRYABLE
|
|
93
|
+
|
|
94
|
+
def to_dict(self) -> dict[str, Any]:
|
|
95
|
+
out: dict[str, Any] = {
|
|
96
|
+
"status": _status_for(self.code),
|
|
97
|
+
"error_code": self.code.value,
|
|
98
|
+
"message": self.message,
|
|
99
|
+
"retryable": self.retryable,
|
|
100
|
+
}
|
|
101
|
+
if self.detail:
|
|
102
|
+
out["detail"] = self.detail
|
|
103
|
+
if self.upstream_status is not None:
|
|
104
|
+
out["upstream_status"] = self.upstream_status
|
|
105
|
+
if self.upstream_content_type:
|
|
106
|
+
out["upstream_content_type"] = self.upstream_content_type
|
|
107
|
+
if self.upstream_body_excerpt:
|
|
108
|
+
out["upstream_body_excerpt"] = self.upstream_body_excerpt
|
|
109
|
+
if self.hint:
|
|
110
|
+
out["hint"] = self.hint
|
|
111
|
+
return out
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _status_for(code: ErrorCode) -> str:
|
|
115
|
+
return {
|
|
116
|
+
ErrorCode.NO_RESULTS: "no_results",
|
|
117
|
+
ErrorCode.UPSTREAM_BLOCKED: "upstream_blocked",
|
|
118
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT: "upstream_contract_drift",
|
|
119
|
+
ErrorCode.UPSTREAM_ERROR: "upstream_error",
|
|
120
|
+
ErrorCode.INVALID_ARGUMENT: "invalid_argument",
|
|
121
|
+
ErrorCode.CACHE_MISS: "cache_miss",
|
|
122
|
+
ErrorCode.SESSION_FAILED: "session_failed",
|
|
123
|
+
}[code]
|
|
File without changes
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
"""Response classification -- applied to every upstream response before parsing.
|
|
2
|
+
|
|
3
|
+
This module is the reason a WAF block page can never be mistaken for "no trials
|
|
4
|
+
exist". The primary check is structural (content type + header shape), not
|
|
5
|
+
textual: if we asked for a CSV and did not get a CSV with the expected header,
|
|
6
|
+
that is a failure regardless of status code and regardless of body wording.
|
|
7
|
+
|
|
8
|
+
Text markers exist only to improve the `hint`; they are never load-bearing,
|
|
9
|
+
because the upstream block page's exact wording is not something we control or
|
|
10
|
+
should depend on.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import csv
|
|
16
|
+
import io
|
|
17
|
+
import re
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
from ..data.columns import REQUIRED_COLUMNS
|
|
21
|
+
from ..errors import ErrorCode, IctrpError
|
|
22
|
+
|
|
23
|
+
#: MIME type the portal uses for a successful export.
|
|
24
|
+
CSV_CONTENT_TYPE = "application/vnd.ms-excel"
|
|
25
|
+
|
|
26
|
+
#: Filename the portal advertises on a successful export.
|
|
27
|
+
CSV_FILENAME = "IctrpResults.csv"
|
|
28
|
+
|
|
29
|
+
#: Secondary classifier only. Used to pick a better hint, never to decide
|
|
30
|
+
#: success or failure.
|
|
31
|
+
_BLOCK_MARKERS = (
|
|
32
|
+
"access denied",
|
|
33
|
+
"request rejected",
|
|
34
|
+
"web application firewall",
|
|
35
|
+
"attention required",
|
|
36
|
+
"just a moment",
|
|
37
|
+
"security threat",
|
|
38
|
+
"your request has been blocked",
|
|
39
|
+
"访问被阻断",
|
|
40
|
+
"aliyunwaf",
|
|
41
|
+
"acw_sc__v2",
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
#: Marker in the redirect target that means we misused the session.
|
|
45
|
+
_NOACCESS_MARKER = "/noaccess.aspx"
|
|
46
|
+
|
|
47
|
+
_MAX_EXCERPT = 512
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class CsvPayload:
|
|
52
|
+
"""A validated export response."""
|
|
53
|
+
|
|
54
|
+
header: tuple[str, ...]
|
|
55
|
+
rows: list[list[str]]
|
|
56
|
+
content_type: str | None
|
|
57
|
+
reported_total: int | None = None
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def row_count(self) -> int:
|
|
61
|
+
return len(self.rows)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def is_empty(self) -> bool:
|
|
65
|
+
"""True only for a structurally valid export with zero data rows."""
|
|
66
|
+
return len(self.rows) == 0
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _excerpt(body: bytes | str, limit: int = _MAX_EXCERPT) -> str:
|
|
70
|
+
if isinstance(body, bytes):
|
|
71
|
+
text = body.decode("utf-8", "replace")
|
|
72
|
+
else:
|
|
73
|
+
text = body
|
|
74
|
+
text = re.sub(r"<script\b[^>]*>.*?</script>", " ", text, flags=re.S | re.I)
|
|
75
|
+
text = re.sub(r"<style\b[^>]*>.*?</style>", " ", text, flags=re.S | re.I)
|
|
76
|
+
text = re.sub(r"<[^>]+>", " ", text)
|
|
77
|
+
text = re.sub(r"\s+", " ", text).strip()
|
|
78
|
+
return text[:limit]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def looks_like_block_page(body: bytes | str) -> bool:
|
|
82
|
+
text = _excerpt(body, 8000).lower()
|
|
83
|
+
return any(marker in text for marker in _BLOCK_MARKERS)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _is_successful_csv_envelope(
|
|
87
|
+
status: int, content_type: str | None, content_disposition: str | None
|
|
88
|
+
) -> bool:
|
|
89
|
+
if status != 200:
|
|
90
|
+
return False
|
|
91
|
+
if not content_type:
|
|
92
|
+
return False
|
|
93
|
+
base = content_type.split(";")[0].strip().lower()
|
|
94
|
+
if base == CSV_CONTENT_TYPE or base in ("text/csv", "application/csv"):
|
|
95
|
+
return True
|
|
96
|
+
# Some responses omit the exact MIME but still advertise the filename.
|
|
97
|
+
if content_disposition and CSV_FILENAME.lower() in content_disposition.lower():
|
|
98
|
+
return True
|
|
99
|
+
return False
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _parse_csv(body: bytes) -> tuple[tuple[str, ...], list[list[str]]]:
|
|
103
|
+
text = body.decode("utf-8-sig", "replace")
|
|
104
|
+
reader = csv.reader(io.StringIO(text))
|
|
105
|
+
try:
|
|
106
|
+
header = next(reader)
|
|
107
|
+
except StopIteration:
|
|
108
|
+
return (), []
|
|
109
|
+
header = tuple(h.strip() for h in header)
|
|
110
|
+
rows = [r for r in reader if r and any(c.strip() for c in r)]
|
|
111
|
+
return header, rows
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _validate_header(header: tuple[str, ...]) -> None:
|
|
115
|
+
missing = REQUIRED_COLUMNS - set(header)
|
|
116
|
+
if missing:
|
|
117
|
+
raise IctrpError(
|
|
118
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
119
|
+
"ICTRP export header is missing required column(s)",
|
|
120
|
+
detail=f"missing: {sorted(missing)}",
|
|
121
|
+
hint=(
|
|
122
|
+
"The portal's export format has changed. Update data/columns.py "
|
|
123
|
+
"and the field mapping before relying on results."
|
|
124
|
+
),
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def classify_export(
|
|
129
|
+
*,
|
|
130
|
+
status: int,
|
|
131
|
+
content_type: str | None,
|
|
132
|
+
content_disposition: str | None,
|
|
133
|
+
body: bytes,
|
|
134
|
+
location: str | None = None,
|
|
135
|
+
reported_total: int | None = None,
|
|
136
|
+
) -> CsvPayload:
|
|
137
|
+
"""Classify an export response.
|
|
138
|
+
|
|
139
|
+
Returns a validated `CsvPayload` (possibly empty) or raises `IctrpError`.
|
|
140
|
+
An empty payload is the ONLY way a caller may report zero results.
|
|
141
|
+
"""
|
|
142
|
+
if location and _NOACCESS_MARKER in location.lower():
|
|
143
|
+
raise IctrpError(
|
|
144
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
145
|
+
"Portal redirected the export to NoAccess.aspx",
|
|
146
|
+
upstream_status=status,
|
|
147
|
+
upstream_content_type=content_type,
|
|
148
|
+
hint=(
|
|
149
|
+
"The export POST was rejected. Verify that the results-page hidden "
|
|
150
|
+
"inputs were used verbatim and that TextBox1/Button1 were NOT sent, "
|
|
151
|
+
"since the results page does not contain them."
|
|
152
|
+
),
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
if status in (403, 405, 429, 503):
|
|
156
|
+
raise IctrpError(
|
|
157
|
+
ErrorCode.UPSTREAM_BLOCKED,
|
|
158
|
+
f"Upstream refused the request (HTTP {status})",
|
|
159
|
+
upstream_status=status,
|
|
160
|
+
upstream_content_type=content_type,
|
|
161
|
+
upstream_body_excerpt=_excerpt(body),
|
|
162
|
+
hint=(
|
|
163
|
+
"The portal blocked this request. This is not an empty result set; "
|
|
164
|
+
"do not report zero trials. Retry later, and reduce request rate."
|
|
165
|
+
),
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
if status >= 500:
|
|
169
|
+
raise IctrpError(
|
|
170
|
+
ErrorCode.UPSTREAM_ERROR,
|
|
171
|
+
f"Upstream server error (HTTP {status})",
|
|
172
|
+
upstream_status=status,
|
|
173
|
+
upstream_content_type=content_type,
|
|
174
|
+
upstream_body_excerpt=_excerpt(body),
|
|
175
|
+
hint="Transient upstream failure. Retry with backoff.",
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
if not _is_successful_csv_envelope(status, content_type, content_disposition):
|
|
179
|
+
hint = "Response was not a CSV export."
|
|
180
|
+
if looks_like_block_page(body):
|
|
181
|
+
hint = (
|
|
182
|
+
"Response looks like an HTML block/interstitial page, not an export. "
|
|
183
|
+
"Treat as blocked, not as zero results."
|
|
184
|
+
)
|
|
185
|
+
raise IctrpError(
|
|
186
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
187
|
+
"Export response was not a CSV",
|
|
188
|
+
upstream_status=status,
|
|
189
|
+
upstream_content_type=content_type,
|
|
190
|
+
upstream_body_excerpt=_excerpt(body),
|
|
191
|
+
hint=hint,
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
header, rows = _parse_csv(body)
|
|
195
|
+
if not header:
|
|
196
|
+
raise IctrpError(
|
|
197
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
198
|
+
"Export body contained no CSV header row",
|
|
199
|
+
upstream_status=status,
|
|
200
|
+
upstream_content_type=content_type,
|
|
201
|
+
upstream_body_excerpt=_excerpt(body),
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
_validate_header(header)
|
|
205
|
+
|
|
206
|
+
# Pad short rows so column indexing is always safe. Upstream can emit rows
|
|
207
|
+
# with trailing empty fields omitted.
|
|
208
|
+
width = len(header)
|
|
209
|
+
rows = [(r + [""] * width)[:width] for r in rows]
|
|
210
|
+
|
|
211
|
+
return CsvPayload(
|
|
212
|
+
header=header,
|
|
213
|
+
rows=rows,
|
|
214
|
+
content_type=content_type,
|
|
215
|
+
reported_total=reported_total,
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def classify_form_page(*, status: int, body: bytes) -> str:
|
|
220
|
+
"""Validate the initial GET of the search form. Returns the HTML text."""
|
|
221
|
+
text = body.decode("utf-8", "replace")
|
|
222
|
+
if status in (403, 405, 429, 503):
|
|
223
|
+
raise IctrpError(
|
|
224
|
+
ErrorCode.UPSTREAM_BLOCKED,
|
|
225
|
+
f"Search form was refused (HTTP {status})",
|
|
226
|
+
upstream_status=status,
|
|
227
|
+
upstream_body_excerpt=_excerpt(body),
|
|
228
|
+
)
|
|
229
|
+
if status >= 500:
|
|
230
|
+
raise IctrpError(
|
|
231
|
+
ErrorCode.UPSTREAM_ERROR,
|
|
232
|
+
f"Search form returned HTTP {status}",
|
|
233
|
+
upstream_status=status,
|
|
234
|
+
upstream_body_excerpt=_excerpt(body),
|
|
235
|
+
)
|
|
236
|
+
if status != 200:
|
|
237
|
+
raise IctrpError(
|
|
238
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
239
|
+
f"Unexpected status for search form: HTTP {status}",
|
|
240
|
+
upstream_status=status,
|
|
241
|
+
upstream_body_excerpt=_excerpt(body),
|
|
242
|
+
)
|
|
243
|
+
if looks_like_block_page(body):
|
|
244
|
+
raise IctrpError(
|
|
245
|
+
ErrorCode.UPSTREAM_BLOCKED,
|
|
246
|
+
"Search form returned a block/interstitial page",
|
|
247
|
+
upstream_status=status,
|
|
248
|
+
upstream_body_excerpt=_excerpt(body),
|
|
249
|
+
)
|
|
250
|
+
return text
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
_RECORDS_RE = re.compile(r"([\d,]+)\s*records", re.I)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def parse_reported_total(html: str) -> int | None:
|
|
257
|
+
"""Extract the portal's own 'N records' figure from a results page.
|
|
258
|
+
|
|
259
|
+
This is the portal's claim, NOT a verified count of what we received. The
|
|
260
|
+
two routinely differ -- measured shortfalls of 0.4% to 29% -- so callers must
|
|
261
|
+
surface both and never conflate them. See docs/MEASUREMENTS.md section 2.
|
|
262
|
+
"""
|
|
263
|
+
match = _RECORDS_RE.search(html)
|
|
264
|
+
if not match:
|
|
265
|
+
return None
|
|
266
|
+
try:
|
|
267
|
+
return int(match.group(1).replace(",", ""))
|
|
268
|
+
except ValueError:
|
|
269
|
+
return None
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""HTML form-state extraction.
|
|
2
|
+
|
|
3
|
+
The portal is ASP.NET WebForms. Every step of the chain depends on hidden inputs
|
|
4
|
+
carried forward from the previous response, so extraction has to be exact and
|
|
5
|
+
lossless -- a dropped field produces a 302 to NoAccess.aspx, which is
|
|
6
|
+
indistinguishable from a block unless you check the redirect target.
|
|
7
|
+
|
|
8
|
+
The extraction is deliberately dumb (regex over input tags) rather than
|
|
9
|
+
DOM-based, because the pages mix well-formed and sloppy markup and we only ever
|
|
10
|
+
need attributed name/value pairs.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import html as _html
|
|
16
|
+
import re
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
|
|
19
|
+
#: Inputs that carry WebForms state and must be replayed on the next POST.
|
|
20
|
+
STATE_FIELDS: tuple[str, ...] = (
|
|
21
|
+
"__VIEWSTATE",
|
|
22
|
+
"__VIEWSTATEGENERATOR",
|
|
23
|
+
"__VIEWSTATEENCRYPTED",
|
|
24
|
+
"__EVENTVALIDATION",
|
|
25
|
+
"__EVENTTARGET",
|
|
26
|
+
"__EVENTARGUMENT",
|
|
27
|
+
"__LASTFOCUS",
|
|
28
|
+
"ToolkitScriptManager_HiddenField",
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
_INPUT_RE = re.compile(r"<input\b[^>]*>", re.I)
|
|
32
|
+
_ATTR_RE = re.compile(r"""([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+))""")
|
|
33
|
+
_TAG_RE = re.compile(r"<[^>]+>")
|
|
34
|
+
_WS_RE = re.compile(r"\s+")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class FormState:
|
|
39
|
+
"""Hidden inputs harvested from one page, ready to replay."""
|
|
40
|
+
|
|
41
|
+
fields: dict[str, str]
|
|
42
|
+
|
|
43
|
+
def merged_with(self, overrides: dict[str, str]) -> dict[str, str]:
|
|
44
|
+
"""Return the state fields plus caller-supplied control values."""
|
|
45
|
+
out = dict(self.fields)
|
|
46
|
+
out.update(overrides)
|
|
47
|
+
return out
|
|
48
|
+
|
|
49
|
+
def has(self, name: str) -> bool:
|
|
50
|
+
return name in self.fields
|
|
51
|
+
|
|
52
|
+
def missing_state(self) -> list[str]:
|
|
53
|
+
"""State fields we expect on a form page but did not find."""
|
|
54
|
+
required = ("__VIEWSTATE", "__EVENTVALIDATION")
|
|
55
|
+
return [f for f in required if f not in self.fields]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _attrs(tag: str) -> dict[str, str]:
|
|
59
|
+
got: dict[str, str] = {}
|
|
60
|
+
for m in _ATTR_RE.finditer(tag):
|
|
61
|
+
name = m.group(1).lower()
|
|
62
|
+
value = m.group(2) if m.group(2) is not None else (
|
|
63
|
+
m.group(3) if m.group(3) is not None else (m.group(4) or "")
|
|
64
|
+
)
|
|
65
|
+
got[name] = _html.unescape(value)
|
|
66
|
+
return got
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def extract_inputs(html: str) -> dict[str, str]:
|
|
70
|
+
"""Return every named input's value from a page.
|
|
71
|
+
|
|
72
|
+
Includes buttons and text boxes, because the export step needs to know which
|
|
73
|
+
controls exist -- notably to *exclude* controls that the results page does
|
|
74
|
+
not carry.
|
|
75
|
+
"""
|
|
76
|
+
found: dict[str, str] = {}
|
|
77
|
+
for tag in _INPUT_RE.finditer(html):
|
|
78
|
+
attrs = _attrs(tag.group(0))
|
|
79
|
+
name = attrs.get("name")
|
|
80
|
+
if not name:
|
|
81
|
+
continue
|
|
82
|
+
found[name] = attrs.get("value", "")
|
|
83
|
+
return found
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def extract_form_state(html: str) -> FormState:
|
|
87
|
+
"""Extract just the WebForms state fields."""
|
|
88
|
+
all_inputs = extract_inputs(html)
|
|
89
|
+
return FormState({k: v for k, v in all_inputs.items() if k in STATE_FIELDS})
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def extract_select_options(html: str, select_name: str) -> list[tuple[str, str]]:
|
|
93
|
+
"""Return (value, label) pairs for a named <select>."""
|
|
94
|
+
pattern = re.compile(
|
|
95
|
+
r"<select\b[^>]*\bname\s*=\s*[\"']?" + re.escape(select_name) + r"[\"']?[^>]*>(.*?)</select>",
|
|
96
|
+
re.I | re.S,
|
|
97
|
+
)
|
|
98
|
+
match = pattern.search(html)
|
|
99
|
+
if not match:
|
|
100
|
+
return []
|
|
101
|
+
options: list[tuple[str, str]] = []
|
|
102
|
+
for opt in re.finditer(r"<option\b([^>]*)>(.*?)</option>", match.group(1), re.I | re.S):
|
|
103
|
+
attrs = _attrs(opt.group(1))
|
|
104
|
+
label = _WS_RE.sub(" ", _TAG_RE.sub("", opt.group(2))).strip()
|
|
105
|
+
options.append((_html.unescape(attrs.get("value", "")), label))
|
|
106
|
+
return options
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def text_of(html: str) -> str:
|
|
110
|
+
"""Collapse a fragment to plain text."""
|
|
111
|
+
return _WS_RE.sub(" ", _TAG_RE.sub(" ", html)).strip()
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def build_export_body(results_html: str, *, export_control: str = "Button7") -> dict[str, str]:
|
|
115
|
+
"""Build the export POST body from a results page.
|
|
116
|
+
|
|
117
|
+
CRITICAL: the body is built ONLY from fields the results page actually
|
|
118
|
+
carries, plus the export button. The search controls (`TextBox1`, `Button1`)
|
|
119
|
+
are deliberately NOT sent, even when present. Measured: including controls
|
|
120
|
+
that the results page does not carry makes the portal answer `302 Found` to
|
|
121
|
+
`/NoAccess.aspx?aspxerrorpath=/Default.aspx`.
|
|
122
|
+
|
|
123
|
+
The caller is responsible for verifying the search controls are absent via
|
|
124
|
+
`assert_export_body_excludes_search_controls`.
|
|
125
|
+
"""
|
|
126
|
+
state = extract_form_state(results_html)
|
|
127
|
+
body = dict(state.fields)
|
|
128
|
+
body[export_control] = "Export to CSV"
|
|
129
|
+
return body
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def assert_export_body_excludes_search_controls(body: dict[str, str]) -> None:
|
|
133
|
+
"""Guard against reintroducing the measured 302-to-NoAccess failure.
|
|
134
|
+
|
|
135
|
+
`TextBox1`/`Button1` are search-form controls. Sending them on the export
|
|
136
|
+
POST is the specific mistake that produced the measured error redirect.
|
|
137
|
+
"""
|
|
138
|
+
for forbidden in ("TextBox1", "Button1"):
|
|
139
|
+
if forbidden in body:
|
|
140
|
+
raise AssertionError(
|
|
141
|
+
f"export body must not contain search control {forbidden!r}; "
|
|
142
|
+
"including it produces a 302 to /NoAccess.aspx"
|
|
143
|
+
)
|