stata-code 0.12.1__tar.gz → 0.12.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {stata_code-0.12.1 → stata_code-0.12.2}/CHANGELOG.md +57 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/PKG-INFO +1 -1
- {stata_code-0.12.1 → stata_code-0.12.2}/SCHEMA.md +16 -6
- {stata_code-0.12.1 → stata_code-0.12.2}/pyproject.toml +1 -1
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/__init__.py +1 -1
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/estimation.py +152 -12
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/jobs.py +16 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/runner.py +93 -2
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp/server.py +1 -1
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_agent_ergonomics.py +43 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_estimation.py +114 -2
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_real_stata.py +74 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_runner_helpers.py +28 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/.gitignore +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/LICENSE +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/LICENSE-POLICY.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/PUBLISHING.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/README.en.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/README.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/docs/competitive-landscape.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/docs/design/hard_timeout.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/docs/industry-leader-roadmap.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/docs/quickstart.zh.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/01-basic-regression.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/02-did-card-krueger.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/03-graphs.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/04-multi-session.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/05-large-matrix.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/06-cross-stack-parity-audit.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/07-data-mcp-handoff.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/examples/README.md +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/schema/run_result.schema.json +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/scripts/build_skill_zip.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/scripts/build_standalone.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/scripts/check_github_actions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/scripts/check_versions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/scripts/export_schema.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/__main__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/cli.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/_pool.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/_refs.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/_runtime.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/console.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/daemon.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/errors.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/handoff.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/lint.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/log_artifacts.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/notebook.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/policy.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/provenance.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/run_index.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/schema.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/doctor.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/__main__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/assets/logo-32x32.png +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/assets/logo-64x64.png +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/assets/logo-svg.svg +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/kernel.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp/__main__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp_setup.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/conftest.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/fixtures/.gitkeep +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_benchmark.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_bugfix_regressions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_cancel.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_cli.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_console.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_daemon.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_doctor.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_errors.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_github_actions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_handoff.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_kernel.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_lint.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_lint_mcp.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_log_artifacts.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_mcp.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_mcp_kernel_extra.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_mcp_transport.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_method_prompts.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_new_tools.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_notebook.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_notebook_phase2.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_policy.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_pool.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_pool_refs_extra.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_provenance.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_public_api.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_release_versions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_run_index.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_runner.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_runtime_discovery.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_schema.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_schema_artifact.py +0 -0
- {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_skill_package.py +0 -0
|
@@ -6,6 +6,63 @@ to semver-major.minor for the result schema (see `SCHEMA.md` §6).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## 0.12.2 — 2026-08-11
|
|
10
|
+
|
|
11
|
+
Correctness release. `results.estimation` could report confidence intervals
|
|
12
|
+
and p-values that disagreed with the Stata log printed beside them; if you
|
|
13
|
+
read coefficients out of `results.estimation` rather than the log, upgrade.
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
|
|
17
|
+
- **`results.estimation` reported normal-approximation inference for `t` models
|
|
18
|
+
whenever `r(table)` had been cleared.** `r(table)` only survives until the
|
|
19
|
+
next command, so any block that runs something after its estimation —
|
|
20
|
+
`graph export`, `esttab`, a final `summarize` — fell back to rebuilding the
|
|
21
|
+
table from `e(b)` / `e(V)`, and that path hardcoded a normal approximation.
|
|
22
|
+
The result was a payload that quietly disagreed with the log beside it:
|
|
23
|
+
|
|
24
|
+
```text
|
|
25
|
+
regress price_k mpg weight foreign
|
|
26
|
+
scatter price_k mpg // clears r(table)
|
|
27
|
+
|
|
28
|
+
log: mpg 95% CI [-.1261758, .169883] P>|t| 0.769
|
|
29
|
+
results.estimation: mpg 95% CI [-0.12362, 0.16732] p 0.7684
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
The fallback now follows Stata's own rule — a *t* table on `e(df_r)` degrees
|
|
33
|
+
of freedom when that scalar is set, *z* otherwise — so the rebuilt table
|
|
34
|
+
matches the printed one to the last digit. `ci_level` also honours `e(level)`
|
|
35
|
+
instead of assuming 95, so `regress, level(90)` is no longer mislabelled.
|
|
36
|
+
Agents told to prefer `structuredContent` over the log were the ones exposed
|
|
37
|
+
to this; the `r(table)` path was always correct.
|
|
38
|
+
|
|
39
|
+
A new `estimation_from_e_b_v` warning records when the rebuild happened. It
|
|
40
|
+
fires only for an estimation produced by the current run, since `e()` is
|
|
41
|
+
session-global and would otherwise re-warn on every later call.
|
|
42
|
+
|
|
43
|
+
- **`include_results: "none"` hollowed out `results.estimation`.** `n_obs`,
|
|
44
|
+
`df_model`, `df_resid` and `model_stats` are all read from `e()` scalars, so
|
|
45
|
+
suppressing the `r()` / `e()` *echo* also removed the model-level numbers —
|
|
46
|
+
even though `include_estimation` is the documented knob for that block, and
|
|
47
|
+
SCHEMA.md §4 already promised `include_results` "never affects
|
|
48
|
+
`results.estimation`". The scalars are now always read when estimation is
|
|
49
|
+
wanted, and only withheld from the wire.
|
|
50
|
+
|
|
51
|
+
- **A finished background run reported a duration that kept growing.**
|
|
52
|
+
`Job.elapsed_ms` recomputed `now - submitted` on every poll instead of
|
|
53
|
+
freezing at completion, so a 1.7 s bootstrap polled ten minutes later
|
|
54
|
+
reported ten minutes. Callers attributing wall-clock time to Stata were
|
|
55
|
+
reading their own polling delay. The value is now frozen before the terminal
|
|
56
|
+
status is published, on both the success and error paths.
|
|
57
|
+
|
|
58
|
+
### Changed
|
|
59
|
+
|
|
60
|
+
- **Macro values longer than 256 characters are elided on the wire** with an
|
|
61
|
+
explicit `… (N more chars elided)` marker. This is aimed at `e(rngstate)`,
|
|
62
|
+
roughly 2 KB of hex emitted on every `bootstrap` / `permute` / `simulate`
|
|
63
|
+
run that no consumer can act on. Names are always kept, and the estimation
|
|
64
|
+
contract is still built from the uncapped values.
|
|
65
|
+
|
|
9
66
|
## 0.12.1 — 2026-07-28
|
|
10
67
|
|
|
11
68
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: stata-code
|
|
3
|
-
Version: 0.12.
|
|
3
|
+
Version: 0.12.2
|
|
4
4
|
Summary: Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)
|
|
5
5
|
Project-URL: Homepage, https://github.com/brycewang-stanford/stata-code
|
|
6
6
|
Project-URL: Repository, https://github.com/brycewang-stanford/stata-code
|
|
@@ -343,7 +343,7 @@ Stata's `r()` and `e()` return dictionaries, structurally separated. Each follow
|
|
|
343
343
|
| Sub-field | Type | Notes |
|
|
344
344
|
| --- | --- | --- |
|
|
345
345
|
| `scalars` | `dict<str, number \| null>` | Native floats / ints. Stata's system missing (`.`) → JSON `null`. Extended missings (`.a`–`.z`) → `null` with information loss. |
|
|
346
|
-
| `macros` | `dict<str, string>` | Stata macro values verbatim. |
|
|
346
|
+
| `macros` | `dict<str, string>` | Stata macro values verbatim, except that a value longer than 256 characters is truncated and suffixed with `… (N more chars elided)`. The cap exists for macros like `e(rngstate)`, which is ~2 KB of hex on every `bootstrap` / `permute` / `simulate` run and carries nothing a consumer can act on. The name is always kept, and `results.estimation` is derived from the uncapped values. |
|
|
347
347
|
| `matrices` | `dict<str, Matrix>` | See `Matrix` below. |
|
|
348
348
|
|
|
349
349
|
**`Matrix` shape:**
|
|
@@ -383,9 +383,9 @@ Stata's `r()` and `e()` return dictionaries, structurally separated. Each follow
|
|
|
383
383
|
| `n_obs` | `int \| null` | Integer form of `e(N)` when available. |
|
|
384
384
|
| `df_model` | `number \| null` | Mirrors `e(df_m)`. |
|
|
385
385
|
| `df_resid` | `number \| null` | Mirrors `e(df_r)`. |
|
|
386
|
-
| `statistic_kind` | `"t" \| "z"` | Which statistic fills each coefficient's `statistic` field. |
|
|
387
|
-
| `source` | `"r_table" \| "e_b_v"` | `r_table` means values were copied from Stata's displayed `r(table)` after verifying its columns and `b` row match `e(b)`; `e_b_v` means point estimates come from `e(b)` and inference, when present, is computed from `e(V)
|
|
388
|
-
| `ci_level` | `number` | Confidence level used for `ci_low` / `ci_high`;
|
|
386
|
+
| `statistic_kind` | `"t" \| "z"` | Which statistic fills each coefficient's `statistic` field, and which distribution produced `p_value` / `ci_low` / `ci_high`. On the `e_b_v` path this follows Stata's own rule — `t` on `df_resid` degrees of freedom when `e(df_r)` is set, `z` otherwise — so a rebuilt table agrees with the printed log rather than reporting a normal-approximation interval next to a `P>\|t\|` column. |
|
|
387
|
+
| `source` | `"r_table" \| "e_b_v"` | `r_table` means values were copied from Stata's displayed `r(table)` after verifying its columns and `b` row match `e(b)`; `e_b_v` means point estimates come from `e(b)` and inference, when present, is computed from `e(V)`. A matrix returned by `ref` is resolved before use, so a deferred `e(V)` still yields standard errors. |
|
|
388
|
+
| `ci_level` | `number` | Confidence level used for `ci_low` / `ci_high`. Mirrors `e(level)` when the command stored it, so `regress, level(90)` reports `90.0`; defaults to `95.0` otherwise. |
|
|
389
389
|
| `coefficients` | `array<Coefficient>` | One row per term in `e(b)`, subject to the caller's `include_estimation` / `max_coefficients` budget. |
|
|
390
390
|
| `n_coefficients` | `int` | The model's true term count. Equals `coefficients.length` unless the caller trimmed the table, so `12` rows out of `n_coefficients: 141` is never mistaken for a 12-term model. |
|
|
391
391
|
| `coefficients_truncated` | `bool` | `true` when rows were dropped to satisfy the budget. |
|
|
@@ -564,11 +564,21 @@ The rc-to-kind table is approximate and lives in code (`stata_code.core.errors`)
|
|
|
564
564
|
|
|
565
565
|
| Field | Type | Notes |
|
|
566
566
|
| --- | --- | --- |
|
|
567
|
-
| `kind` | `string` | Open enum. Common values: `convergence`, `singular`, `boundary`, `omitted_collinear`, `non_finite`, `dataset_variables_truncated`, `unknown`. |
|
|
567
|
+
| `kind` | `string` | Open enum. Common values: `convergence`, `singular`, `boundary`, `omitted_collinear`, `non_finite`, `dataset_variables_truncated`, `log_closed`, `output_tracking_skipped`, `estimation_from_e_b_v`, `unknown`. |
|
|
568
568
|
| `message` | `string` | Human-readable, single line. Truncated to 1,024 characters. |
|
|
569
569
|
|
|
570
570
|
Warnings are de-duplicated by `(kind, message)`.
|
|
571
571
|
|
|
572
|
+
`estimation_from_e_b_v` reports that this run performed an estimation whose
|
|
573
|
+
`r(table)` was already gone by the time results were read — a later command in
|
|
574
|
+
the same submission cleared it — so `results.estimation` was rebuilt from
|
|
575
|
+
`e(b)` / `e(V)`. The rebuilt numbers still match the printed log (see
|
|
576
|
+
`statistic_kind` in §3.5), so this is provenance rather than a correctness
|
|
577
|
+
alarm; putting the estimation last in the block restores the `r_table` path.
|
|
578
|
+
It is emitted only when the estimation was produced by *this* run: `e()` is
|
|
579
|
+
session-global, so a later `summarize` in the same session keeps reporting the
|
|
580
|
+
inherited table through the same fallback without re-warning.
|
|
581
|
+
|
|
572
582
|
### 3.9 `origin`
|
|
573
583
|
|
|
574
584
|
```json
|
|
@@ -607,7 +617,7 @@ The schema also dictates what callers may *ask for*. Every frontend exposes the
|
|
|
607
617
|
| `include_graphs` | `"ref" \| "inline" \| "none"` | `"ref"` | `"none"` skips graph capture entirely (cheapest); `"ref"` captures and returns refs; `"inline"` base64-encodes bytes into `inline`. |
|
|
608
618
|
| `graph_format` | `"png" \| "svg" \| "pdf"` | `"png"` | Render format. |
|
|
609
619
|
| `include_dataset_variables` | `bool` | `true` | Set `false` to omit `dataset.variables`. |
|
|
610
|
-
| `include_results` | `"none" \| "scalars" \| "full"` | `"scalars"` | Payload budget for `results.r` / `results.e`. `"scalars"` inlines scalars and macros and emits every matrix as a stub (§3.4); `"full"` inlines matrix values up to the ~10,000-cell cap; `"none"` omits `r()` / `e()` entirely. Never affects `results.estimation
|
|
620
|
+
| `include_results` | `"none" \| "scalars" \| "full"` | `"scalars"` | Payload budget for `results.r` / `results.e`. `"scalars"` inlines scalars and macros and emits every matrix as a stub (§3.4); `"full"` inlines matrix values up to the ~10,000-cell cap; `"none"` omits `r()` / `e()` entirely. Never affects `results.estimation`: the model-level fields (`n_obs`, `df_model`, `df_resid`, `model_stats`, `depvar`) are read from `e()` regardless of this setting and merely withheld from the wire, so `"none"` does not hollow out the estimation contract. Use `include_estimation` to trim that block. |
|
|
611
621
|
| `include_estimation` | `"none" \| "summary" \| "full"` | `"full"` | Payload budget for `results.estimation`. `"summary"` keeps the model-level block and drops per-term rows. |
|
|
612
622
|
| `max_coefficients` | `int \| null` | `null` | Cap on `estimation.coefficients` rows. `n_coefficients` still reports the true count. |
|
|
613
623
|
| `timeout_ms` | `int \| null` | `600000` (10 min) | Hard timeout. `null` disables. On expiry, returns `ok: false`, `error.kind: "timeout"`, `rc: -2`. The budget covers **queueing**: a call waiting on a session whose Stata process is mid-run returns `rc: -5`, `error.kind: "session_busy"` rather than blocking past its deadline. Frontends MAY override the default if their use case demands. |
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "stata-code"
|
|
7
|
-
version = "0.12.
|
|
7
|
+
version = "0.12.2"
|
|
8
8
|
description = "Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)"
|
|
9
9
|
# The repo default README (README.md) is Chinese; keep the PyPI long description
|
|
10
10
|
# English by pointing at README.en.md.
|
|
@@ -9,8 +9,18 @@ parse from log prose:
|
|
|
9
9
|
exactly what Stata printed, so we copy them verbatim.
|
|
10
10
|
* ``e(b)`` / ``e(V)`` — the coefficient vector and its variance–covariance
|
|
11
11
|
matrix. Always present after an estimation command. We use these as a
|
|
12
|
-
fallback, computing ``se`` from ``diag(V)`` and
|
|
13
|
-
|
|
12
|
+
fallback, computing ``se`` from ``diag(V)`` and the statistic / p-value / CI
|
|
13
|
+
ourselves (flagged via ``source="e_b_v"``).
|
|
14
|
+
|
|
15
|
+
The fallback matters more than "fallback" suggests: ``r(table)`` is wiped by
|
|
16
|
+
the *next* command, so any block that ends in ``graph export`` / ``esttab`` /
|
|
17
|
+
``summarize`` after its estimation lands here. It therefore has to reproduce
|
|
18
|
+
what Stata displayed, not merely something defensible. Stata's own rule (see
|
|
19
|
+
``_coef_table``) is distributional: when ``e(df_r)`` is set the table is a
|
|
20
|
+
*t* table on that many residual degrees of freedom, and otherwise a *z* table.
|
|
21
|
+
We follow exactly that rule, so the fallback CI agrees with the printed log
|
|
22
|
+
digit for digit instead of quietly reporting a normal-approximation interval
|
|
23
|
+
alongside a log that says ``P>|t|``.
|
|
14
24
|
|
|
15
25
|
The result is :class:`EstimationResult`, attached to ``RunResult.results.
|
|
16
26
|
estimation`` so every frontend (MCP, kernel, VS Code) gets the same typed
|
|
@@ -22,6 +32,7 @@ from __future__ import annotations
|
|
|
22
32
|
|
|
23
33
|
import math
|
|
24
34
|
from collections.abc import Callable
|
|
35
|
+
from functools import lru_cache
|
|
25
36
|
|
|
26
37
|
from stata_code.core.schema import (
|
|
27
38
|
Coefficient,
|
|
@@ -162,6 +173,111 @@ def _two_sided_normal_p(z: float) -> float:
|
|
|
162
173
|
return 2.0 * (1.0 - _normal_cdf(abs(z)))
|
|
163
174
|
|
|
164
175
|
|
|
176
|
+
def _betacf(a: float, b: float, x: float) -> float:
|
|
177
|
+
"""Continued fraction for the incomplete beta function (Lentz's method)."""
|
|
178
|
+
tiny = 1e-30
|
|
179
|
+
qab, qap, qam = a + b, a + 1.0, a - 1.0
|
|
180
|
+
c = 1.0
|
|
181
|
+
d = 1.0 - qab * x / qap
|
|
182
|
+
if abs(d) < tiny:
|
|
183
|
+
d = tiny
|
|
184
|
+
d = 1.0 / d
|
|
185
|
+
h = d
|
|
186
|
+
for m in range(1, 301):
|
|
187
|
+
m2 = 2 * m
|
|
188
|
+
aa = m * (b - m) * x / ((qam + m2) * (a + m2))
|
|
189
|
+
d = 1.0 + aa * d
|
|
190
|
+
if abs(d) < tiny:
|
|
191
|
+
d = tiny
|
|
192
|
+
c = 1.0 + aa / c
|
|
193
|
+
if abs(c) < tiny:
|
|
194
|
+
c = tiny
|
|
195
|
+
d = 1.0 / d
|
|
196
|
+
h *= d * c
|
|
197
|
+
aa = -(a + m) * (qab + m) * x / ((a + m2) * (qap + m2))
|
|
198
|
+
d = 1.0 + aa * d
|
|
199
|
+
if abs(d) < tiny:
|
|
200
|
+
d = tiny
|
|
201
|
+
c = 1.0 + aa / c
|
|
202
|
+
if abs(c) < tiny:
|
|
203
|
+
c = tiny
|
|
204
|
+
d = 1.0 / d
|
|
205
|
+
delta = d * c
|
|
206
|
+
h *= delta
|
|
207
|
+
if abs(delta - 1.0) < 1e-15:
|
|
208
|
+
break
|
|
209
|
+
return h
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _betainc_reg(a: float, b: float, x: float) -> float:
|
|
213
|
+
"""Regularized incomplete beta ``I_x(a, b)``, stdlib only."""
|
|
214
|
+
if x <= 0.0:
|
|
215
|
+
return 0.0
|
|
216
|
+
if x >= 1.0:
|
|
217
|
+
return 1.0
|
|
218
|
+
ln_front = (
|
|
219
|
+
math.lgamma(a + b) - math.lgamma(a) - math.lgamma(b) + a * math.log(x) + b * math.log1p(-x)
|
|
220
|
+
)
|
|
221
|
+
front = math.exp(ln_front)
|
|
222
|
+
if x < (a + 1.0) / (a + b + 2.0):
|
|
223
|
+
return front * _betacf(a, b, x) / a
|
|
224
|
+
return 1.0 - front * _betacf(b, a, 1.0 - x) / b
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _two_sided_t_p(t: float, df: float) -> float:
|
|
228
|
+
"""Two-sided p-value for a t statistic on ``df`` degrees of freedom.
|
|
229
|
+
|
|
230
|
+
``P(|T| > |t|) = I_{df/(df+t²)}(df/2, 1/2)`` — exact, no table lookup.
|
|
231
|
+
"""
|
|
232
|
+
if df <= 0.0:
|
|
233
|
+
return _two_sided_normal_p(t)
|
|
234
|
+
if math.isinf(t):
|
|
235
|
+
return 0.0
|
|
236
|
+
return _betainc_reg(df / 2.0, 0.5, df / (df + t * t))
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
@lru_cache(maxsize=256)
|
|
240
|
+
def _t_crit(df: float, level: float) -> float:
|
|
241
|
+
"""Two-sided critical value of the t distribution for a CI ``level`` in %.
|
|
242
|
+
|
|
243
|
+
Found by bisection on :func:`_two_sided_t_p`, which is strictly decreasing
|
|
244
|
+
in ``|t|`` — robust for any ``df`` without shipping a quantile table.
|
|
245
|
+
"""
|
|
246
|
+
alpha = 1.0 - level / 100.0
|
|
247
|
+
if df <= 0.0 or not (0.0 < alpha < 1.0):
|
|
248
|
+
return _z_crit(level)
|
|
249
|
+
lo, hi = 0.0, 1.0
|
|
250
|
+
while _two_sided_t_p(hi, df) > alpha and hi < 1e12:
|
|
251
|
+
hi *= 2.0
|
|
252
|
+
for _ in range(200):
|
|
253
|
+
mid = 0.5 * (lo + hi)
|
|
254
|
+
if _two_sided_t_p(mid, df) > alpha:
|
|
255
|
+
lo = mid
|
|
256
|
+
else:
|
|
257
|
+
hi = mid
|
|
258
|
+
if hi - lo < 1e-13 * max(1.0, hi):
|
|
259
|
+
break
|
|
260
|
+
return 0.5 * (lo + hi)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
@lru_cache(maxsize=64)
|
|
264
|
+
def _z_crit(level: float) -> float:
|
|
265
|
+
"""Two-sided standard-normal critical value for a CI ``level`` in %."""
|
|
266
|
+
alpha = 1.0 - level / 100.0
|
|
267
|
+
if not (0.0 < alpha < 1.0):
|
|
268
|
+
return _Z_CRIT_95
|
|
269
|
+
lo, hi = 0.0, 40.0
|
|
270
|
+
for _ in range(200):
|
|
271
|
+
mid = 0.5 * (lo + hi)
|
|
272
|
+
if _two_sided_normal_p(mid) > alpha:
|
|
273
|
+
lo = mid
|
|
274
|
+
else:
|
|
275
|
+
hi = mid
|
|
276
|
+
if hi - lo < 1e-14 * max(1.0, hi):
|
|
277
|
+
break
|
|
278
|
+
return 0.5 * (lo + hi)
|
|
279
|
+
|
|
280
|
+
|
|
165
281
|
def _cell(row: list[float | None] | None, j: int) -> float | None:
|
|
166
282
|
if row is None or j >= len(row):
|
|
167
283
|
return None
|
|
@@ -284,13 +400,16 @@ def _from_b_v(
|
|
|
284
400
|
v: Matrix | None,
|
|
285
401
|
resolve: MatrixResolver | None = None,
|
|
286
402
|
b_values: list[list[float | None]] | None = None,
|
|
287
|
-
|
|
403
|
+
df_resid: float | None = None,
|
|
404
|
+
level: float = 95.0,
|
|
405
|
+
) -> tuple[list[Coefficient], str]:
|
|
288
406
|
"""Compute coefficient rows from e(b) (and e(V) when available).
|
|
289
407
|
|
|
290
408
|
``se``/``statistic``/``p_value``/CI are filled only when e(V) is
|
|
291
409
|
obtainable — inline or through its ref; otherwise just the point estimates
|
|
292
|
-
are returned.
|
|
293
|
-
|
|
410
|
+
are returned. Returns ``(coefficients, statistic_kind)``: a *t* table on
|
|
411
|
+
``df_resid`` degrees of freedom when Stata set ``e(df_r)``, matching what
|
|
412
|
+
the command printed, and a *z* table otherwise.
|
|
294
413
|
"""
|
|
295
414
|
if b_values is None:
|
|
296
415
|
b_values = _values(b, resolve)
|
|
@@ -305,6 +424,10 @@ def _from_b_v(
|
|
|
305
424
|
else:
|
|
306
425
|
v_diag = [None] * len(terms)
|
|
307
426
|
|
|
427
|
+
df = float(df_resid) if df_resid is not None and df_resid > 0.0 else None
|
|
428
|
+
stat_kind = "t" if df is not None else "z"
|
|
429
|
+
crit = _t_crit(df, level) if df is not None else _z_crit(level)
|
|
430
|
+
|
|
308
431
|
coeffs: list[Coefficient] = []
|
|
309
432
|
for j, term in enumerate(terms):
|
|
310
433
|
b_val = _cell(b_row, j)
|
|
@@ -316,9 +439,9 @@ def _from_b_v(
|
|
|
316
439
|
ci_high: float | None = None
|
|
317
440
|
if b_val is not None and se is not None and se > 0.0:
|
|
318
441
|
stat = b_val / se
|
|
319
|
-
p_val = _two_sided_normal_p(stat)
|
|
320
|
-
ci_low = b_val -
|
|
321
|
-
ci_high = b_val +
|
|
442
|
+
p_val = _two_sided_t_p(stat, df) if df is not None else _two_sided_normal_p(stat)
|
|
443
|
+
ci_low = b_val - crit * se
|
|
444
|
+
ci_high = b_val + crit * se
|
|
322
445
|
coeffs.append(
|
|
323
446
|
Coefficient(
|
|
324
447
|
term=term,
|
|
@@ -330,7 +453,7 @@ def _from_b_v(
|
|
|
330
453
|
ci_high=ci_high,
|
|
331
454
|
)
|
|
332
455
|
)
|
|
333
|
-
return coeffs
|
|
456
|
+
return coeffs, stat_kind
|
|
334
457
|
|
|
335
458
|
|
|
336
459
|
def _model_stats(scalars: dict[str, float | None]) -> dict[str, float | None]:
|
|
@@ -406,14 +529,30 @@ def build_estimation_from_returns(
|
|
|
406
529
|
statistic_kind: str
|
|
407
530
|
source: str
|
|
408
531
|
|
|
532
|
+
df_resid = e.scalars.get("df_r")
|
|
533
|
+
# e(level) is the CI level the command actually displayed; absent means the
|
|
534
|
+
# Stata default of 95. Both the critical value and the reported ci_level
|
|
535
|
+
# follow it, so `regress, level(90)` does not come back labelled 95.
|
|
536
|
+
level_raw = e.scalars.get("level")
|
|
537
|
+
ci_level = float(level_raw) if level_raw is not None else 95.0
|
|
538
|
+
|
|
409
539
|
table = r.matrices.get("table")
|
|
410
540
|
parsed = _from_r_table(table, resolve_matrix) if table is not None else None
|
|
411
541
|
if parsed is not None and table is not None and _r_table_matches_b(table, b, resolve_matrix):
|
|
412
542
|
coeffs, statistic_kind = parsed
|
|
413
543
|
source = "r_table"
|
|
414
544
|
else:
|
|
415
|
-
|
|
416
|
-
|
|
545
|
+
# r(table) is gone — cleared by whatever command ran after the
|
|
546
|
+
# estimation. Rebuild the table from e(b)/e(V) on the same
|
|
547
|
+
# distribution Stata used, so the numbers still match the log.
|
|
548
|
+
coeffs, statistic_kind = _from_b_v(
|
|
549
|
+
b,
|
|
550
|
+
v,
|
|
551
|
+
resolve_matrix,
|
|
552
|
+
b_values,
|
|
553
|
+
df_resid=float(df_resid) if df_resid is not None else None,
|
|
554
|
+
level=ci_level,
|
|
555
|
+
)
|
|
417
556
|
source = "e_b_v"
|
|
418
557
|
|
|
419
558
|
n_obs: int | None = None
|
|
@@ -430,9 +569,10 @@ def build_estimation_from_returns(
|
|
|
430
569
|
depvar=depvar,
|
|
431
570
|
n_obs=n_obs,
|
|
432
571
|
df_model=e.scalars.get("df_m"),
|
|
433
|
-
df_resid=
|
|
572
|
+
df_resid=df_resid,
|
|
434
573
|
statistic_kind=statistic_kind, # type: ignore[arg-type]
|
|
435
574
|
source=source, # type: ignore[arg-type]
|
|
575
|
+
ci_level=ci_level,
|
|
436
576
|
coefficients=coeffs,
|
|
437
577
|
model_stats=_model_stats(e.scalars),
|
|
438
578
|
diagnostics=_command_diagnostics(command, e.scalars),
|
|
@@ -60,6 +60,7 @@ class Job:
|
|
|
60
60
|
"error",
|
|
61
61
|
"_done",
|
|
62
62
|
"_started",
|
|
63
|
+
"_elapsed_ms",
|
|
63
64
|
)
|
|
64
65
|
|
|
65
66
|
def __init__(self, job_id: str, session_id: str, code: str) -> None:
|
|
@@ -73,6 +74,7 @@ class Job:
|
|
|
73
74
|
self.error: str | None = None
|
|
74
75
|
self._done = threading.Event()
|
|
75
76
|
self._started = time.monotonic()
|
|
77
|
+
self._elapsed_ms: int | None = None
|
|
76
78
|
|
|
77
79
|
@property
|
|
78
80
|
def done(self) -> bool:
|
|
@@ -83,6 +85,15 @@ class Job:
|
|
|
83
85
|
return self._done.wait(timeout=timeout_s)
|
|
84
86
|
|
|
85
87
|
def elapsed_ms(self) -> int:
|
|
88
|
+
"""How long the run took, or has been running so far.
|
|
89
|
+
|
|
90
|
+
Frozen once the job reaches a terminal state. Recomputing it on every
|
|
91
|
+
poll made a finished job's reported duration grow without bound — a
|
|
92
|
+
1.7s bootstrap polled ten minutes later reported ten minutes — so
|
|
93
|
+
callers attributing wall time to Stata were reading their own latency.
|
|
94
|
+
"""
|
|
95
|
+
if self._elapsed_ms is not None:
|
|
96
|
+
return self._elapsed_ms
|
|
86
97
|
return max(0, int((time.monotonic() - self._started) * 1000))
|
|
87
98
|
|
|
88
99
|
def summary(self) -> dict[str, Any]:
|
|
@@ -140,6 +151,11 @@ class JobRegistry:
|
|
|
140
151
|
# sees a terminal status is guaranteed to see the payload with it.
|
|
141
152
|
job.result = result
|
|
142
153
|
job.error = error
|
|
154
|
+
# Freeze the clock before publishing a terminal status, so no
|
|
155
|
+
# reader can observe "done" alongside a still-advancing duration.
|
|
156
|
+
job._elapsed_ms = max( # noqa: SLF001 - same-module private handshake
|
|
157
|
+
0, int((time.monotonic() - job._started) * 1000) # noqa: SLF001
|
|
158
|
+
)
|
|
143
159
|
job.finished_at = _utc_iso_ms()
|
|
144
160
|
job.status = status
|
|
145
161
|
job._done.set() # noqa: SLF001 - same-module private handshake
|
|
@@ -140,6 +140,13 @@ _DATASET_VAR_CAP = 200
|
|
|
140
140
|
# a matrix would inline more than ~10,000 cells."
|
|
141
141
|
MATRIX_INLINE_CELL_CAP = 10_000
|
|
142
142
|
|
|
143
|
+
# Longest macro value put on the wire verbatim. Stata macros are mostly short
|
|
144
|
+
# (`e(cmd)`, `e(depvar)`, `e(vcetype)`), but a few are enormous and carry
|
|
145
|
+
# nothing an agent can act on — `e(rngstate)` alone is ~2 KB of hex on every
|
|
146
|
+
# bootstrap / permute / simulate run. Anything longer is elided with an
|
|
147
|
+
# explicit marker rather than dropped, so the name still shows up.
|
|
148
|
+
MACRO_INLINE_CHAR_CAP = 256
|
|
149
|
+
|
|
143
150
|
# Stata's system missing `.` is exactly maxdouble (2**1023); the 26 extended
|
|
144
151
|
# missings `.a`–`.z` occupy the next representable doubles above it. `sfi`
|
|
145
152
|
# hands all of them back as ordinary Python floats, so any value at or above
|
|
@@ -659,7 +666,23 @@ def _project_returns(
|
|
|
659
666
|
n_rows=n_rows,
|
|
660
667
|
n_cols=n_cols,
|
|
661
668
|
)
|
|
662
|
-
return StataReturns(
|
|
669
|
+
return StataReturns(
|
|
670
|
+
scalars=rv.scalars,
|
|
671
|
+
macros={name: _cap_macro(v) for name, v in rv.macros.items()},
|
|
672
|
+
matrices=matrices,
|
|
673
|
+
)
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
def _cap_macro(value: str) -> str:
|
|
677
|
+
"""Elide a macro value too long to be worth its tokens on the wire.
|
|
678
|
+
|
|
679
|
+
Applied only at projection time — the estimation contract is built from the
|
|
680
|
+
uncapped values, so nothing downstream loses precision.
|
|
681
|
+
"""
|
|
682
|
+
if len(value) <= MACRO_INLINE_CHAR_CAP:
|
|
683
|
+
return value
|
|
684
|
+
dropped = len(value) - MACRO_INLINE_CHAR_CAP
|
|
685
|
+
return f"{value[:MACRO_INLINE_CHAR_CAP]}… ({dropped} more chars elided)"
|
|
663
686
|
|
|
664
687
|
|
|
665
688
|
def _collect_dataset(rt: Any, include_variables: bool) -> DatasetInfo:
|
|
@@ -1329,6 +1352,17 @@ def execute(
|
|
|
1329
1352
|
# Snapshot existing graph names before user code so we can take a delta
|
|
1330
1353
|
# afterward.
|
|
1331
1354
|
pre_graphs = _list_graph_names(rt) if include_graphs != "none" else []
|
|
1355
|
+
|
|
1356
|
+
# e() is session-global and outlives the run that set it, so
|
|
1357
|
+
# `results.estimation` may describe an estimation from an earlier call.
|
|
1358
|
+
# Fingerprint it now to tell "this run estimated something" from "this
|
|
1359
|
+
# run inherited an estimation" — the difference decides whether an
|
|
1360
|
+
# e(b)/e(V) rebuild is worth reporting.
|
|
1361
|
+
pre_estimation = (
|
|
1362
|
+
_estimation_fingerprint(rt)
|
|
1363
|
+
if estimation_mode != IncludeEstimation.NONE
|
|
1364
|
+
else None
|
|
1365
|
+
)
|
|
1332
1366
|
graph_source_hints, unnamed_graph_source_hints = (
|
|
1333
1367
|
_graph_source_hints(code) if include_graphs != "none" else ({}, [])
|
|
1334
1368
|
)
|
|
@@ -1437,6 +1471,34 @@ def execute(
|
|
|
1437
1471
|
ok = error is None and rc == 0
|
|
1438
1472
|
warnings = _extract_warnings(log_text)
|
|
1439
1473
|
|
|
1474
|
+
# r(table) is cleared by the next command, so a block that runs anything
|
|
1475
|
+
# after its estimation gets a table rebuilt from e(b)/e(V). The numbers
|
|
1476
|
+
# follow Stata's own t-vs-z rule and so still match the log, but the
|
|
1477
|
+
# provenance is worth stating: a caller comparing against `r(table)` rows
|
|
1478
|
+
# it did not capture should know which path produced these.
|
|
1479
|
+
#
|
|
1480
|
+
# Gated on the estimation being *this run's*. e() outlives the call that
|
|
1481
|
+
# set it, so every later `summarize` in the session re-reports the same
|
|
1482
|
+
# inherited table through the same fallback — warning on those would be
|
|
1483
|
+
# pure noise about a rebuild the caller did not ask for and cannot act on.
|
|
1484
|
+
estimated_here = (
|
|
1485
|
+
pre_estimation is not None and _estimation_fingerprint(rt) != pre_estimation
|
|
1486
|
+
)
|
|
1487
|
+
if estimated_here and estimation is not None and estimation.source == "e_b_v":
|
|
1488
|
+
warnings.append(
|
|
1489
|
+
StataWarning(
|
|
1490
|
+
kind="estimation_from_e_b_v",
|
|
1491
|
+
message=(
|
|
1492
|
+
"results.estimation was rebuilt from e(b)/e(V) because r(table) "
|
|
1493
|
+
"no longer described the current estimation — a later command "
|
|
1494
|
+
"cleared it. Inference uses "
|
|
1495
|
+
f"{estimation.statistic_kind!r} as Stata would "
|
|
1496
|
+
f"(df_r={estimation.df_resid!r}). Put the estimation last in the "
|
|
1497
|
+
"block, or re-run it alone, to get the r(table) values verbatim."
|
|
1498
|
+
),
|
|
1499
|
+
)
|
|
1500
|
+
)
|
|
1501
|
+
|
|
1440
1502
|
# A failed run may have left `log using` handles open. Close only the ones
|
|
1441
1503
|
# this run opened: a log the caller opened in an earlier successful run is
|
|
1442
1504
|
# theirs to manage and must survive.
|
|
@@ -1599,6 +1661,14 @@ def _collect_results(
|
|
|
1599
1661
|
# With results suppressed but estimation wanted, read only the matrices the
|
|
1600
1662
|
# contract is derived from rather than every matrix in scope.
|
|
1601
1663
|
named = results_mode != IncludeResults.NONE
|
|
1664
|
+
# e() scalars and macros are inputs to the estimation contract, not just
|
|
1665
|
+
# payload: n_obs, df_model, df_resid, model_stats, the t-vs-z choice and
|
|
1666
|
+
# depvar all come from them. Suppressing them for `include_results="none"`
|
|
1667
|
+
# silently hollowed out `estimation` — a caller who asked only to stop
|
|
1668
|
+
# *echoing* r()/e() lost the model-level numbers that `include_estimation`
|
|
1669
|
+
# is the knob for. Read them whenever estimation is wanted; the wire
|
|
1670
|
+
# projection below still drops them when the caller said "none".
|
|
1671
|
+
e_named = named or want_estimation
|
|
1602
1672
|
raw = ResultsInfo(
|
|
1603
1673
|
r=_collect_returns(
|
|
1604
1674
|
rt,
|
|
@@ -1609,7 +1679,7 @@ def _collect_results(
|
|
|
1609
1679
|
e=_collect_returns(
|
|
1610
1680
|
rt,
|
|
1611
1681
|
"e",
|
|
1612
|
-
want_named=
|
|
1682
|
+
want_named=e_named,
|
|
1613
1683
|
matrix_names=None if named else _ESTIMATION_MATRICES["e"],
|
|
1614
1684
|
),
|
|
1615
1685
|
last_estimation_cmd=_last_estimation_cmd(rt),
|
|
@@ -1655,6 +1725,27 @@ def _last_estimation_cmd(rt: Any) -> str | None:
|
|
|
1655
1725
|
return None
|
|
1656
1726
|
|
|
1657
1727
|
|
|
1728
|
+
def _estimation_fingerprint(rt: Any) -> tuple[str | None, str | None, float | None]:
|
|
1729
|
+
"""Cheap identity for the estimation currently in e-scope.
|
|
1730
|
+
|
|
1731
|
+
Three reads, no matrices: enough to tell one estimation from the next
|
|
1732
|
+
without paying to fetch e(b). Used only to compare before/after a run.
|
|
1733
|
+
"""
|
|
1734
|
+
sfi = rt.sfi
|
|
1735
|
+
|
|
1736
|
+
def _macro(name: str) -> str | None:
|
|
1737
|
+
try:
|
|
1738
|
+
return sfi.Macro.getGlobal(name) or None
|
|
1739
|
+
except Exception: # noqa: BLE001
|
|
1740
|
+
return None
|
|
1741
|
+
|
|
1742
|
+
try:
|
|
1743
|
+
n = sfi.Scalar.getValue("e(N)")
|
|
1744
|
+
except Exception: # noqa: BLE001
|
|
1745
|
+
n = None
|
|
1746
|
+
return (_macro("e(cmd)"), _macro("e(cmdline)"), _norm_stata_number(n))
|
|
1747
|
+
|
|
1748
|
+
|
|
1658
1749
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
1659
1750
|
# Multi-session via Stata frames (Module 4)
|
|
1660
1751
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
@@ -102,7 +102,7 @@ from stata_code.core.runner import (
|
|
|
102
102
|
)
|
|
103
103
|
from stata_code.core.schema import RunResult
|
|
104
104
|
|
|
105
|
-
__version__ = "0.12.
|
|
105
|
+
__version__ = "0.12.2"
|
|
106
106
|
|
|
107
107
|
SERVER_INSTRUCTIONS = (
|
|
108
108
|
"Use stata-code for running and inspecting Stata code. Prefer structuredContent "
|
|
@@ -374,6 +374,49 @@ class TestJobRegistry:
|
|
|
374
374
|
assert job.result == "RESULT"
|
|
375
375
|
assert job.finished_at is not None
|
|
376
376
|
|
|
377
|
+
def test_elapsed_ms_freezes_when_the_job_finishes(self, monkeypatch):
|
|
378
|
+
# Recomputing elapsed on every poll made a finished job's duration
|
|
379
|
+
# grow with the caller's polling delay, so a 2s run polled a minute
|
|
380
|
+
# later reported a minute of Stata time.
|
|
381
|
+
from stata_code.core import jobs
|
|
382
|
+
|
|
383
|
+
monkeypatch.setattr(jobs, "pool_execute", lambda code, **kw: "RESULT") # noqa: ARG005
|
|
384
|
+
registry = jobs.JobRegistry()
|
|
385
|
+
job = registry.submit("display 1")
|
|
386
|
+
assert job.wait(5) is True
|
|
387
|
+
|
|
388
|
+
first = job.elapsed_ms()
|
|
389
|
+
time.sleep(0.15)
|
|
390
|
+
assert job.elapsed_ms() == first
|
|
391
|
+
assert job.summary()["elapsed_ms"] == first
|
|
392
|
+
|
|
393
|
+
def test_elapsed_ms_still_advances_while_running(self, monkeypatch):
|
|
394
|
+
from stata_code.core import jobs
|
|
395
|
+
|
|
396
|
+
gate = threading.Event()
|
|
397
|
+
monkeypatch.setattr(jobs, "pool_execute", lambda code, **kw: gate.wait(10)) # noqa: ARG005
|
|
398
|
+
registry = jobs.JobRegistry()
|
|
399
|
+
job = registry.submit("display 1")
|
|
400
|
+
first = job.elapsed_ms()
|
|
401
|
+
time.sleep(0.05)
|
|
402
|
+
assert job.elapsed_ms() >= first
|
|
403
|
+
gate.set()
|
|
404
|
+
job.wait(5)
|
|
405
|
+
|
|
406
|
+
def test_elapsed_ms_is_frozen_on_the_error_path_too(self, monkeypatch):
|
|
407
|
+
from stata_code.core import jobs
|
|
408
|
+
|
|
409
|
+
def _boom(code, **kwargs): # noqa: ARG001
|
|
410
|
+
raise ValueError("bad option")
|
|
411
|
+
|
|
412
|
+
monkeypatch.setattr(jobs, "pool_execute", _boom)
|
|
413
|
+
registry = jobs.JobRegistry()
|
|
414
|
+
job = registry.submit("display 1")
|
|
415
|
+
assert job.wait(5) is True
|
|
416
|
+
first = job.elapsed_ms()
|
|
417
|
+
time.sleep(0.15)
|
|
418
|
+
assert job.elapsed_ms() == first
|
|
419
|
+
|
|
377
420
|
def test_failure_is_recorded_not_swallowed(self, monkeypatch):
|
|
378
421
|
from stata_code.core import jobs
|
|
379
422
|
|
|
@@ -2,14 +2,20 @@
|
|
|
2
2
|
|
|
3
3
|
Covers both extraction paths:
|
|
4
4
|
* ``r(table)`` (referee-grade — values copied exactly as Stata displayed them)
|
|
5
|
-
* ``e(b)`` / ``e(V)`` fallback (se/
|
|
5
|
+
* ``e(b)`` / ``e(V)`` fallback (se / statistic / p / CI computed here, on the
|
|
6
|
+
same distribution Stata used: t when ``e(df_r)`` is set, z otherwise)
|
|
6
7
|
"""
|
|
7
8
|
|
|
8
9
|
from __future__ import annotations
|
|
9
10
|
|
|
10
11
|
import math
|
|
11
12
|
|
|
13
|
+
import pytest
|
|
14
|
+
|
|
12
15
|
from stata_code.core.estimation import (
|
|
16
|
+
_t_crit,
|
|
17
|
+
_two_sided_t_p,
|
|
18
|
+
_z_crit,
|
|
13
19
|
build_estimation_from_returns,
|
|
14
20
|
build_estimation_result,
|
|
15
21
|
)
|
|
@@ -98,7 +104,56 @@ class TestRTablePath:
|
|
|
98
104
|
|
|
99
105
|
|
|
100
106
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
101
|
-
#
|
|
107
|
+
# Distribution helpers — checked against published t-table values
|
|
108
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class TestDistributions:
|
|
112
|
+
@pytest.mark.parametrize(
|
|
113
|
+
("df", "expected"),
|
|
114
|
+
[
|
|
115
|
+
(1, 12.706204736),
|
|
116
|
+
(5, 2.570581836),
|
|
117
|
+
(10, 2.228138852),
|
|
118
|
+
(30, 2.042272456),
|
|
119
|
+
(70, 1.994437112),
|
|
120
|
+
(120, 1.979930405),
|
|
121
|
+
(100000, 1.959987707),
|
|
122
|
+
],
|
|
123
|
+
)
|
|
124
|
+
def test_t_crit_matches_published_two_sided_95pct_values(self, df, expected):
|
|
125
|
+
assert math.isclose(_t_crit(float(df), 95.0), expected, rel_tol=1e-8)
|
|
126
|
+
|
|
127
|
+
def test_t_crit_converges_to_the_normal_as_df_grows(self):
|
|
128
|
+
assert math.isclose(_t_crit(1e9, 95.0), _z_crit(95.0), rel_tol=1e-6)
|
|
129
|
+
|
|
130
|
+
@pytest.mark.parametrize(
|
|
131
|
+
("t", "df", "expected"),
|
|
132
|
+
[
|
|
133
|
+
(2.042272456, 30, 0.05),
|
|
134
|
+
(1.0, 10, 0.340893),
|
|
135
|
+
(0.29443916691636146, 70, 0.769293740812962),
|
|
136
|
+
(5.493002930280353, 70, 5.991178e-7),
|
|
137
|
+
],
|
|
138
|
+
)
|
|
139
|
+
def test_two_sided_t_p_matches_reference_values(self, t, df, expected):
|
|
140
|
+
assert math.isclose(_two_sided_t_p(t, float(df)), expected, rel_tol=1e-5)
|
|
141
|
+
|
|
142
|
+
def test_t_p_is_symmetric_and_bounded(self):
|
|
143
|
+
for t in (0.0, 0.5, 3.0, 40.0):
|
|
144
|
+
p = _two_sided_t_p(t, 12.0)
|
|
145
|
+
assert 0.0 <= p <= 1.0
|
|
146
|
+
assert math.isclose(p, _two_sided_t_p(-t, 12.0), rel_tol=1e-15)
|
|
147
|
+
assert math.isclose(_two_sided_t_p(0.0, 12.0), 1.0, rel_tol=1e-12)
|
|
148
|
+
|
|
149
|
+
def test_z_crit_matches_known_levels(self):
|
|
150
|
+
assert math.isclose(_z_crit(95.0), 1.959963984540054, rel_tol=1e-12)
|
|
151
|
+
assert math.isclose(_z_crit(99.0), 2.5758293035489004, rel_tol=1e-10)
|
|
152
|
+
assert math.isclose(_z_crit(90.0), 1.6448536269514722, rel_tol=1e-10)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
156
|
+
# e(b)/e(V) fallback — computed se / statistic / p / CI
|
|
102
157
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
103
158
|
|
|
104
159
|
|
|
@@ -128,6 +183,63 @@ class TestBVFallback:
|
|
|
128
183
|
assert math.isclose(c.ci_low, 10.0 - 1.959963984540054 * 2.0, rel_tol=1e-12)
|
|
129
184
|
assert math.isclose(c.ci_high, 10.0 + 1.959963984540054 * 2.0, rel_tol=1e-12)
|
|
130
185
|
|
|
186
|
+
def test_df_r_switches_the_fallback_to_a_t_table(self):
|
|
187
|
+
# Stata prints a t table whenever e(df_r) is set, so the fallback must
|
|
188
|
+
# too — otherwise the rebuilt CI silently disagrees with the log.
|
|
189
|
+
cols = ["x"]
|
|
190
|
+
e = StataReturns(
|
|
191
|
+
scalars={"df_r": 70},
|
|
192
|
+
matrices={"b": _b([10.0], cols), "V": _v([4.0], cols)},
|
|
193
|
+
)
|
|
194
|
+
est = build_estimation_from_returns(e, StataReturns())
|
|
195
|
+
assert est.source == "e_b_v"
|
|
196
|
+
assert est.statistic_kind == "t"
|
|
197
|
+
# t(.975, 70) = 1.9944371... — wider than the 1.95996 normal interval.
|
|
198
|
+
c = est.coefficients[0]
|
|
199
|
+
assert math.isclose(c.ci_low, 10.0 - 1.9944371112999 * 2.0, rel_tol=1e-9)
|
|
200
|
+
assert math.isclose(c.ci_high, 10.0 + 1.9944371112999 * 2.0, rel_tol=1e-9)
|
|
201
|
+
|
|
202
|
+
def test_fallback_reproduces_statas_printed_regress_table(self):
|
|
203
|
+
# Values taken from a live `regress price_k mpg weight foreign` on
|
|
204
|
+
# sysuse auto (N=74, df_r=70). Before the t fix this path returned
|
|
205
|
+
# [-0.12362, 0.16732] against a log that printed [-.1261758, .169883].
|
|
206
|
+
cols = ["mpg"]
|
|
207
|
+
b, se = 0.02185360997213235, 0.07422113776846848
|
|
208
|
+
e = StataReturns(
|
|
209
|
+
scalars={"N": 74, "df_m": 3, "df_r": 70},
|
|
210
|
+
macros={"cmd": "regress", "depvar": "price_k"},
|
|
211
|
+
matrices={"b": _b([b], cols), "V": _v([se * se], cols)},
|
|
212
|
+
)
|
|
213
|
+
est = build_estimation_from_returns(e, StataReturns())
|
|
214
|
+
c = est.coefficients[0]
|
|
215
|
+
assert math.isclose(c.ci_low, -0.1261757816711833, rel_tol=1e-9)
|
|
216
|
+
assert math.isclose(c.ci_high, 0.16988300161544798, rel_tol=1e-9)
|
|
217
|
+
assert math.isclose(c.p_value, 0.769293740812962, rel_tol=1e-9)
|
|
218
|
+
|
|
219
|
+
def test_no_df_r_stays_on_the_normal_approximation(self):
|
|
220
|
+
# Commands that do not set e(df_r) (ml-family, bootstrap, ...) print a
|
|
221
|
+
# z table; the fallback must not invent residual degrees of freedom.
|
|
222
|
+
cols = ["x"]
|
|
223
|
+
e = StataReturns(
|
|
224
|
+
scalars={"N": 500},
|
|
225
|
+
macros={"cmd": "logit"},
|
|
226
|
+
matrices={"b": _b([10.0], cols), "V": _v([4.0], cols)},
|
|
227
|
+
)
|
|
228
|
+
est = build_estimation_from_returns(e, StataReturns())
|
|
229
|
+
assert est.statistic_kind == "z"
|
|
230
|
+
assert math.isclose(est.coefficients[0].ci_high, 10.0 + 1.959963984540054 * 2.0)
|
|
231
|
+
|
|
232
|
+
def test_e_level_drives_both_ci_level_and_the_critical_value(self):
|
|
233
|
+
cols = ["x"]
|
|
234
|
+
e = StataReturns(
|
|
235
|
+
scalars={"df_r": 70, "level": 90},
|
|
236
|
+
matrices={"b": _b([10.0], cols), "V": _v([4.0], cols)},
|
|
237
|
+
)
|
|
238
|
+
est = build_estimation_from_returns(e, StataReturns())
|
|
239
|
+
assert est.ci_level == 90.0
|
|
240
|
+
# t(.95, 70) = 1.6669145... — narrower than the 95% interval.
|
|
241
|
+
assert math.isclose(est.coefficients[0].ci_high, 10.0 + 1.6669145 * 2.0, rel_tol=1e-7)
|
|
242
|
+
|
|
131
243
|
def test_no_vcov_yields_point_estimates_only(self):
|
|
132
244
|
cols = ["x"]
|
|
133
245
|
e = StataReturns(matrices={"b": _b([3.0], cols)}) # no V
|
|
@@ -123,6 +123,80 @@ class TestEstimationContractReal:
|
|
|
123
123
|
assert r.ok, r.error
|
|
124
124
|
assert r.results.estimation is None
|
|
125
125
|
|
|
126
|
+
def test_fallback_table_agrees_with_r_table_to_the_last_digit(self):
|
|
127
|
+
# r(table) is cleared by whatever runs after the estimation, so a block
|
|
128
|
+
# ending in `graph export` / `esttab` / `summarize` lands on the
|
|
129
|
+
# e(b)/e(V) path. Both paths must describe the same regression: this
|
|
130
|
+
# once returned normal-approximation intervals against a log printing
|
|
131
|
+
# `P>|t|`, so structuredContent silently disagreed with the log.
|
|
132
|
+
setup = "sysuse auto, clear\ngen price_k = price/1000\n"
|
|
133
|
+
clean = _run(setup + "regress price_k mpg weight foreign", "rs_rt")
|
|
134
|
+
stale = _run(setup + "regress price_k mpg weight foreign\nsummarize mpg", "rs_bv")
|
|
135
|
+
assert clean.ok and stale.ok, (clean.error, stale.error)
|
|
136
|
+
|
|
137
|
+
assert clean.results.estimation.source == "r_table"
|
|
138
|
+
assert stale.results.estimation.source == "e_b_v"
|
|
139
|
+
assert stale.results.estimation.statistic_kind == "t"
|
|
140
|
+
assert stale.results.estimation.df_resid == 70
|
|
141
|
+
|
|
142
|
+
by_term = {c.term: c for c in clean.results.estimation.coefficients}
|
|
143
|
+
for c in stale.results.estimation.coefficients:
|
|
144
|
+
ref = by_term[c.term]
|
|
145
|
+
assert c.b == pytest.approx(ref.b, rel=1e-12)
|
|
146
|
+
assert c.se == pytest.approx(ref.se, rel=1e-12)
|
|
147
|
+
assert c.statistic == pytest.approx(ref.statistic, rel=1e-9)
|
|
148
|
+
assert c.p_value == pytest.approx(ref.p_value, rel=1e-6, abs=1e-12)
|
|
149
|
+
assert c.ci_low == pytest.approx(ref.ci_low, rel=1e-9)
|
|
150
|
+
assert c.ci_high == pytest.approx(ref.ci_high, rel=1e-9)
|
|
151
|
+
|
|
152
|
+
def test_rebuilt_table_is_flagged_in_warnings(self):
|
|
153
|
+
r = _run("sysuse auto, clear\nregress price mpg\nsummarize mpg", "rs_bvwarn")
|
|
154
|
+
assert r.ok, r.error
|
|
155
|
+
assert r.results.estimation.source == "e_b_v"
|
|
156
|
+
kinds = {w.kind for w in r.warnings}
|
|
157
|
+
assert "estimation_from_e_b_v" in kinds
|
|
158
|
+
|
|
159
|
+
def test_inherited_estimation_does_not_warn_on_later_runs(self):
|
|
160
|
+
# e() is session-global, so `estimation` keeps describing this regress
|
|
161
|
+
# on every subsequent call. Those runs rebuild from e(b)/e(V) too, but
|
|
162
|
+
# they estimated nothing — warning on them would be noise.
|
|
163
|
+
first = _run("sysuse auto, clear\nregress price mpg", "rs_bvquiet")
|
|
164
|
+
assert first.ok, first.error
|
|
165
|
+
later = _run("summarize mpg", "rs_bvquiet")
|
|
166
|
+
assert later.ok, later.error
|
|
167
|
+
assert later.results.estimation is not None # inherited from e()
|
|
168
|
+
assert later.results.estimation.source == "e_b_v"
|
|
169
|
+
assert [w.kind for w in later.warnings] == []
|
|
170
|
+
|
|
171
|
+
def test_estimation_survives_include_results_none(self):
|
|
172
|
+
# include_results governs how much r()/e() is *echoed*; the model-level
|
|
173
|
+
# numbers in `estimation` are include_estimation's business. Reading
|
|
174
|
+
# e() scalars is what makes n_obs / df_resid / model_stats available.
|
|
175
|
+
r = run(
|
|
176
|
+
"sysuse auto, clear\nregress price mpg weight",
|
|
177
|
+
session_id="rs_noresults",
|
|
178
|
+
include_results="none",
|
|
179
|
+
include_full_log=False,
|
|
180
|
+
)
|
|
181
|
+
assert r.ok, r.error
|
|
182
|
+
# Nothing echoed on the wire...
|
|
183
|
+
assert r.results.e.scalars == {} and r.results.e.macros == {}
|
|
184
|
+
# ...but the contract is intact.
|
|
185
|
+
est = r.results.estimation
|
|
186
|
+
assert est is not None
|
|
187
|
+
assert est.n_obs == 74
|
|
188
|
+
assert est.df_resid == 71
|
|
189
|
+
assert est.model_stats.get("r2") == pytest.approx(0.2934, abs=1e-3)
|
|
190
|
+
assert len(est.coefficients) == 3
|
|
191
|
+
|
|
192
|
+
def test_rngstate_macro_is_elided_not_shipped_whole(self):
|
|
193
|
+
r = _run("sysuse auto, clear\nbootstrap r(mean), reps(20): summarize price", "rs_rng")
|
|
194
|
+
assert r.ok, r.error
|
|
195
|
+
rngstate = r.results.e.macros.get("rngstate")
|
|
196
|
+
assert rngstate is not None, "bootstrap should still report the macro name"
|
|
197
|
+
assert len(rngstate) < 400
|
|
198
|
+
assert "chars elided" in rngstate
|
|
199
|
+
|
|
126
200
|
|
|
127
201
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
128
202
|
# Error taxonomy — real rc codes, labels, recovery, suggestions
|
|
@@ -1040,6 +1040,34 @@ class TestProjectReturns:
|
|
|
1040
1040
|
)
|
|
1041
1041
|
assert out.scalars == {} and out.macros == {} and out.matrices == {}
|
|
1042
1042
|
|
|
1043
|
+
def test_oversized_macro_is_elided_with_a_marker(self):
|
|
1044
|
+
# e(rngstate) is ~2 KB of hex on every bootstrap / permute / simulate
|
|
1045
|
+
# run and carries nothing an agent can branch on.
|
|
1046
|
+
from stata_code.core.schema import IncludeResults, StataReturns
|
|
1047
|
+
|
|
1048
|
+
rngstate = "X" + "a1b2c3d4" * 260
|
|
1049
|
+
collected = StataReturns(macros={"cmd": "bootstrap", "rngstate": rngstate})
|
|
1050
|
+
out = runner._project_returns(
|
|
1051
|
+
collected, prefix="e", request_id="req-m", mode=IncludeResults.SCALARS
|
|
1052
|
+
)
|
|
1053
|
+
assert out.macros["cmd"] == "bootstrap" # short macros pass through
|
|
1054
|
+
capped = out.macros["rngstate"]
|
|
1055
|
+
assert len(capped) < len(rngstate)
|
|
1056
|
+
assert capped.startswith(rngstate[: runner.MACRO_INLINE_CHAR_CAP])
|
|
1057
|
+
assert "chars elided" in capped
|
|
1058
|
+
|
|
1059
|
+
def test_macro_exactly_at_the_cap_is_untouched(self):
|
|
1060
|
+
from stata_code.core.schema import IncludeResults, StataReturns
|
|
1061
|
+
|
|
1062
|
+
exact = "z" * runner.MACRO_INLINE_CHAR_CAP
|
|
1063
|
+
out = runner._project_returns(
|
|
1064
|
+
StataReturns(macros={"m": exact}),
|
|
1065
|
+
prefix="e",
|
|
1066
|
+
request_id="req-e",
|
|
1067
|
+
mode=IncludeResults.SCALARS,
|
|
1068
|
+
)
|
|
1069
|
+
assert out.macros["m"] == exact
|
|
1070
|
+
|
|
1043
1071
|
|
|
1044
1072
|
class TestCollectDataset:
|
|
1045
1073
|
def test_full_metadata_with_variables(self):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|