stata-code 0.12.1__tar.gz → 0.12.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {stata_code-0.12.1 → stata_code-0.12.2}/CHANGELOG.md +57 -0
  2. {stata_code-0.12.1 → stata_code-0.12.2}/PKG-INFO +1 -1
  3. {stata_code-0.12.1 → stata_code-0.12.2}/SCHEMA.md +16 -6
  4. {stata_code-0.12.1 → stata_code-0.12.2}/pyproject.toml +1 -1
  5. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/__init__.py +1 -1
  6. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/estimation.py +152 -12
  7. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/jobs.py +16 -0
  8. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/runner.py +93 -2
  9. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp/server.py +1 -1
  10. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_agent_ergonomics.py +43 -0
  11. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_estimation.py +114 -2
  12. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_real_stata.py +74 -0
  13. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_runner_helpers.py +28 -0
  14. {stata_code-0.12.1 → stata_code-0.12.2}/.gitignore +0 -0
  15. {stata_code-0.12.1 → stata_code-0.12.2}/LICENSE +0 -0
  16. {stata_code-0.12.1 → stata_code-0.12.2}/LICENSE-POLICY.md +0 -0
  17. {stata_code-0.12.1 → stata_code-0.12.2}/PUBLISHING.md +0 -0
  18. {stata_code-0.12.1 → stata_code-0.12.2}/README.en.md +0 -0
  19. {stata_code-0.12.1 → stata_code-0.12.2}/README.md +0 -0
  20. {stata_code-0.12.1 → stata_code-0.12.2}/docs/competitive-landscape.md +0 -0
  21. {stata_code-0.12.1 → stata_code-0.12.2}/docs/design/hard_timeout.md +0 -0
  22. {stata_code-0.12.1 → stata_code-0.12.2}/docs/industry-leader-roadmap.md +0 -0
  23. {stata_code-0.12.1 → stata_code-0.12.2}/docs/quickstart.zh.md +0 -0
  24. {stata_code-0.12.1 → stata_code-0.12.2}/examples/01-basic-regression.md +0 -0
  25. {stata_code-0.12.1 → stata_code-0.12.2}/examples/02-did-card-krueger.md +0 -0
  26. {stata_code-0.12.1 → stata_code-0.12.2}/examples/03-graphs.md +0 -0
  27. {stata_code-0.12.1 → stata_code-0.12.2}/examples/04-multi-session.md +0 -0
  28. {stata_code-0.12.1 → stata_code-0.12.2}/examples/05-large-matrix.md +0 -0
  29. {stata_code-0.12.1 → stata_code-0.12.2}/examples/06-cross-stack-parity-audit.md +0 -0
  30. {stata_code-0.12.1 → stata_code-0.12.2}/examples/07-data-mcp-handoff.md +0 -0
  31. {stata_code-0.12.1 → stata_code-0.12.2}/examples/README.md +0 -0
  32. {stata_code-0.12.1 → stata_code-0.12.2}/schema/run_result.schema.json +0 -0
  33. {stata_code-0.12.1 → stata_code-0.12.2}/scripts/build_skill_zip.py +0 -0
  34. {stata_code-0.12.1 → stata_code-0.12.2}/scripts/build_standalone.py +0 -0
  35. {stata_code-0.12.1 → stata_code-0.12.2}/scripts/check_github_actions.py +0 -0
  36. {stata_code-0.12.1 → stata_code-0.12.2}/scripts/check_versions.py +0 -0
  37. {stata_code-0.12.1 → stata_code-0.12.2}/scripts/export_schema.py +0 -0
  38. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/__main__.py +0 -0
  39. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/cli.py +0 -0
  40. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/__init__.py +0 -0
  41. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/_pool.py +0 -0
  42. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/_refs.py +0 -0
  43. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/_runtime.py +0 -0
  44. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/console.py +0 -0
  45. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/daemon.py +0 -0
  46. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/errors.py +0 -0
  47. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/handoff.py +0 -0
  48. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/lint.py +0 -0
  49. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/log_artifacts.py +0 -0
  50. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/notebook.py +0 -0
  51. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/policy.py +0 -0
  52. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/provenance.py +0 -0
  53. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/run_index.py +0 -0
  54. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/core/schema.py +0 -0
  55. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/doctor.py +0 -0
  56. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/__init__.py +0 -0
  57. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/__main__.py +0 -0
  58. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/assets/logo-32x32.png +0 -0
  59. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/assets/logo-64x64.png +0 -0
  60. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/assets/logo-svg.svg +0 -0
  61. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/kernel/kernel.py +0 -0
  62. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp/__init__.py +0 -0
  63. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp/__main__.py +0 -0
  64. {stata_code-0.12.1 → stata_code-0.12.2}/stata_code/mcp_setup.py +0 -0
  65. {stata_code-0.12.1 → stata_code-0.12.2}/tests/__init__.py +0 -0
  66. {stata_code-0.12.1 → stata_code-0.12.2}/tests/conftest.py +0 -0
  67. {stata_code-0.12.1 → stata_code-0.12.2}/tests/fixtures/.gitkeep +0 -0
  68. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_benchmark.py +0 -0
  69. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_bugfix_regressions.py +0 -0
  70. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_cancel.py +0 -0
  71. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_cli.py +0 -0
  72. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_console.py +0 -0
  73. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_daemon.py +0 -0
  74. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_doctor.py +0 -0
  75. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_errors.py +0 -0
  76. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_github_actions.py +0 -0
  77. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_handoff.py +0 -0
  78. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_kernel.py +0 -0
  79. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_lint.py +0 -0
  80. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_lint_mcp.py +0 -0
  81. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_log_artifacts.py +0 -0
  82. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_mcp.py +0 -0
  83. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_mcp_kernel_extra.py +0 -0
  84. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_mcp_transport.py +0 -0
  85. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_method_prompts.py +0 -0
  86. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_new_tools.py +0 -0
  87. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_notebook.py +0 -0
  88. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_notebook_phase2.py +0 -0
  89. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_policy.py +0 -0
  90. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_pool.py +0 -0
  91. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_pool_refs_extra.py +0 -0
  92. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_provenance.py +0 -0
  93. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_public_api.py +0 -0
  94. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_release_versions.py +0 -0
  95. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_run_index.py +0 -0
  96. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_runner.py +0 -0
  97. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_runtime_discovery.py +0 -0
  98. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_schema.py +0 -0
  99. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_schema_artifact.py +0 -0
  100. {stata_code-0.12.1 → stata_code-0.12.2}/tests/test_skill_package.py +0 -0
@@ -6,6 +6,63 @@ to semver-major.minor for the result schema (see `SCHEMA.md` §6).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## 0.12.2 — 2026-08-11
10
+
11
+ Correctness release. `results.estimation` could report confidence intervals
12
+ and p-values that disagreed with the Stata log printed beside them; if you
13
+ read coefficients out of `results.estimation` rather than the log, upgrade.
14
+
15
+ ### Fixed
16
+
17
+ - **`results.estimation` reported normal-approximation inference for `t` models
18
+ whenever `r(table)` had been cleared.** `r(table)` only survives until the
19
+ next command, so any block that runs something after its estimation —
20
+ `graph export`, `esttab`, a final `summarize` — fell back to rebuilding the
21
+ table from `e(b)` / `e(V)`, and that path hardcoded a normal approximation.
22
+ The result was a payload that quietly disagreed with the log beside it:
23
+
24
+ ```text
25
+ regress price_k mpg weight foreign
26
+ scatter price_k mpg // clears r(table)
27
+
28
+ log: mpg 95% CI [-.1261758, .169883] P>|t| 0.769
29
+ results.estimation: mpg 95% CI [-0.12362, 0.16732] p 0.7684
30
+ ```
31
+
32
+ The fallback now follows Stata's own rule — a *t* table on `e(df_r)` degrees
33
+ of freedom when that scalar is set, *z* otherwise — so the rebuilt table
34
+ matches the printed one to the last digit. `ci_level` also honours `e(level)`
35
+ instead of assuming 95, so `regress, level(90)` is no longer mislabelled.
36
+ Agents told to prefer `structuredContent` over the log were the ones exposed
37
+ to this; the `r(table)` path was always correct.
38
+
39
+ A new `estimation_from_e_b_v` warning records when the rebuild happened. It
40
+ fires only for an estimation produced by the current run, since `e()` is
41
+ session-global and would otherwise re-warn on every later call.
42
+
43
+ - **`include_results: "none"` hollowed out `results.estimation`.** `n_obs`,
44
+ `df_model`, `df_resid` and `model_stats` are all read from `e()` scalars, so
45
+ suppressing the `r()` / `e()` *echo* also removed the model-level numbers —
46
+ even though `include_estimation` is the documented knob for that block, and
47
+ SCHEMA.md §4 already promised `include_results` "never affects
48
+ `results.estimation`". The scalars are now always read when estimation is
49
+ wanted, and only withheld from the wire.
50
+
51
+ - **A finished background run reported a duration that kept growing.**
52
+ `Job.elapsed_ms` recomputed `now - submitted` on every poll instead of
53
+ freezing at completion, so a 1.7 s bootstrap polled ten minutes later
54
+ reported ten minutes. Callers attributing wall-clock time to Stata were
55
+ reading their own polling delay. The value is now frozen before the terminal
56
+ status is published, on both the success and error paths.
57
+
58
+ ### Changed
59
+
60
+ - **Macro values longer than 256 characters are elided on the wire** with an
61
+ explicit `… (N more chars elided)` marker. This is aimed at `e(rngstate)`,
62
+ roughly 2 KB of hex emitted on every `bootstrap` / `permute` / `simulate`
63
+ run that no consumer can act on. Names are always kept, and the estimation
64
+ contract is still built from the uncapped values.
65
+
9
66
  ## 0.12.1 — 2026-07-28
10
67
 
11
68
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: stata-code
3
- Version: 0.12.1
3
+ Version: 0.12.2
4
4
  Summary: Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)
5
5
  Project-URL: Homepage, https://github.com/brycewang-stanford/stata-code
6
6
  Project-URL: Repository, https://github.com/brycewang-stanford/stata-code
@@ -343,7 +343,7 @@ Stata's `r()` and `e()` return dictionaries, structurally separated. Each follow
343
343
  | Sub-field | Type | Notes |
344
344
  | --- | --- | --- |
345
345
  | `scalars` | `dict<str, number \| null>` | Native floats / ints. Stata's system missing (`.`) → JSON `null`. Extended missings (`.a`–`.z`) → `null` with information loss. |
346
- | `macros` | `dict<str, string>` | Stata macro values verbatim. |
346
+ | `macros` | `dict<str, string>` | Stata macro values verbatim, except that a value longer than 256 characters is truncated and suffixed with `… (N more chars elided)`. The cap exists for macros like `e(rngstate)`, which is ~2 KB of hex on every `bootstrap` / `permute` / `simulate` run and carries nothing a consumer can act on. The name is always kept, and `results.estimation` is derived from the uncapped values. |
347
347
  | `matrices` | `dict<str, Matrix>` | See `Matrix` below. |
348
348
 
349
349
  **`Matrix` shape:**
@@ -383,9 +383,9 @@ Stata's `r()` and `e()` return dictionaries, structurally separated. Each follow
383
383
  | `n_obs` | `int \| null` | Integer form of `e(N)` when available. |
384
384
  | `df_model` | `number \| null` | Mirrors `e(df_m)`. |
385
385
  | `df_resid` | `number \| null` | Mirrors `e(df_r)`. |
386
- | `statistic_kind` | `"t" \| "z"` | Which statistic fills each coefficient's `statistic` field. |
387
- | `source` | `"r_table" \| "e_b_v"` | `r_table` means values were copied from Stata's displayed `r(table)` after verifying its columns and `b` row match `e(b)`; `e_b_v` means point estimates come from `e(b)` and inference, when present, is computed from `e(V)` with a normal approximation. A matrix returned by `ref` is resolved before use, so a deferred `e(V)` still yields standard errors. |
388
- | `ci_level` | `number` | Confidence level used for `ci_low` / `ci_high`; currently `95.0`. |
386
+ | `statistic_kind` | `"t" \| "z"` | Which statistic fills each coefficient's `statistic` field, and which distribution produced `p_value` / `ci_low` / `ci_high`. On the `e_b_v` path this follows Stata's own rule — `t` on `df_resid` degrees of freedom when `e(df_r)` is set, `z` otherwise — so a rebuilt table agrees with the printed log rather than reporting a normal-approximation interval next to a `P>\|t\|` column. |
387
+ | `source` | `"r_table" \| "e_b_v"` | `r_table` means values were copied from Stata's displayed `r(table)` after verifying its columns and `b` row match `e(b)`; `e_b_v` means point estimates come from `e(b)` and inference, when present, is computed from `e(V)`. A matrix returned by `ref` is resolved before use, so a deferred `e(V)` still yields standard errors. |
388
+ | `ci_level` | `number` | Confidence level used for `ci_low` / `ci_high`. Mirrors `e(level)` when the command stored it, so `regress, level(90)` reports `90.0`; defaults to `95.0` otherwise. |
389
389
  | `coefficients` | `array<Coefficient>` | One row per term in `e(b)`, subject to the caller's `include_estimation` / `max_coefficients` budget. |
390
390
  | `n_coefficients` | `int` | The model's true term count. Equals `coefficients.length` unless the caller trimmed the table, so `12` rows out of `n_coefficients: 141` is never mistaken for a 12-term model. |
391
391
  | `coefficients_truncated` | `bool` | `true` when rows were dropped to satisfy the budget. |
@@ -564,11 +564,21 @@ The rc-to-kind table is approximate and lives in code (`stata_code.core.errors`)
564
564
 
565
565
  | Field | Type | Notes |
566
566
  | --- | --- | --- |
567
- | `kind` | `string` | Open enum. Common values: `convergence`, `singular`, `boundary`, `omitted_collinear`, `non_finite`, `dataset_variables_truncated`, `unknown`. |
567
+ | `kind` | `string` | Open enum. Common values: `convergence`, `singular`, `boundary`, `omitted_collinear`, `non_finite`, `dataset_variables_truncated`, `log_closed`, `output_tracking_skipped`, `estimation_from_e_b_v`, `unknown`. |
568
568
  | `message` | `string` | Human-readable, single line. Truncated to 1,024 characters. |
569
569
 
570
570
  Warnings are de-duplicated by `(kind, message)`.
571
571
 
572
+ `estimation_from_e_b_v` reports that this run performed an estimation whose
573
+ `r(table)` was already gone by the time results were read — a later command in
574
+ the same submission cleared it — so `results.estimation` was rebuilt from
575
+ `e(b)` / `e(V)`. The rebuilt numbers still match the printed log (see
576
+ `statistic_kind` in §3.5), so this is provenance rather than a correctness
577
+ alarm; putting the estimation last in the block restores the `r_table` path.
578
+ It is emitted only when the estimation was produced by *this* run: `e()` is
579
+ session-global, so a later `summarize` in the same session keeps reporting the
580
+ inherited table through the same fallback without re-warning.
581
+
572
582
  ### 3.9 `origin`
573
583
 
574
584
  ```json
@@ -607,7 +617,7 @@ The schema also dictates what callers may *ask for*. Every frontend exposes the
607
617
  | `include_graphs` | `"ref" \| "inline" \| "none"` | `"ref"` | `"none"` skips graph capture entirely (cheapest); `"ref"` captures and returns refs; `"inline"` base64-encodes bytes into `inline`. |
608
618
  | `graph_format` | `"png" \| "svg" \| "pdf"` | `"png"` | Render format. |
609
619
  | `include_dataset_variables` | `bool` | `true` | Set `false` to omit `dataset.variables`. |
610
- | `include_results` | `"none" \| "scalars" \| "full"` | `"scalars"` | Payload budget for `results.r` / `results.e`. `"scalars"` inlines scalars and macros and emits every matrix as a stub (§3.4); `"full"` inlines matrix values up to the ~10,000-cell cap; `"none"` omits `r()` / `e()` entirely. Never affects `results.estimation`. |
620
+ | `include_results` | `"none" \| "scalars" \| "full"` | `"scalars"` | Payload budget for `results.r` / `results.e`. `"scalars"` inlines scalars and macros and emits every matrix as a stub (§3.4); `"full"` inlines matrix values up to the ~10,000-cell cap; `"none"` omits `r()` / `e()` entirely. Never affects `results.estimation`: the model-level fields (`n_obs`, `df_model`, `df_resid`, `model_stats`, `depvar`) are read from `e()` regardless of this setting and merely withheld from the wire, so `"none"` does not hollow out the estimation contract. Use `include_estimation` to trim that block. |
611
621
  | `include_estimation` | `"none" \| "summary" \| "full"` | `"full"` | Payload budget for `results.estimation`. `"summary"` keeps the model-level block and drops per-term rows. |
612
622
  | `max_coefficients` | `int \| null` | `null` | Cap on `estimation.coefficients` rows. `n_coefficients` still reports the true count. |
613
623
  | `timeout_ms` | `int \| null` | `600000` (10 min) | Hard timeout. `null` disables. On expiry, returns `ok: false`, `error.kind: "timeout"`, `rc: -2`. The budget covers **queueing**: a call waiting on a session whose Stata process is mid-run returns `rc: -5`, `error.kind: "session_busy"` rather than blocking past its deadline. Frontends MAY override the default if their use case demands. |
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "stata-code"
7
- version = "0.12.1"
7
+ version = "0.12.2"
8
8
  description = "Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)"
9
9
  # The repo default README (README.md) is Chinese; keep the PyPI long description
10
10
  # English by pointing at README.en.md.
@@ -260,7 +260,7 @@ def is_available() -> bool:
260
260
  return True
261
261
 
262
262
 
263
- __version__ = "0.12.1"
263
+ __version__ = "0.12.2"
264
264
 
265
265
  __all__ = [
266
266
  # Primary entry points
@@ -9,8 +9,18 @@ parse from log prose:
9
9
  exactly what Stata printed, so we copy them verbatim.
10
10
  * ``e(b)`` / ``e(V)`` — the coefficient vector and its variance–covariance
11
11
  matrix. Always present after an estimation command. We use these as a
12
- fallback, computing ``se`` from ``diag(V)`` and ``z``/``p``/CI under a normal
13
- approximation (clearly flagged via ``source`` / ``statistic_kind``).
12
+ fallback, computing ``se`` from ``diag(V)`` and the statistic / p-value / CI
13
+ ourselves (flagged via ``source="e_b_v"``).
14
+
15
+ The fallback matters more than "fallback" suggests: ``r(table)`` is wiped by
16
+ the *next* command, so any block that ends in ``graph export`` / ``esttab`` /
17
+ ``summarize`` after its estimation lands here. It therefore has to reproduce
18
+ what Stata displayed, not merely something defensible. Stata's own rule (see
19
+ ``_coef_table``) is distributional: when ``e(df_r)`` is set the table is a
20
+ *t* table on that many residual degrees of freedom, and otherwise a *z* table.
21
+ We follow exactly that rule, so the fallback CI agrees with the printed log
22
+ digit for digit instead of quietly reporting a normal-approximation interval
23
+ alongside a log that says ``P>|t|``.
14
24
 
15
25
  The result is :class:`EstimationResult`, attached to ``RunResult.results.
16
26
  estimation`` so every frontend (MCP, kernel, VS Code) gets the same typed
@@ -22,6 +32,7 @@ from __future__ import annotations
22
32
 
23
33
  import math
24
34
  from collections.abc import Callable
35
+ from functools import lru_cache
25
36
 
26
37
  from stata_code.core.schema import (
27
38
  Coefficient,
@@ -162,6 +173,111 @@ def _two_sided_normal_p(z: float) -> float:
162
173
  return 2.0 * (1.0 - _normal_cdf(abs(z)))
163
174
 
164
175
 
176
+ def _betacf(a: float, b: float, x: float) -> float:
177
+ """Continued fraction for the incomplete beta function (Lentz's method)."""
178
+ tiny = 1e-30
179
+ qab, qap, qam = a + b, a + 1.0, a - 1.0
180
+ c = 1.0
181
+ d = 1.0 - qab * x / qap
182
+ if abs(d) < tiny:
183
+ d = tiny
184
+ d = 1.0 / d
185
+ h = d
186
+ for m in range(1, 301):
187
+ m2 = 2 * m
188
+ aa = m * (b - m) * x / ((qam + m2) * (a + m2))
189
+ d = 1.0 + aa * d
190
+ if abs(d) < tiny:
191
+ d = tiny
192
+ c = 1.0 + aa / c
193
+ if abs(c) < tiny:
194
+ c = tiny
195
+ d = 1.0 / d
196
+ h *= d * c
197
+ aa = -(a + m) * (qab + m) * x / ((a + m2) * (qap + m2))
198
+ d = 1.0 + aa * d
199
+ if abs(d) < tiny:
200
+ d = tiny
201
+ c = 1.0 + aa / c
202
+ if abs(c) < tiny:
203
+ c = tiny
204
+ d = 1.0 / d
205
+ delta = d * c
206
+ h *= delta
207
+ if abs(delta - 1.0) < 1e-15:
208
+ break
209
+ return h
210
+
211
+
212
+ def _betainc_reg(a: float, b: float, x: float) -> float:
213
+ """Regularized incomplete beta ``I_x(a, b)``, stdlib only."""
214
+ if x <= 0.0:
215
+ return 0.0
216
+ if x >= 1.0:
217
+ return 1.0
218
+ ln_front = (
219
+ math.lgamma(a + b) - math.lgamma(a) - math.lgamma(b) + a * math.log(x) + b * math.log1p(-x)
220
+ )
221
+ front = math.exp(ln_front)
222
+ if x < (a + 1.0) / (a + b + 2.0):
223
+ return front * _betacf(a, b, x) / a
224
+ return 1.0 - front * _betacf(b, a, 1.0 - x) / b
225
+
226
+
227
+ def _two_sided_t_p(t: float, df: float) -> float:
228
+ """Two-sided p-value for a t statistic on ``df`` degrees of freedom.
229
+
230
+ ``P(|T| > |t|) = I_{df/(df+t²)}(df/2, 1/2)`` — exact, no table lookup.
231
+ """
232
+ if df <= 0.0:
233
+ return _two_sided_normal_p(t)
234
+ if math.isinf(t):
235
+ return 0.0
236
+ return _betainc_reg(df / 2.0, 0.5, df / (df + t * t))
237
+
238
+
239
+ @lru_cache(maxsize=256)
240
+ def _t_crit(df: float, level: float) -> float:
241
+ """Two-sided critical value of the t distribution for a CI ``level`` in %.
242
+
243
+ Found by bisection on :func:`_two_sided_t_p`, which is strictly decreasing
244
+ in ``|t|`` — robust for any ``df`` without shipping a quantile table.
245
+ """
246
+ alpha = 1.0 - level / 100.0
247
+ if df <= 0.0 or not (0.0 < alpha < 1.0):
248
+ return _z_crit(level)
249
+ lo, hi = 0.0, 1.0
250
+ while _two_sided_t_p(hi, df) > alpha and hi < 1e12:
251
+ hi *= 2.0
252
+ for _ in range(200):
253
+ mid = 0.5 * (lo + hi)
254
+ if _two_sided_t_p(mid, df) > alpha:
255
+ lo = mid
256
+ else:
257
+ hi = mid
258
+ if hi - lo < 1e-13 * max(1.0, hi):
259
+ break
260
+ return 0.5 * (lo + hi)
261
+
262
+
263
+ @lru_cache(maxsize=64)
264
+ def _z_crit(level: float) -> float:
265
+ """Two-sided standard-normal critical value for a CI ``level`` in %."""
266
+ alpha = 1.0 - level / 100.0
267
+ if not (0.0 < alpha < 1.0):
268
+ return _Z_CRIT_95
269
+ lo, hi = 0.0, 40.0
270
+ for _ in range(200):
271
+ mid = 0.5 * (lo + hi)
272
+ if _two_sided_normal_p(mid) > alpha:
273
+ lo = mid
274
+ else:
275
+ hi = mid
276
+ if hi - lo < 1e-14 * max(1.0, hi):
277
+ break
278
+ return 0.5 * (lo + hi)
279
+
280
+
165
281
  def _cell(row: list[float | None] | None, j: int) -> float | None:
166
282
  if row is None or j >= len(row):
167
283
  return None
@@ -284,13 +400,16 @@ def _from_b_v(
284
400
  v: Matrix | None,
285
401
  resolve: MatrixResolver | None = None,
286
402
  b_values: list[list[float | None]] | None = None,
287
- ) -> list[Coefficient]:
403
+ df_resid: float | None = None,
404
+ level: float = 95.0,
405
+ ) -> tuple[list[Coefficient], str]:
288
406
  """Compute coefficient rows from e(b) (and e(V) when available).
289
407
 
290
408
  ``se``/``statistic``/``p_value``/CI are filled only when e(V) is
291
409
  obtainable — inline or through its ref; otherwise just the point estimates
292
- are returned. Inference uses the normal approximation — callers flag this
293
- via ``source="e_b_v"``.
410
+ are returned. Returns ``(coefficients, statistic_kind)``: a *t* table on
411
+ ``df_resid`` degrees of freedom when Stata set ``e(df_r)``, matching what
412
+ the command printed, and a *z* table otherwise.
294
413
  """
295
414
  if b_values is None:
296
415
  b_values = _values(b, resolve)
@@ -305,6 +424,10 @@ def _from_b_v(
305
424
  else:
306
425
  v_diag = [None] * len(terms)
307
426
 
427
+ df = float(df_resid) if df_resid is not None and df_resid > 0.0 else None
428
+ stat_kind = "t" if df is not None else "z"
429
+ crit = _t_crit(df, level) if df is not None else _z_crit(level)
430
+
308
431
  coeffs: list[Coefficient] = []
309
432
  for j, term in enumerate(terms):
310
433
  b_val = _cell(b_row, j)
@@ -316,9 +439,9 @@ def _from_b_v(
316
439
  ci_high: float | None = None
317
440
  if b_val is not None and se is not None and se > 0.0:
318
441
  stat = b_val / se
319
- p_val = _two_sided_normal_p(stat)
320
- ci_low = b_val - _Z_CRIT_95 * se
321
- ci_high = b_val + _Z_CRIT_95 * se
442
+ p_val = _two_sided_t_p(stat, df) if df is not None else _two_sided_normal_p(stat)
443
+ ci_low = b_val - crit * se
444
+ ci_high = b_val + crit * se
322
445
  coeffs.append(
323
446
  Coefficient(
324
447
  term=term,
@@ -330,7 +453,7 @@ def _from_b_v(
330
453
  ci_high=ci_high,
331
454
  )
332
455
  )
333
- return coeffs
456
+ return coeffs, stat_kind
334
457
 
335
458
 
336
459
  def _model_stats(scalars: dict[str, float | None]) -> dict[str, float | None]:
@@ -406,14 +529,30 @@ def build_estimation_from_returns(
406
529
  statistic_kind: str
407
530
  source: str
408
531
 
532
+ df_resid = e.scalars.get("df_r")
533
+ # e(level) is the CI level the command actually displayed; absent means the
534
+ # Stata default of 95. Both the critical value and the reported ci_level
535
+ # follow it, so `regress, level(90)` does not come back labelled 95.
536
+ level_raw = e.scalars.get("level")
537
+ ci_level = float(level_raw) if level_raw is not None else 95.0
538
+
409
539
  table = r.matrices.get("table")
410
540
  parsed = _from_r_table(table, resolve_matrix) if table is not None else None
411
541
  if parsed is not None and table is not None and _r_table_matches_b(table, b, resolve_matrix):
412
542
  coeffs, statistic_kind = parsed
413
543
  source = "r_table"
414
544
  else:
415
- coeffs = _from_b_v(b, v, resolve_matrix, b_values)
416
- statistic_kind = "z"
545
+ # r(table) is gone — cleared by whatever command ran after the
546
+ # estimation. Rebuild the table from e(b)/e(V) on the same
547
+ # distribution Stata used, so the numbers still match the log.
548
+ coeffs, statistic_kind = _from_b_v(
549
+ b,
550
+ v,
551
+ resolve_matrix,
552
+ b_values,
553
+ df_resid=float(df_resid) if df_resid is not None else None,
554
+ level=ci_level,
555
+ )
417
556
  source = "e_b_v"
418
557
 
419
558
  n_obs: int | None = None
@@ -430,9 +569,10 @@ def build_estimation_from_returns(
430
569
  depvar=depvar,
431
570
  n_obs=n_obs,
432
571
  df_model=e.scalars.get("df_m"),
433
- df_resid=e.scalars.get("df_r"),
572
+ df_resid=df_resid,
434
573
  statistic_kind=statistic_kind, # type: ignore[arg-type]
435
574
  source=source, # type: ignore[arg-type]
575
+ ci_level=ci_level,
436
576
  coefficients=coeffs,
437
577
  model_stats=_model_stats(e.scalars),
438
578
  diagnostics=_command_diagnostics(command, e.scalars),
@@ -60,6 +60,7 @@ class Job:
60
60
  "error",
61
61
  "_done",
62
62
  "_started",
63
+ "_elapsed_ms",
63
64
  )
64
65
 
65
66
  def __init__(self, job_id: str, session_id: str, code: str) -> None:
@@ -73,6 +74,7 @@ class Job:
73
74
  self.error: str | None = None
74
75
  self._done = threading.Event()
75
76
  self._started = time.monotonic()
77
+ self._elapsed_ms: int | None = None
76
78
 
77
79
  @property
78
80
  def done(self) -> bool:
@@ -83,6 +85,15 @@ class Job:
83
85
  return self._done.wait(timeout=timeout_s)
84
86
 
85
87
  def elapsed_ms(self) -> int:
88
+ """How long the run took, or has been running so far.
89
+
90
+ Frozen once the job reaches a terminal state. Recomputing it on every
91
+ poll made a finished job's reported duration grow without bound — a
92
+ 1.7s bootstrap polled ten minutes later reported ten minutes — so
93
+ callers attributing wall time to Stata were reading their own latency.
94
+ """
95
+ if self._elapsed_ms is not None:
96
+ return self._elapsed_ms
86
97
  return max(0, int((time.monotonic() - self._started) * 1000))
87
98
 
88
99
  def summary(self) -> dict[str, Any]:
@@ -140,6 +151,11 @@ class JobRegistry:
140
151
  # sees a terminal status is guaranteed to see the payload with it.
141
152
  job.result = result
142
153
  job.error = error
154
+ # Freeze the clock before publishing a terminal status, so no
155
+ # reader can observe "done" alongside a still-advancing duration.
156
+ job._elapsed_ms = max( # noqa: SLF001 - same-module private handshake
157
+ 0, int((time.monotonic() - job._started) * 1000) # noqa: SLF001
158
+ )
143
159
  job.finished_at = _utc_iso_ms()
144
160
  job.status = status
145
161
  job._done.set() # noqa: SLF001 - same-module private handshake
@@ -140,6 +140,13 @@ _DATASET_VAR_CAP = 200
140
140
  # a matrix would inline more than ~10,000 cells."
141
141
  MATRIX_INLINE_CELL_CAP = 10_000
142
142
 
143
+ # Longest macro value put on the wire verbatim. Stata macros are mostly short
144
+ # (`e(cmd)`, `e(depvar)`, `e(vcetype)`), but a few are enormous and carry
145
+ # nothing an agent can act on — `e(rngstate)` alone is ~2 KB of hex on every
146
+ # bootstrap / permute / simulate run. Anything longer is elided with an
147
+ # explicit marker rather than dropped, so the name still shows up.
148
+ MACRO_INLINE_CHAR_CAP = 256
149
+
143
150
  # Stata's system missing `.` is exactly maxdouble (2**1023); the 26 extended
144
151
  # missings `.a`–`.z` occupy the next representable doubles above it. `sfi`
145
152
  # hands all of them back as ordinary Python floats, so any value at or above
@@ -659,7 +666,23 @@ def _project_returns(
659
666
  n_rows=n_rows,
660
667
  n_cols=n_cols,
661
668
  )
662
- return StataReturns(scalars=rv.scalars, macros=rv.macros, matrices=matrices)
669
+ return StataReturns(
670
+ scalars=rv.scalars,
671
+ macros={name: _cap_macro(v) for name, v in rv.macros.items()},
672
+ matrices=matrices,
673
+ )
674
+
675
+
676
+ def _cap_macro(value: str) -> str:
677
+ """Elide a macro value too long to be worth its tokens on the wire.
678
+
679
+ Applied only at projection time — the estimation contract is built from the
680
+ uncapped values, so nothing downstream loses precision.
681
+ """
682
+ if len(value) <= MACRO_INLINE_CHAR_CAP:
683
+ return value
684
+ dropped = len(value) - MACRO_INLINE_CHAR_CAP
685
+ return f"{value[:MACRO_INLINE_CHAR_CAP]}… ({dropped} more chars elided)"
663
686
 
664
687
 
665
688
  def _collect_dataset(rt: Any, include_variables: bool) -> DatasetInfo:
@@ -1329,6 +1352,17 @@ def execute(
1329
1352
  # Snapshot existing graph names before user code so we can take a delta
1330
1353
  # afterward.
1331
1354
  pre_graphs = _list_graph_names(rt) if include_graphs != "none" else []
1355
+
1356
+ # e() is session-global and outlives the run that set it, so
1357
+ # `results.estimation` may describe an estimation from an earlier call.
1358
+ # Fingerprint it now to tell "this run estimated something" from "this
1359
+ # run inherited an estimation" — the difference decides whether an
1360
+ # e(b)/e(V) rebuild is worth reporting.
1361
+ pre_estimation = (
1362
+ _estimation_fingerprint(rt)
1363
+ if estimation_mode != IncludeEstimation.NONE
1364
+ else None
1365
+ )
1332
1366
  graph_source_hints, unnamed_graph_source_hints = (
1333
1367
  _graph_source_hints(code) if include_graphs != "none" else ({}, [])
1334
1368
  )
@@ -1437,6 +1471,34 @@ def execute(
1437
1471
  ok = error is None and rc == 0
1438
1472
  warnings = _extract_warnings(log_text)
1439
1473
 
1474
+ # r(table) is cleared by the next command, so a block that runs anything
1475
+ # after its estimation gets a table rebuilt from e(b)/e(V). The numbers
1476
+ # follow Stata's own t-vs-z rule and so still match the log, but the
1477
+ # provenance is worth stating: a caller comparing against `r(table)` rows
1478
+ # it did not capture should know which path produced these.
1479
+ #
1480
+ # Gated on the estimation being *this run's*. e() outlives the call that
1481
+ # set it, so every later `summarize` in the session re-reports the same
1482
+ # inherited table through the same fallback — warning on those would be
1483
+ # pure noise about a rebuild the caller did not ask for and cannot act on.
1484
+ estimated_here = (
1485
+ pre_estimation is not None and _estimation_fingerprint(rt) != pre_estimation
1486
+ )
1487
+ if estimated_here and estimation is not None and estimation.source == "e_b_v":
1488
+ warnings.append(
1489
+ StataWarning(
1490
+ kind="estimation_from_e_b_v",
1491
+ message=(
1492
+ "results.estimation was rebuilt from e(b)/e(V) because r(table) "
1493
+ "no longer described the current estimation — a later command "
1494
+ "cleared it. Inference uses "
1495
+ f"{estimation.statistic_kind!r} as Stata would "
1496
+ f"(df_r={estimation.df_resid!r}). Put the estimation last in the "
1497
+ "block, or re-run it alone, to get the r(table) values verbatim."
1498
+ ),
1499
+ )
1500
+ )
1501
+
1440
1502
  # A failed run may have left `log using` handles open. Close only the ones
1441
1503
  # this run opened: a log the caller opened in an earlier successful run is
1442
1504
  # theirs to manage and must survive.
@@ -1599,6 +1661,14 @@ def _collect_results(
1599
1661
  # With results suppressed but estimation wanted, read only the matrices the
1600
1662
  # contract is derived from rather than every matrix in scope.
1601
1663
  named = results_mode != IncludeResults.NONE
1664
+ # e() scalars and macros are inputs to the estimation contract, not just
1665
+ # payload: n_obs, df_model, df_resid, model_stats, the t-vs-z choice and
1666
+ # depvar all come from them. Suppressing them for `include_results="none"`
1667
+ # silently hollowed out `estimation` — a caller who asked only to stop
1668
+ # *echoing* r()/e() lost the model-level numbers that `include_estimation`
1669
+ # is the knob for. Read them whenever estimation is wanted; the wire
1670
+ # projection below still drops them when the caller said "none".
1671
+ e_named = named or want_estimation
1602
1672
  raw = ResultsInfo(
1603
1673
  r=_collect_returns(
1604
1674
  rt,
@@ -1609,7 +1679,7 @@ def _collect_results(
1609
1679
  e=_collect_returns(
1610
1680
  rt,
1611
1681
  "e",
1612
- want_named=named,
1682
+ want_named=e_named,
1613
1683
  matrix_names=None if named else _ESTIMATION_MATRICES["e"],
1614
1684
  ),
1615
1685
  last_estimation_cmd=_last_estimation_cmd(rt),
@@ -1655,6 +1725,27 @@ def _last_estimation_cmd(rt: Any) -> str | None:
1655
1725
  return None
1656
1726
 
1657
1727
 
1728
+ def _estimation_fingerprint(rt: Any) -> tuple[str | None, str | None, float | None]:
1729
+ """Cheap identity for the estimation currently in e-scope.
1730
+
1731
+ Three reads, no matrices: enough to tell one estimation from the next
1732
+ without paying to fetch e(b). Used only to compare before/after a run.
1733
+ """
1734
+ sfi = rt.sfi
1735
+
1736
+ def _macro(name: str) -> str | None:
1737
+ try:
1738
+ return sfi.Macro.getGlobal(name) or None
1739
+ except Exception: # noqa: BLE001
1740
+ return None
1741
+
1742
+ try:
1743
+ n = sfi.Scalar.getValue("e(N)")
1744
+ except Exception: # noqa: BLE001
1745
+ n = None
1746
+ return (_macro("e(cmd)"), _macro("e(cmdline)"), _norm_stata_number(n))
1747
+
1748
+
1658
1749
  # ─────────────────────────────────────────────────────────────────────────────
1659
1750
  # Multi-session via Stata frames (Module 4)
1660
1751
  # ─────────────────────────────────────────────────────────────────────────────
@@ -102,7 +102,7 @@ from stata_code.core.runner import (
102
102
  )
103
103
  from stata_code.core.schema import RunResult
104
104
 
105
- __version__ = "0.12.1"
105
+ __version__ = "0.12.2"
106
106
 
107
107
  SERVER_INSTRUCTIONS = (
108
108
  "Use stata-code for running and inspecting Stata code. Prefer structuredContent "
@@ -374,6 +374,49 @@ class TestJobRegistry:
374
374
  assert job.result == "RESULT"
375
375
  assert job.finished_at is not None
376
376
 
377
+ def test_elapsed_ms_freezes_when_the_job_finishes(self, monkeypatch):
378
+ # Recomputing elapsed on every poll made a finished job's duration
379
+ # grow with the caller's polling delay, so a 2s run polled a minute
380
+ # later reported a minute of Stata time.
381
+ from stata_code.core import jobs
382
+
383
+ monkeypatch.setattr(jobs, "pool_execute", lambda code, **kw: "RESULT") # noqa: ARG005
384
+ registry = jobs.JobRegistry()
385
+ job = registry.submit("display 1")
386
+ assert job.wait(5) is True
387
+
388
+ first = job.elapsed_ms()
389
+ time.sleep(0.15)
390
+ assert job.elapsed_ms() == first
391
+ assert job.summary()["elapsed_ms"] == first
392
+
393
+ def test_elapsed_ms_still_advances_while_running(self, monkeypatch):
394
+ from stata_code.core import jobs
395
+
396
+ gate = threading.Event()
397
+ monkeypatch.setattr(jobs, "pool_execute", lambda code, **kw: gate.wait(10)) # noqa: ARG005
398
+ registry = jobs.JobRegistry()
399
+ job = registry.submit("display 1")
400
+ first = job.elapsed_ms()
401
+ time.sleep(0.05)
402
+ assert job.elapsed_ms() >= first
403
+ gate.set()
404
+ job.wait(5)
405
+
406
+ def test_elapsed_ms_is_frozen_on_the_error_path_too(self, monkeypatch):
407
+ from stata_code.core import jobs
408
+
409
+ def _boom(code, **kwargs): # noqa: ARG001
410
+ raise ValueError("bad option")
411
+
412
+ monkeypatch.setattr(jobs, "pool_execute", _boom)
413
+ registry = jobs.JobRegistry()
414
+ job = registry.submit("display 1")
415
+ assert job.wait(5) is True
416
+ first = job.elapsed_ms()
417
+ time.sleep(0.15)
418
+ assert job.elapsed_ms() == first
419
+
377
420
  def test_failure_is_recorded_not_swallowed(self, monkeypatch):
378
421
  from stata_code.core import jobs
379
422
 
@@ -2,14 +2,20 @@
2
2
 
3
3
  Covers both extraction paths:
4
4
  * ``r(table)`` (referee-grade — values copied exactly as Stata displayed them)
5
- * ``e(b)`` / ``e(V)`` fallback (se/z/p/CI computed under a normal approximation)
5
+ * ``e(b)`` / ``e(V)`` fallback (se / statistic / p / CI computed here, on the
6
+ same distribution Stata used: t when ``e(df_r)`` is set, z otherwise)
6
7
  """
7
8
 
8
9
  from __future__ import annotations
9
10
 
10
11
  import math
11
12
 
13
+ import pytest
14
+
12
15
  from stata_code.core.estimation import (
16
+ _t_crit,
17
+ _two_sided_t_p,
18
+ _z_crit,
13
19
  build_estimation_from_returns,
14
20
  build_estimation_result,
15
21
  )
@@ -98,7 +104,56 @@ class TestRTablePath:
98
104
 
99
105
 
100
106
  # ─────────────────────────────────────────────────────────────────────────────
101
- # e(b)/e(V) fallback — computed se/z/p/CI
107
+ # Distribution helpers — checked against published t-table values
108
+ # ─────────────────────────────────────────────────────────────────────────────
109
+
110
+
111
+ class TestDistributions:
112
+ @pytest.mark.parametrize(
113
+ ("df", "expected"),
114
+ [
115
+ (1, 12.706204736),
116
+ (5, 2.570581836),
117
+ (10, 2.228138852),
118
+ (30, 2.042272456),
119
+ (70, 1.994437112),
120
+ (120, 1.979930405),
121
+ (100000, 1.959987707),
122
+ ],
123
+ )
124
+ def test_t_crit_matches_published_two_sided_95pct_values(self, df, expected):
125
+ assert math.isclose(_t_crit(float(df), 95.0), expected, rel_tol=1e-8)
126
+
127
+ def test_t_crit_converges_to_the_normal_as_df_grows(self):
128
+ assert math.isclose(_t_crit(1e9, 95.0), _z_crit(95.0), rel_tol=1e-6)
129
+
130
+ @pytest.mark.parametrize(
131
+ ("t", "df", "expected"),
132
+ [
133
+ (2.042272456, 30, 0.05),
134
+ (1.0, 10, 0.340893),
135
+ (0.29443916691636146, 70, 0.769293740812962),
136
+ (5.493002930280353, 70, 5.991178e-7),
137
+ ],
138
+ )
139
+ def test_two_sided_t_p_matches_reference_values(self, t, df, expected):
140
+ assert math.isclose(_two_sided_t_p(t, float(df)), expected, rel_tol=1e-5)
141
+
142
+ def test_t_p_is_symmetric_and_bounded(self):
143
+ for t in (0.0, 0.5, 3.0, 40.0):
144
+ p = _two_sided_t_p(t, 12.0)
145
+ assert 0.0 <= p <= 1.0
146
+ assert math.isclose(p, _two_sided_t_p(-t, 12.0), rel_tol=1e-15)
147
+ assert math.isclose(_two_sided_t_p(0.0, 12.0), 1.0, rel_tol=1e-12)
148
+
149
+ def test_z_crit_matches_known_levels(self):
150
+ assert math.isclose(_z_crit(95.0), 1.959963984540054, rel_tol=1e-12)
151
+ assert math.isclose(_z_crit(99.0), 2.5758293035489004, rel_tol=1e-10)
152
+ assert math.isclose(_z_crit(90.0), 1.6448536269514722, rel_tol=1e-10)
153
+
154
+
155
+ # ─────────────────────────────────────────────────────────────────────────────
156
+ # e(b)/e(V) fallback — computed se / statistic / p / CI
102
157
  # ─────────────────────────────────────────────────────────────────────────────
103
158
 
104
159
 
@@ -128,6 +183,63 @@ class TestBVFallback:
128
183
  assert math.isclose(c.ci_low, 10.0 - 1.959963984540054 * 2.0, rel_tol=1e-12)
129
184
  assert math.isclose(c.ci_high, 10.0 + 1.959963984540054 * 2.0, rel_tol=1e-12)
130
185
 
186
+ def test_df_r_switches_the_fallback_to_a_t_table(self):
187
+ # Stata prints a t table whenever e(df_r) is set, so the fallback must
188
+ # too — otherwise the rebuilt CI silently disagrees with the log.
189
+ cols = ["x"]
190
+ e = StataReturns(
191
+ scalars={"df_r": 70},
192
+ matrices={"b": _b([10.0], cols), "V": _v([4.0], cols)},
193
+ )
194
+ est = build_estimation_from_returns(e, StataReturns())
195
+ assert est.source == "e_b_v"
196
+ assert est.statistic_kind == "t"
197
+ # t(.975, 70) = 1.9944371... — wider than the 1.95996 normal interval.
198
+ c = est.coefficients[0]
199
+ assert math.isclose(c.ci_low, 10.0 - 1.9944371112999 * 2.0, rel_tol=1e-9)
200
+ assert math.isclose(c.ci_high, 10.0 + 1.9944371112999 * 2.0, rel_tol=1e-9)
201
+
202
+ def test_fallback_reproduces_statas_printed_regress_table(self):
203
+ # Values taken from a live `regress price_k mpg weight foreign` on
204
+ # sysuse auto (N=74, df_r=70). Before the t fix this path returned
205
+ # [-0.12362, 0.16732] against a log that printed [-.1261758, .169883].
206
+ cols = ["mpg"]
207
+ b, se = 0.02185360997213235, 0.07422113776846848
208
+ e = StataReturns(
209
+ scalars={"N": 74, "df_m": 3, "df_r": 70},
210
+ macros={"cmd": "regress", "depvar": "price_k"},
211
+ matrices={"b": _b([b], cols), "V": _v([se * se], cols)},
212
+ )
213
+ est = build_estimation_from_returns(e, StataReturns())
214
+ c = est.coefficients[0]
215
+ assert math.isclose(c.ci_low, -0.1261757816711833, rel_tol=1e-9)
216
+ assert math.isclose(c.ci_high, 0.16988300161544798, rel_tol=1e-9)
217
+ assert math.isclose(c.p_value, 0.769293740812962, rel_tol=1e-9)
218
+
219
+ def test_no_df_r_stays_on_the_normal_approximation(self):
220
+ # Commands that do not set e(df_r) (ml-family, bootstrap, ...) print a
221
+ # z table; the fallback must not invent residual degrees of freedom.
222
+ cols = ["x"]
223
+ e = StataReturns(
224
+ scalars={"N": 500},
225
+ macros={"cmd": "logit"},
226
+ matrices={"b": _b([10.0], cols), "V": _v([4.0], cols)},
227
+ )
228
+ est = build_estimation_from_returns(e, StataReturns())
229
+ assert est.statistic_kind == "z"
230
+ assert math.isclose(est.coefficients[0].ci_high, 10.0 + 1.959963984540054 * 2.0)
231
+
232
+ def test_e_level_drives_both_ci_level_and_the_critical_value(self):
233
+ cols = ["x"]
234
+ e = StataReturns(
235
+ scalars={"df_r": 70, "level": 90},
236
+ matrices={"b": _b([10.0], cols), "V": _v([4.0], cols)},
237
+ )
238
+ est = build_estimation_from_returns(e, StataReturns())
239
+ assert est.ci_level == 90.0
240
+ # t(.95, 70) = 1.6669145... — narrower than the 95% interval.
241
+ assert math.isclose(est.coefficients[0].ci_high, 10.0 + 1.6669145 * 2.0, rel_tol=1e-7)
242
+
131
243
  def test_no_vcov_yields_point_estimates_only(self):
132
244
  cols = ["x"]
133
245
  e = StataReturns(matrices={"b": _b([3.0], cols)}) # no V
@@ -123,6 +123,80 @@ class TestEstimationContractReal:
123
123
  assert r.ok, r.error
124
124
  assert r.results.estimation is None
125
125
 
126
+ def test_fallback_table_agrees_with_r_table_to_the_last_digit(self):
127
+ # r(table) is cleared by whatever runs after the estimation, so a block
128
+ # ending in `graph export` / `esttab` / `summarize` lands on the
129
+ # e(b)/e(V) path. Both paths must describe the same regression: this
130
+ # once returned normal-approximation intervals against a log printing
131
+ # `P>|t|`, so structuredContent silently disagreed with the log.
132
+ setup = "sysuse auto, clear\ngen price_k = price/1000\n"
133
+ clean = _run(setup + "regress price_k mpg weight foreign", "rs_rt")
134
+ stale = _run(setup + "regress price_k mpg weight foreign\nsummarize mpg", "rs_bv")
135
+ assert clean.ok and stale.ok, (clean.error, stale.error)
136
+
137
+ assert clean.results.estimation.source == "r_table"
138
+ assert stale.results.estimation.source == "e_b_v"
139
+ assert stale.results.estimation.statistic_kind == "t"
140
+ assert stale.results.estimation.df_resid == 70
141
+
142
+ by_term = {c.term: c for c in clean.results.estimation.coefficients}
143
+ for c in stale.results.estimation.coefficients:
144
+ ref = by_term[c.term]
145
+ assert c.b == pytest.approx(ref.b, rel=1e-12)
146
+ assert c.se == pytest.approx(ref.se, rel=1e-12)
147
+ assert c.statistic == pytest.approx(ref.statistic, rel=1e-9)
148
+ assert c.p_value == pytest.approx(ref.p_value, rel=1e-6, abs=1e-12)
149
+ assert c.ci_low == pytest.approx(ref.ci_low, rel=1e-9)
150
+ assert c.ci_high == pytest.approx(ref.ci_high, rel=1e-9)
151
+
152
+ def test_rebuilt_table_is_flagged_in_warnings(self):
153
+ r = _run("sysuse auto, clear\nregress price mpg\nsummarize mpg", "rs_bvwarn")
154
+ assert r.ok, r.error
155
+ assert r.results.estimation.source == "e_b_v"
156
+ kinds = {w.kind for w in r.warnings}
157
+ assert "estimation_from_e_b_v" in kinds
158
+
159
+ def test_inherited_estimation_does_not_warn_on_later_runs(self):
160
+ # e() is session-global, so `estimation` keeps describing this regress
161
+ # on every subsequent call. Those runs rebuild from e(b)/e(V) too, but
162
+ # they estimated nothing — warning on them would be noise.
163
+ first = _run("sysuse auto, clear\nregress price mpg", "rs_bvquiet")
164
+ assert first.ok, first.error
165
+ later = _run("summarize mpg", "rs_bvquiet")
166
+ assert later.ok, later.error
167
+ assert later.results.estimation is not None # inherited from e()
168
+ assert later.results.estimation.source == "e_b_v"
169
+ assert [w.kind for w in later.warnings] == []
170
+
171
+ def test_estimation_survives_include_results_none(self):
172
+ # include_results governs how much r()/e() is *echoed*; the model-level
173
+ # numbers in `estimation` are include_estimation's business. Reading
174
+ # e() scalars is what makes n_obs / df_resid / model_stats available.
175
+ r = run(
176
+ "sysuse auto, clear\nregress price mpg weight",
177
+ session_id="rs_noresults",
178
+ include_results="none",
179
+ include_full_log=False,
180
+ )
181
+ assert r.ok, r.error
182
+ # Nothing echoed on the wire...
183
+ assert r.results.e.scalars == {} and r.results.e.macros == {}
184
+ # ...but the contract is intact.
185
+ est = r.results.estimation
186
+ assert est is not None
187
+ assert est.n_obs == 74
188
+ assert est.df_resid == 71
189
+ assert est.model_stats.get("r2") == pytest.approx(0.2934, abs=1e-3)
190
+ assert len(est.coefficients) == 3
191
+
192
+ def test_rngstate_macro_is_elided_not_shipped_whole(self):
193
+ r = _run("sysuse auto, clear\nbootstrap r(mean), reps(20): summarize price", "rs_rng")
194
+ assert r.ok, r.error
195
+ rngstate = r.results.e.macros.get("rngstate")
196
+ assert rngstate is not None, "bootstrap should still report the macro name"
197
+ assert len(rngstate) < 400
198
+ assert "chars elided" in rngstate
199
+
126
200
 
127
201
  # ─────────────────────────────────────────────────────────────────────────────
128
202
  # Error taxonomy — real rc codes, labels, recovery, suggestions
@@ -1040,6 +1040,34 @@ class TestProjectReturns:
1040
1040
  )
1041
1041
  assert out.scalars == {} and out.macros == {} and out.matrices == {}
1042
1042
 
1043
+ def test_oversized_macro_is_elided_with_a_marker(self):
1044
+ # e(rngstate) is ~2 KB of hex on every bootstrap / permute / simulate
1045
+ # run and carries nothing an agent can branch on.
1046
+ from stata_code.core.schema import IncludeResults, StataReturns
1047
+
1048
+ rngstate = "X" + "a1b2c3d4" * 260
1049
+ collected = StataReturns(macros={"cmd": "bootstrap", "rngstate": rngstate})
1050
+ out = runner._project_returns(
1051
+ collected, prefix="e", request_id="req-m", mode=IncludeResults.SCALARS
1052
+ )
1053
+ assert out.macros["cmd"] == "bootstrap" # short macros pass through
1054
+ capped = out.macros["rngstate"]
1055
+ assert len(capped) < len(rngstate)
1056
+ assert capped.startswith(rngstate[: runner.MACRO_INLINE_CHAR_CAP])
1057
+ assert "chars elided" in capped
1058
+
1059
+ def test_macro_exactly_at_the_cap_is_untouched(self):
1060
+ from stata_code.core.schema import IncludeResults, StataReturns
1061
+
1062
+ exact = "z" * runner.MACRO_INLINE_CHAR_CAP
1063
+ out = runner._project_returns(
1064
+ StataReturns(macros={"m": exact}),
1065
+ prefix="e",
1066
+ request_id="req-e",
1067
+ mode=IncludeResults.SCALARS,
1068
+ )
1069
+ assert out.macros["m"] == exact
1070
+
1043
1071
 
1044
1072
  class TestCollectDataset:
1045
1073
  def test_full_metadata_with_variables(self):
File without changes
File without changes
File without changes
File without changes
File without changes