stata-code 0.12.1__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {stata_code-0.12.1 → stata_code-0.13.0}/CHANGELOG.md +142 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/PKG-INFO +3 -3
- {stata_code-0.12.1 → stata_code-0.13.0}/README.en.md +1 -1
- {stata_code-0.12.1 → stata_code-0.13.0}/README.md +1 -1
- {stata_code-0.12.1 → stata_code-0.13.0}/SCHEMA.md +27 -7
- {stata_code-0.12.1 → stata_code-0.13.0}/pyproject.toml +1 -1
- {stata_code-0.12.1 → stata_code-0.13.0}/schema/run_result.schema.json +24 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/__init__.py +1 -1
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/console.py +16 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/estimation.py +152 -12
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/jobs.py +16 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/runner.py +123 -10
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/schema.py +48 -1
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/mcp/server.py +1 -1
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_agent_ergonomics.py +43 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_console.py +27 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_estimation.py +114 -2
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_real_stata.py +108 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_runner_helpers.py +28 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_schema.py +44 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/.gitignore +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/LICENSE +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/LICENSE-POLICY.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/PUBLISHING.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/docs/competitive-landscape.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/docs/design/hard_timeout.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/docs/industry-leader-roadmap.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/docs/quickstart.zh.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/01-basic-regression.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/02-did-card-krueger.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/03-graphs.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/04-multi-session.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/05-large-matrix.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/06-cross-stack-parity-audit.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/07-data-mcp-handoff.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/examples/README.md +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/scripts/build_skill_zip.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/scripts/build_standalone.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/scripts/check_github_actions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/scripts/check_versions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/scripts/export_schema.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/__main__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/cli.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/_pool.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/_refs.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/_runtime.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/daemon.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/errors.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/handoff.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/lint.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/log_artifacts.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/notebook.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/policy.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/provenance.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/core/run_index.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/doctor.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/kernel/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/kernel/__main__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/kernel/assets/logo-32x32.png +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/kernel/assets/logo-64x64.png +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/kernel/assets/logo-svg.svg +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/kernel/kernel.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/mcp/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/mcp/__main__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/stata_code/mcp_setup.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/__init__.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/conftest.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/fixtures/.gitkeep +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_benchmark.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_bugfix_regressions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_cancel.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_cli.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_daemon.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_doctor.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_errors.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_github_actions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_handoff.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_kernel.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_lint.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_lint_mcp.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_log_artifacts.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_mcp.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_mcp_kernel_extra.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_mcp_transport.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_method_prompts.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_new_tools.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_notebook.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_notebook_phase2.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_policy.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_pool.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_pool_refs_extra.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_provenance.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_public_api.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_release_versions.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_run_index.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_runner.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_runtime_discovery.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_schema_artifact.py +0 -0
- {stata_code-0.12.1 → stata_code-0.13.0}/tests/test_skill_package.py +0 -0
|
@@ -6,6 +6,148 @@ to semver-major.minor for the result schema (see `SCHEMA.md` §6).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## 0.13.0 — 2026-10-03
|
|
10
|
+
|
|
11
|
+
A data viewer for `.dta` files. Opening a dataset in VS Code used to mean
|
|
12
|
+
either launching Stata or converting to CSV and losing the variable labels,
|
|
13
|
+
value labels, formats and notes on the way. The extension now reads `.dta`
|
|
14
|
+
directly and shows all of it.
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
|
|
18
|
+
- **VS Code: a Stata-free `.dta` data viewer.** Double-click any `.dta` file
|
|
19
|
+
and it opens in a read-only grid, registered as the default editor for
|
|
20
|
+
`*.dta`. The file is parsed in the extension itself, from StataCorp's
|
|
21
|
+
published format documentation, so it needs neither Stata, Python, nor the
|
|
22
|
+
MCP server. What the viewer keeps that a CSV round trip drops:
|
|
23
|
+
- variable labels (under each column name and in the variables panel);
|
|
24
|
+
- value labels, shown in place of their codes, with a *Value labels* toggle
|
|
25
|
+
to see the underlying numbers;
|
|
26
|
+
- display formats — `%td` / `%tc` / `%tm` / `%tq` / `%th` / `%tw` / `%ty`
|
|
27
|
+
dates including custom detail codes (`%tdCCYY-NN-DD`), `%w.df`, `%w.de`,
|
|
28
|
+
`%w.dg`, comma variants — printed as Stata prints them;
|
|
29
|
+
- the 27 missing values as `.`, `.a` … `.z`, including value labels attached
|
|
30
|
+
to extended missing values;
|
|
31
|
+
- dataset and variable notes, the dataset label, sort order and timestamp;
|
|
32
|
+
- `strL` variables and non-ASCII text.
|
|
33
|
+
|
|
34
|
+
Observations are fixed-width records, so the viewer reads only the rows on
|
|
35
|
+
screen; a 3-million-row file opens and jumps to its last row as fast as a
|
|
36
|
+
5-row one. A variables panel filters by name or label and jumps to a column;
|
|
37
|
+
*Go to row*, arrow / Page / Home / End keys and `Cmd/Ctrl+C` work as
|
|
38
|
+
expected; the viewer reloads when Stata re-saves the file.
|
|
39
|
+
|
|
40
|
+
Formats covered: 113–115 (Stata 8–12), 117 (Stata 13), 118 / 119
|
|
41
|
+
(Stata 14+), 120 / 121 (Stata 18 alias variables), in either byte order.
|
|
42
|
+
The parser is tested against fixtures written by a real Stata, and its
|
|
43
|
+
formatted cells are asserted equal to Stata's own `list` output.
|
|
44
|
+
- **VS Code: `stataCode.dtaLegacyEncoding` setting** (default `auto`) for
|
|
45
|
+
`.dta` formats older than 118, which predate UTF-8. `auto` tries UTF-8,
|
|
46
|
+
then GB18030, then Windows-1252.
|
|
47
|
+
- **`dataset.variables[*].format` and `.value_label`.** Each variable now
|
|
48
|
+
reports its display format when that differs from the storage type's
|
|
49
|
+
default, and the name of its value label when one is attached — so a
|
|
50
|
+
consumer can tell that an integer is a `%td` date or that `foreign` is
|
|
51
|
+
labelled by `origin` without running `describe`. Both keys are *omitted*
|
|
52
|
+
(not `null`) when they have nothing to say, so the common variable costs no
|
|
53
|
+
extra tokens. Additive; `schema_version` stays `1.0`. Reported by both the
|
|
54
|
+
pystata and the console backend.
|
|
55
|
+
- **VS Code: the Data sidebar shows those two fields** next to each variable
|
|
56
|
+
(`long %td · Sale date`, `byte · [origin] · Car origin`).
|
|
57
|
+
|
|
58
|
+
### Changed
|
|
59
|
+
|
|
60
|
+
- **VS Code: "View data preview" opens the data viewer instead of a text
|
|
61
|
+
listing.** The first `stataCode.dataPreviewObs` observations in memory are
|
|
62
|
+
copied through a scratch frame into a temporary `.dta` and shown in the same
|
|
63
|
+
grid as a file on disk, so in-memory data gets labels, formats and notes
|
|
64
|
+
too. The user's frame is not touched — not its data, sort order,
|
|
65
|
+
`c(filename)`, `c(changed)`, nor `r()` — and the temporary file is deleted
|
|
66
|
+
when the panel closes. One panel per session; previewing again refreshes it.
|
|
67
|
+
On Stata 15 or older (no frames) and with the console backend the command
|
|
68
|
+
falls back to the text listing below.
|
|
69
|
+
- **VS Code: `stataCode.dataPreviewObs` now defaults to `1000`** (was `50`)
|
|
70
|
+
and accepts up to `100000` (was `10000`): the grid scrolls, a text document
|
|
71
|
+
does not. The text fallback is capped at 200 rows regardless.
|
|
72
|
+
- **VS Code: the fallback text listing is shorter and no longer line-wrapped.**
|
|
73
|
+
`list` is run with a widened `linesize` (restored afterwards) so wide
|
|
74
|
+
datasets stop folding into `>` continuation lines; the command echo is
|
|
75
|
+
stripped from the body; the header collapses to a session line plus a
|
|
76
|
+
`74 obs x 12 vars - showing all 74` summary and the dataset path; and the
|
|
77
|
+
variable list is column-aligned instead of tab-separated.
|
|
78
|
+
- **VS Code: utility runs (data preview, `pwd`, `cd`) no longer hijack the
|
|
79
|
+
Output panel.** They used to force it visible and echo their whole log into
|
|
80
|
+
it. They now log one status line on success and only surface the panel on
|
|
81
|
+
failure.
|
|
82
|
+
|
|
83
|
+
### Fixed
|
|
84
|
+
|
|
85
|
+
- **VS Code: "View data preview" failed on every dataset with fewer than 100
|
|
86
|
+
observations.** The preview listed rows with `list in 1/100`, and an `in`
|
|
87
|
+
range whose upper bound exceeds `_N` is an error in Stata, so the command
|
|
88
|
+
returned `r(198) observation numbers out of range` for `auto.dta` (74 obs)
|
|
89
|
+
and most other teaching datasets — while the document header still claimed
|
|
90
|
+
`showing: 0 of 74 observations`. The preview now selects rows with
|
|
91
|
+
`if _n <= N`, which lists whatever is there (including nothing at all), and
|
|
92
|
+
a failed preview reports its `rc` and message instead of a row count.
|
|
93
|
+
|
|
94
|
+
## 0.12.2 — 2026-08-11
|
|
95
|
+
|
|
96
|
+
Correctness release. `results.estimation` could report confidence intervals
|
|
97
|
+
and p-values that disagreed with the Stata log printed beside them; if you
|
|
98
|
+
read coefficients out of `results.estimation` rather than the log, upgrade.
|
|
99
|
+
|
|
100
|
+
### Fixed
|
|
101
|
+
|
|
102
|
+
- **`results.estimation` reported normal-approximation inference for `t` models
|
|
103
|
+
whenever `r(table)` had been cleared.** `r(table)` only survives until the
|
|
104
|
+
next command, so any block that runs something after its estimation —
|
|
105
|
+
`graph export`, `esttab`, a final `summarize` — fell back to rebuilding the
|
|
106
|
+
table from `e(b)` / `e(V)`, and that path hardcoded a normal approximation.
|
|
107
|
+
The result was a payload that quietly disagreed with the log beside it:
|
|
108
|
+
|
|
109
|
+
```text
|
|
110
|
+
regress price_k mpg weight foreign
|
|
111
|
+
scatter price_k mpg // clears r(table)
|
|
112
|
+
|
|
113
|
+
log: mpg 95% CI [-.1261758, .169883] P>|t| 0.769
|
|
114
|
+
results.estimation: mpg 95% CI [-0.12362, 0.16732] p 0.7684
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
The fallback now follows Stata's own rule — a *t* table on `e(df_r)` degrees
|
|
118
|
+
of freedom when that scalar is set, *z* otherwise — so the rebuilt table
|
|
119
|
+
matches the printed one to the last digit. `ci_level` also honours `e(level)`
|
|
120
|
+
instead of assuming 95, so `regress, level(90)` is no longer mislabelled.
|
|
121
|
+
Agents told to prefer `structuredContent` over the log were the ones exposed
|
|
122
|
+
to this; the `r(table)` path was always correct.
|
|
123
|
+
|
|
124
|
+
A new `estimation_from_e_b_v` warning records when the rebuild happened. It
|
|
125
|
+
fires only for an estimation produced by the current run, since `e()` is
|
|
126
|
+
session-global and would otherwise re-warn on every later call.
|
|
127
|
+
|
|
128
|
+
- **`include_results: "none"` hollowed out `results.estimation`.** `n_obs`,
|
|
129
|
+
`df_model`, `df_resid` and `model_stats` are all read from `e()` scalars, so
|
|
130
|
+
suppressing the `r()` / `e()` *echo* also removed the model-level numbers —
|
|
131
|
+
even though `include_estimation` is the documented knob for that block, and
|
|
132
|
+
SCHEMA.md §4 already promised `include_results` "never affects
|
|
133
|
+
`results.estimation`". The scalars are now always read when estimation is
|
|
134
|
+
wanted, and only withheld from the wire.
|
|
135
|
+
|
|
136
|
+
- **A finished background run reported a duration that kept growing.**
|
|
137
|
+
`Job.elapsed_ms` recomputed `now - submitted` on every poll instead of
|
|
138
|
+
freezing at completion, so a 1.7 s bootstrap polled ten minutes later
|
|
139
|
+
reported ten minutes. Callers attributing wall-clock time to Stata were
|
|
140
|
+
reading their own polling delay. The value is now frozen before the terminal
|
|
141
|
+
status is published, on both the success and error paths.
|
|
142
|
+
|
|
143
|
+
### Changed
|
|
144
|
+
|
|
145
|
+
- **Macro values longer than 256 characters are elided on the wire** with an
|
|
146
|
+
explicit `… (N more chars elided)` marker. This is aimed at `e(rngstate)`,
|
|
147
|
+
roughly 2 KB of hex emitted on every `bootstrap` / `permute` / `simulate`
|
|
148
|
+
run that no consumer can act on. Names are always kept, and the estimation
|
|
149
|
+
contract is still built from the uncapped values.
|
|
150
|
+
|
|
9
151
|
## 0.12.1 — 2026-07-28
|
|
10
152
|
|
|
11
153
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: stata-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)
|
|
5
5
|
Project-URL: Homepage, https://github.com/brycewang-stanford/stata-code
|
|
6
6
|
Project-URL: Repository, https://github.com/brycewang-stanford/stata-code
|
|
@@ -527,7 +527,7 @@ Then open Jupyter Notebook / JupyterLab (or a `.ipynb` in VS Code), pick **Stata
|
|
|
527
527
|
|
|
528
528
|
### As a VS Code Extension
|
|
529
529
|
|
|
530
|
-
The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors.
|
|
530
|
+
The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors. It also registers a **Stata-free `.dta` data viewer**: double-click any `.dta` file to browse it in a grid with its variable labels, value labels, display formats, notes and missing-value codes intact, reading only the rows on screen so multi-gigabyte files open instantly.
|
|
531
531
|
|
|
532
532
|
```bash
|
|
533
533
|
# from the VS Code CLI
|
|
@@ -488,7 +488,7 @@ Then open Jupyter Notebook / JupyterLab (or a `.ipynb` in VS Code), pick **Stata
|
|
|
488
488
|
|
|
489
489
|
### As a VS Code Extension
|
|
490
490
|
|
|
491
|
-
The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors.
|
|
491
|
+
The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors. It also registers a **Stata-free `.dta` data viewer**: double-click any `.dta` file to browse it in a grid with its variable labels, value labels, display formats, notes and missing-value codes intact, reading only the rows on screen so multi-gigabyte files open instantly.
|
|
492
492
|
|
|
493
493
|
```bash
|
|
494
494
|
# from the VS Code CLI
|
|
@@ -456,7 +456,7 @@ jupyter kernelspec list
|
|
|
456
456
|
|
|
457
457
|
### 作为 VS Code 扩展
|
|
458
458
|
|
|
459
|
-
配套扩展已发布到 Marketplace:[`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode)。它会以子进程方式启动 `stata-code-mcp`,并提供语法高亮、`**#` section 和 `program define` 的 Outline、`.do` 文件的 code-lens "Run cell" / "Run section"、**七视图侧边栏**(sessions / last result / **data 变量浏览器** / run history / logs / graphs / **outputs**)——其中包含一个 agent-native 版的 Stata **变量窗口**,以及一个把每次运行写到磁盘的 `esttab` 表格和 `export` 文件呈现出来的 **Outputs** 面板——状态栏指示器、补全、帮助跳转、保守变量重命名,以及来自 v1.0 typed errors
|
|
459
|
+
配套扩展已发布到 Marketplace:[`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode)。它会以子进程方式启动 `stata-code-mcp`,并提供语法高亮、`**#` section 和 `program define` 的 Outline、`.do` 文件的 code-lens "Run cell" / "Run section"、**七视图侧边栏**(sessions / last result / **data 变量浏览器** / run history / logs / graphs / **outputs**)——其中包含一个 agent-native 版的 Stata **变量窗口**,以及一个把每次运行写到磁盘的 `esttab` 表格和 `export` 文件呈现出来的 **Outputs** 面板——状态栏指示器、补全、帮助跳转、保守变量重命名,以及来自 v1.0 typed errors 的内联诊断。扩展还注册了一个**不依赖 Stata 的 `.dta` 数据浏览器**:双击任意 `.dta` 文件即可在表格中浏览,变量标签、值标签、显示格式、notes 和缺失值代码全部保留;只读取屏幕上可见的行,几个 GB 的文件也能瞬间打开。
|
|
460
460
|
|
|
461
461
|
```bash
|
|
462
462
|
# 从 VS Code 命令行
|
|
@@ -343,7 +343,7 @@ Stata's `r()` and `e()` return dictionaries, structurally separated. Each follow
|
|
|
343
343
|
| Sub-field | Type | Notes |
|
|
344
344
|
| --- | --- | --- |
|
|
345
345
|
| `scalars` | `dict<str, number \| null>` | Native floats / ints. Stata's system missing (`.`) → JSON `null`. Extended missings (`.a`–`.z`) → `null` with information loss. |
|
|
346
|
-
| `macros` | `dict<str, string>` | Stata macro values verbatim. |
|
|
346
|
+
| `macros` | `dict<str, string>` | Stata macro values verbatim, except that a value longer than 256 characters is truncated and suffixed with `… (N more chars elided)`. The cap exists for macros like `e(rngstate)`, which is ~2 KB of hex on every `bootstrap` / `permute` / `simulate` run and carries nothing a consumer can act on. The name is always kept, and `results.estimation` is derived from the uncapped values. |
|
|
347
347
|
| `matrices` | `dict<str, Matrix>` | See `Matrix` below. |
|
|
348
348
|
|
|
349
349
|
**`Matrix` shape:**
|
|
@@ -383,9 +383,9 @@ Stata's `r()` and `e()` return dictionaries, structurally separated. Each follow
|
|
|
383
383
|
| `n_obs` | `int \| null` | Integer form of `e(N)` when available. |
|
|
384
384
|
| `df_model` | `number \| null` | Mirrors `e(df_m)`. |
|
|
385
385
|
| `df_resid` | `number \| null` | Mirrors `e(df_r)`. |
|
|
386
|
-
| `statistic_kind` | `"t" \| "z"` | Which statistic fills each coefficient's `statistic` field. |
|
|
387
|
-
| `source` | `"r_table" \| "e_b_v"` | `r_table` means values were copied from Stata's displayed `r(table)` after verifying its columns and `b` row match `e(b)`; `e_b_v` means point estimates come from `e(b)` and inference, when present, is computed from `e(V)
|
|
388
|
-
| `ci_level` | `number` | Confidence level used for `ci_low` / `ci_high`;
|
|
386
|
+
| `statistic_kind` | `"t" \| "z"` | Which statistic fills each coefficient's `statistic` field, and which distribution produced `p_value` / `ci_low` / `ci_high`. On the `e_b_v` path this follows Stata's own rule — `t` on `df_resid` degrees of freedom when `e(df_r)` is set, `z` otherwise — so a rebuilt table agrees with the printed log rather than reporting a normal-approximation interval next to a `P>\|t\|` column. |
|
|
387
|
+
| `source` | `"r_table" \| "e_b_v"` | `r_table` means values were copied from Stata's displayed `r(table)` after verifying its columns and `b` row match `e(b)`; `e_b_v` means point estimates come from `e(b)` and inference, when present, is computed from `e(V)`. A matrix returned by `ref` is resolved before use, so a deferred `e(V)` still yields standard errors. |
|
|
388
|
+
| `ci_level` | `number` | Confidence level used for `ci_low` / `ci_high`. Mirrors `e(level)` when the command stored it, so `regress, level(90)` reports `90.0`; defaults to `95.0` otherwise. |
|
|
389
389
|
| `coefficients` | `array<Coefficient>` | One row per term in `e(b)`, subject to the caller's `include_estimation` / `max_coefficients` budget. |
|
|
390
390
|
| `n_coefficients` | `int` | The model's true term count. Equals `coefficients.length` unless the caller trimmed the table, so `12` rows out of `n_coefficients: 141` is never mistaken for a 12-term model. |
|
|
391
391
|
| `coefficients_truncated` | `bool` | `true` when rows were dropped to satisfy the budget. |
|
|
@@ -425,9 +425,19 @@ A summary of the active Stata frame *after* the command ran. Always populated.
|
|
|
425
425
|
|
|
426
426
|
```json
|
|
427
427
|
{ "name": "mpg", "type": "int", "label": "Mileage (mpg)" }
|
|
428
|
+
{ "name": "foreign", "type": "byte", "label": "Car origin", "value_label": "origin" }
|
|
429
|
+
{ "name": "saledate", "type": "long", "label": "Sale date", "format": "%td" }
|
|
428
430
|
```
|
|
429
431
|
|
|
430
|
-
|
|
432
|
+
| Field | Type | Notes |
|
|
433
|
+
| --- | --- | --- |
|
|
434
|
+
| `name` | `string` | Variable name. |
|
|
435
|
+
| `type` | `string` | Stata's storage type (`byte`, `int`, `long`, `float`, `double`, `str#`, `strL`). |
|
|
436
|
+
| `label` | `string` | The variable label, or `""` if none. |
|
|
437
|
+
| `format` | `string`, optional | The display format, **only when it differs from the storage type's default** (`%8.0g` for `byte`/`int`, `%12.0g` for `long`, `%9.0g` for `float`, `%10.0g` for `double`, `%{max(9,#)}s` for `str#`, `%9s` for `strL`). Absent means "the default". This is how a consumer learns that an integer is a date (`%td`, `%tc`, `%tm`, …). |
|
|
438
|
+
| `value_label` | `string`, optional | Name of the value label attached to the variable. Absent when none is attached. The mapping itself is not shipped; run `label list <name>`. |
|
|
439
|
+
|
|
440
|
+
`format` and `value_label` are **omitted, not `null`**, when they have nothing to say: the variable list rides along on every run, so a null on most variables would cost tokens to carry no information. Added in 0.13 (additive; `schema_version` stays `"1.0"`). Consumers must treat a missing key and `null` alike.
|
|
431
441
|
|
|
432
442
|
When `n_vars` is large (default cap: 200), the producer truncates `variables` to the first 200 entries and emits a warning of kind `dataset_variables_truncated`. Agents wanting all variables should call `describe` directly.
|
|
433
443
|
|
|
@@ -564,11 +574,21 @@ The rc-to-kind table is approximate and lives in code (`stata_code.core.errors`)
|
|
|
564
574
|
|
|
565
575
|
| Field | Type | Notes |
|
|
566
576
|
| --- | --- | --- |
|
|
567
|
-
| `kind` | `string` | Open enum. Common values: `convergence`, `singular`, `boundary`, `omitted_collinear`, `non_finite`, `dataset_variables_truncated`, `unknown`. |
|
|
577
|
+
| `kind` | `string` | Open enum. Common values: `convergence`, `singular`, `boundary`, `omitted_collinear`, `non_finite`, `dataset_variables_truncated`, `log_closed`, `output_tracking_skipped`, `estimation_from_e_b_v`, `unknown`. |
|
|
568
578
|
| `message` | `string` | Human-readable, single line. Truncated to 1,024 characters. |
|
|
569
579
|
|
|
570
580
|
Warnings are de-duplicated by `(kind, message)`.
|
|
571
581
|
|
|
582
|
+
`estimation_from_e_b_v` reports that this run performed an estimation whose
|
|
583
|
+
`r(table)` was already gone by the time results were read — a later command in
|
|
584
|
+
the same submission cleared it — so `results.estimation` was rebuilt from
|
|
585
|
+
`e(b)` / `e(V)`. The rebuilt numbers still match the printed log (see
|
|
586
|
+
`statistic_kind` in §3.5), so this is provenance rather than a correctness
|
|
587
|
+
alarm; putting the estimation last in the block restores the `r_table` path.
|
|
588
|
+
It is emitted only when the estimation was produced by *this* run: `e()` is
|
|
589
|
+
session-global, so a later `summarize` in the same session keeps reporting the
|
|
590
|
+
inherited table through the same fallback without re-warning.
|
|
591
|
+
|
|
572
592
|
### 3.9 `origin`
|
|
573
593
|
|
|
574
594
|
```json
|
|
@@ -607,7 +627,7 @@ The schema also dictates what callers may *ask for*. Every frontend exposes the
|
|
|
607
627
|
| `include_graphs` | `"ref" \| "inline" \| "none"` | `"ref"` | `"none"` skips graph capture entirely (cheapest); `"ref"` captures and returns refs; `"inline"` base64-encodes bytes into `inline`. |
|
|
608
628
|
| `graph_format` | `"png" \| "svg" \| "pdf"` | `"png"` | Render format. |
|
|
609
629
|
| `include_dataset_variables` | `bool` | `true` | Set `false` to omit `dataset.variables`. |
|
|
610
|
-
| `include_results` | `"none" \| "scalars" \| "full"` | `"scalars"` | Payload budget for `results.r` / `results.e`. `"scalars"` inlines scalars and macros and emits every matrix as a stub (§3.4); `"full"` inlines matrix values up to the ~10,000-cell cap; `"none"` omits `r()` / `e()` entirely. Never affects `results.estimation
|
|
630
|
+
| `include_results` | `"none" \| "scalars" \| "full"` | `"scalars"` | Payload budget for `results.r` / `results.e`. `"scalars"` inlines scalars and macros and emits every matrix as a stub (§3.4); `"full"` inlines matrix values up to the ~10,000-cell cap; `"none"` omits `r()` / `e()` entirely. Never affects `results.estimation`: the model-level fields (`n_obs`, `df_model`, `df_resid`, `model_stats`, `depvar`) are read from `e()` regardless of this setting and merely withheld from the wire, so `"none"` does not hollow out the estimation contract. Use `include_estimation` to trim that block. |
|
|
611
631
|
| `include_estimation` | `"none" \| "summary" \| "full"` | `"full"` | Payload budget for `results.estimation`. `"summary"` keeps the model-level block and drops per-term rows. |
|
|
612
632
|
| `max_coefficients` | `int \| null` | `null` | Cap on `estimation.coefficients` rows. `n_coefficients` still reports the true count. |
|
|
613
633
|
| `timeout_ms` | `int \| null` | `600000` (10 min) | Hard timeout. `null` disables. On expiry, returns `ok: false`, `error.kind: "timeout"`, `rc: -2`. The budget covers **queueing**: a call waiting on a session whose Stata process is mid-run returns `rc: -5`, `error.kind: "session_busy"` rather than blocking past its deadline. Frontends MAY override the default if their use case demands. |
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "stata-code"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.13.0"
|
|
8
8
|
description = "Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)"
|
|
9
9
|
# The repo default README (README.md) is Chinese; keep the PyPI long description
|
|
10
10
|
# English by pointing at README.en.md.
|
|
@@ -1145,6 +1145,18 @@
|
|
|
1145
1145
|
"VariableInfo": {
|
|
1146
1146
|
"additionalProperties": true,
|
|
1147
1147
|
"properties": {
|
|
1148
|
+
"format": {
|
|
1149
|
+
"anyOf": [
|
|
1150
|
+
{
|
|
1151
|
+
"type": "string"
|
|
1152
|
+
},
|
|
1153
|
+
{
|
|
1154
|
+
"type": "null"
|
|
1155
|
+
}
|
|
1156
|
+
],
|
|
1157
|
+
"default": null,
|
|
1158
|
+
"title": "Format"
|
|
1159
|
+
},
|
|
1148
1160
|
"label": {
|
|
1149
1161
|
"default": "",
|
|
1150
1162
|
"title": "Label",
|
|
@@ -1157,6 +1169,18 @@
|
|
|
1157
1169
|
"type": {
|
|
1158
1170
|
"title": "Type",
|
|
1159
1171
|
"type": "string"
|
|
1172
|
+
},
|
|
1173
|
+
"value_label": {
|
|
1174
|
+
"anyOf": [
|
|
1175
|
+
{
|
|
1176
|
+
"type": "string"
|
|
1177
|
+
},
|
|
1178
|
+
{
|
|
1179
|
+
"type": "null"
|
|
1180
|
+
}
|
|
1181
|
+
],
|
|
1182
|
+
"default": null,
|
|
1183
|
+
"title": "Value Label"
|
|
1160
1184
|
}
|
|
1161
1185
|
},
|
|
1162
1186
|
"required": [
|
|
@@ -57,6 +57,7 @@ from stata_code.core.schema import (
|
|
|
57
57
|
StataInfo,
|
|
58
58
|
StataReturns,
|
|
59
59
|
VariableInfo,
|
|
60
|
+
default_display_format,
|
|
60
61
|
)
|
|
61
62
|
|
|
62
63
|
|
|
@@ -217,6 +218,7 @@ MARK_SECTION = f"{_M}|SECTION|"
|
|
|
217
218
|
MARK_MATRIX = f"{_M}|MATRIX|"
|
|
218
219
|
MARK_DS = f"{_M}|DS|"
|
|
219
220
|
MARK_VAR = f"{_M}|VAR|"
|
|
221
|
+
MARK_VARFMT = f"{_M}|VARFMT|"
|
|
220
222
|
MARK_BEGIN = f"{_M}|BEGIN"
|
|
221
223
|
MARK_END = f"{_M}|END"
|
|
222
224
|
|
|
@@ -274,6 +276,9 @@ def build_wrapper_do(code: str, *, working_dir: str | None = None) -> str:
|
|
|
274
276
|
# Stata: display "<MARK>`__v'|`: type `__v''|`: variable label `__v''"
|
|
275
277
|
# Built by concatenation to avoid f-string quote/backtick collisions.
|
|
276
278
|
" display \"" + MARK_VAR + "`__v'|`: type `__v''|`: variable label `__v''\"",
|
|
279
|
+
# A second line rather than two more fields on the first: a variable
|
|
280
|
+
# label may itself contain "|", so it has to stay the last field.
|
|
281
|
+
" display \"" + MARK_VARFMT + "`__v'|`: format `__v''|`: value label `__v''\"",
|
|
277
282
|
" }",
|
|
278
283
|
"}",
|
|
279
284
|
f'display "{MARK_END}"',
|
|
@@ -298,6 +303,7 @@ _SCALAR_RE = re.compile(r"^\s*[re]\(([A-Za-z_][A-Za-z0-9_]*)\)\s*=\s*(.+?)\s*$")
|
|
|
298
303
|
_MACRO_RE = re.compile(r'^\s*[re]\(([A-Za-z_][A-Za-z0-9_]*)\)\s*:\s*"?(.*?)"?\s*$')
|
|
299
304
|
_DS_RE = re.compile(r"^\s*" + re.escape(MARK_DS) + r"(\w+)\|(.*)$")
|
|
300
305
|
_VAR_RE = re.compile(r"^\s*" + re.escape(MARK_VAR) + r"(.+?)\|(.*?)\|(.*)$")
|
|
306
|
+
_VARFMT_RE = re.compile(r"^\s*" + re.escape(MARK_VARFMT) + r"(.+?)\|(.*?)\|(.*)$")
|
|
301
307
|
# `matrix list` dimension header: e(b)[1,2] or symmetric e(V)[2,2]
|
|
302
308
|
_MATRIX_HEADER_RE = re.compile(r"^\s*(symmetric\s+)?[A-Za-z_][A-Za-z0-9_]*\([^)]*\)\[\d+,\d+\]")
|
|
303
309
|
|
|
@@ -536,6 +542,16 @@ def _parse_dataset(block: str) -> DatasetInfo:
|
|
|
536
542
|
label=v.group(3).strip(),
|
|
537
543
|
)
|
|
538
544
|
)
|
|
545
|
+
continue
|
|
546
|
+
x = _VARFMT_RE.search(raw)
|
|
547
|
+
if x and variables and variables[-1].name == x.group(1).strip():
|
|
548
|
+
# Same contract as the pystata backend: report a format only when
|
|
549
|
+
# it is not the storage type's default, a value label only when set.
|
|
550
|
+
var = variables[-1]
|
|
551
|
+
fmt = x.group(2).strip()
|
|
552
|
+
if fmt and fmt != default_display_format(var.type):
|
|
553
|
+
var.format = fmt
|
|
554
|
+
var.value_label = x.group(3).strip() or None
|
|
539
555
|
return DatasetInfo(
|
|
540
556
|
frame="default",
|
|
541
557
|
n_obs=n_obs,
|
|
@@ -9,8 +9,18 @@ parse from log prose:
|
|
|
9
9
|
exactly what Stata printed, so we copy them verbatim.
|
|
10
10
|
* ``e(b)`` / ``e(V)`` — the coefficient vector and its variance–covariance
|
|
11
11
|
matrix. Always present after an estimation command. We use these as a
|
|
12
|
-
fallback, computing ``se`` from ``diag(V)`` and
|
|
13
|
-
|
|
12
|
+
fallback, computing ``se`` from ``diag(V)`` and the statistic / p-value / CI
|
|
13
|
+
ourselves (flagged via ``source="e_b_v"``).
|
|
14
|
+
|
|
15
|
+
The fallback matters more than "fallback" suggests: ``r(table)`` is wiped by
|
|
16
|
+
the *next* command, so any block that ends in ``graph export`` / ``esttab`` /
|
|
17
|
+
``summarize`` after its estimation lands here. It therefore has to reproduce
|
|
18
|
+
what Stata displayed, not merely something defensible. Stata's own rule (see
|
|
19
|
+
``_coef_table``) is distributional: when ``e(df_r)`` is set the table is a
|
|
20
|
+
*t* table on that many residual degrees of freedom, and otherwise a *z* table.
|
|
21
|
+
We follow exactly that rule, so the fallback CI agrees with the printed log
|
|
22
|
+
digit for digit instead of quietly reporting a normal-approximation interval
|
|
23
|
+
alongside a log that says ``P>|t|``.
|
|
14
24
|
|
|
15
25
|
The result is :class:`EstimationResult`, attached to ``RunResult.results.
|
|
16
26
|
estimation`` so every frontend (MCP, kernel, VS Code) gets the same typed
|
|
@@ -22,6 +32,7 @@ from __future__ import annotations
|
|
|
22
32
|
|
|
23
33
|
import math
|
|
24
34
|
from collections.abc import Callable
|
|
35
|
+
from functools import lru_cache
|
|
25
36
|
|
|
26
37
|
from stata_code.core.schema import (
|
|
27
38
|
Coefficient,
|
|
@@ -162,6 +173,111 @@ def _two_sided_normal_p(z: float) -> float:
|
|
|
162
173
|
return 2.0 * (1.0 - _normal_cdf(abs(z)))
|
|
163
174
|
|
|
164
175
|
|
|
176
|
+
def _betacf(a: float, b: float, x: float) -> float:
|
|
177
|
+
"""Continued fraction for the incomplete beta function (Lentz's method)."""
|
|
178
|
+
tiny = 1e-30
|
|
179
|
+
qab, qap, qam = a + b, a + 1.0, a - 1.0
|
|
180
|
+
c = 1.0
|
|
181
|
+
d = 1.0 - qab * x / qap
|
|
182
|
+
if abs(d) < tiny:
|
|
183
|
+
d = tiny
|
|
184
|
+
d = 1.0 / d
|
|
185
|
+
h = d
|
|
186
|
+
for m in range(1, 301):
|
|
187
|
+
m2 = 2 * m
|
|
188
|
+
aa = m * (b - m) * x / ((qam + m2) * (a + m2))
|
|
189
|
+
d = 1.0 + aa * d
|
|
190
|
+
if abs(d) < tiny:
|
|
191
|
+
d = tiny
|
|
192
|
+
c = 1.0 + aa / c
|
|
193
|
+
if abs(c) < tiny:
|
|
194
|
+
c = tiny
|
|
195
|
+
d = 1.0 / d
|
|
196
|
+
h *= d * c
|
|
197
|
+
aa = -(a + m) * (qab + m) * x / ((a + m2) * (qap + m2))
|
|
198
|
+
d = 1.0 + aa * d
|
|
199
|
+
if abs(d) < tiny:
|
|
200
|
+
d = tiny
|
|
201
|
+
c = 1.0 + aa / c
|
|
202
|
+
if abs(c) < tiny:
|
|
203
|
+
c = tiny
|
|
204
|
+
d = 1.0 / d
|
|
205
|
+
delta = d * c
|
|
206
|
+
h *= delta
|
|
207
|
+
if abs(delta - 1.0) < 1e-15:
|
|
208
|
+
break
|
|
209
|
+
return h
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _betainc_reg(a: float, b: float, x: float) -> float:
|
|
213
|
+
"""Regularized incomplete beta ``I_x(a, b)``, stdlib only."""
|
|
214
|
+
if x <= 0.0:
|
|
215
|
+
return 0.0
|
|
216
|
+
if x >= 1.0:
|
|
217
|
+
return 1.0
|
|
218
|
+
ln_front = (
|
|
219
|
+
math.lgamma(a + b) - math.lgamma(a) - math.lgamma(b) + a * math.log(x) + b * math.log1p(-x)
|
|
220
|
+
)
|
|
221
|
+
front = math.exp(ln_front)
|
|
222
|
+
if x < (a + 1.0) / (a + b + 2.0):
|
|
223
|
+
return front * _betacf(a, b, x) / a
|
|
224
|
+
return 1.0 - front * _betacf(b, a, 1.0 - x) / b
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _two_sided_t_p(t: float, df: float) -> float:
|
|
228
|
+
"""Two-sided p-value for a t statistic on ``df`` degrees of freedom.
|
|
229
|
+
|
|
230
|
+
``P(|T| > |t|) = I_{df/(df+t²)}(df/2, 1/2)`` — exact, no table lookup.
|
|
231
|
+
"""
|
|
232
|
+
if df <= 0.0:
|
|
233
|
+
return _two_sided_normal_p(t)
|
|
234
|
+
if math.isinf(t):
|
|
235
|
+
return 0.0
|
|
236
|
+
return _betainc_reg(df / 2.0, 0.5, df / (df + t * t))
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
@lru_cache(maxsize=256)
|
|
240
|
+
def _t_crit(df: float, level: float) -> float:
|
|
241
|
+
"""Two-sided critical value of the t distribution for a CI ``level`` in %.
|
|
242
|
+
|
|
243
|
+
Found by bisection on :func:`_two_sided_t_p`, which is strictly decreasing
|
|
244
|
+
in ``|t|`` — robust for any ``df`` without shipping a quantile table.
|
|
245
|
+
"""
|
|
246
|
+
alpha = 1.0 - level / 100.0
|
|
247
|
+
if df <= 0.0 or not (0.0 < alpha < 1.0):
|
|
248
|
+
return _z_crit(level)
|
|
249
|
+
lo, hi = 0.0, 1.0
|
|
250
|
+
while _two_sided_t_p(hi, df) > alpha and hi < 1e12:
|
|
251
|
+
hi *= 2.0
|
|
252
|
+
for _ in range(200):
|
|
253
|
+
mid = 0.5 * (lo + hi)
|
|
254
|
+
if _two_sided_t_p(mid, df) > alpha:
|
|
255
|
+
lo = mid
|
|
256
|
+
else:
|
|
257
|
+
hi = mid
|
|
258
|
+
if hi - lo < 1e-13 * max(1.0, hi):
|
|
259
|
+
break
|
|
260
|
+
return 0.5 * (lo + hi)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
@lru_cache(maxsize=64)
|
|
264
|
+
def _z_crit(level: float) -> float:
|
|
265
|
+
"""Two-sided standard-normal critical value for a CI ``level`` in %."""
|
|
266
|
+
alpha = 1.0 - level / 100.0
|
|
267
|
+
if not (0.0 < alpha < 1.0):
|
|
268
|
+
return _Z_CRIT_95
|
|
269
|
+
lo, hi = 0.0, 40.0
|
|
270
|
+
for _ in range(200):
|
|
271
|
+
mid = 0.5 * (lo + hi)
|
|
272
|
+
if _two_sided_normal_p(mid) > alpha:
|
|
273
|
+
lo = mid
|
|
274
|
+
else:
|
|
275
|
+
hi = mid
|
|
276
|
+
if hi - lo < 1e-14 * max(1.0, hi):
|
|
277
|
+
break
|
|
278
|
+
return 0.5 * (lo + hi)
|
|
279
|
+
|
|
280
|
+
|
|
165
281
|
def _cell(row: list[float | None] | None, j: int) -> float | None:
|
|
166
282
|
if row is None or j >= len(row):
|
|
167
283
|
return None
|
|
@@ -284,13 +400,16 @@ def _from_b_v(
|
|
|
284
400
|
v: Matrix | None,
|
|
285
401
|
resolve: MatrixResolver | None = None,
|
|
286
402
|
b_values: list[list[float | None]] | None = None,
|
|
287
|
-
|
|
403
|
+
df_resid: float | None = None,
|
|
404
|
+
level: float = 95.0,
|
|
405
|
+
) -> tuple[list[Coefficient], str]:
|
|
288
406
|
"""Compute coefficient rows from e(b) (and e(V) when available).
|
|
289
407
|
|
|
290
408
|
``se``/``statistic``/``p_value``/CI are filled only when e(V) is
|
|
291
409
|
obtainable — inline or through its ref; otherwise just the point estimates
|
|
292
|
-
are returned.
|
|
293
|
-
|
|
410
|
+
are returned. Returns ``(coefficients, statistic_kind)``: a *t* table on
|
|
411
|
+
``df_resid`` degrees of freedom when Stata set ``e(df_r)``, matching what
|
|
412
|
+
the command printed, and a *z* table otherwise.
|
|
294
413
|
"""
|
|
295
414
|
if b_values is None:
|
|
296
415
|
b_values = _values(b, resolve)
|
|
@@ -305,6 +424,10 @@ def _from_b_v(
|
|
|
305
424
|
else:
|
|
306
425
|
v_diag = [None] * len(terms)
|
|
307
426
|
|
|
427
|
+
df = float(df_resid) if df_resid is not None and df_resid > 0.0 else None
|
|
428
|
+
stat_kind = "t" if df is not None else "z"
|
|
429
|
+
crit = _t_crit(df, level) if df is not None else _z_crit(level)
|
|
430
|
+
|
|
308
431
|
coeffs: list[Coefficient] = []
|
|
309
432
|
for j, term in enumerate(terms):
|
|
310
433
|
b_val = _cell(b_row, j)
|
|
@@ -316,9 +439,9 @@ def _from_b_v(
|
|
|
316
439
|
ci_high: float | None = None
|
|
317
440
|
if b_val is not None and se is not None and se > 0.0:
|
|
318
441
|
stat = b_val / se
|
|
319
|
-
p_val = _two_sided_normal_p(stat)
|
|
320
|
-
ci_low = b_val -
|
|
321
|
-
ci_high = b_val +
|
|
442
|
+
p_val = _two_sided_t_p(stat, df) if df is not None else _two_sided_normal_p(stat)
|
|
443
|
+
ci_low = b_val - crit * se
|
|
444
|
+
ci_high = b_val + crit * se
|
|
322
445
|
coeffs.append(
|
|
323
446
|
Coefficient(
|
|
324
447
|
term=term,
|
|
@@ -330,7 +453,7 @@ def _from_b_v(
|
|
|
330
453
|
ci_high=ci_high,
|
|
331
454
|
)
|
|
332
455
|
)
|
|
333
|
-
return coeffs
|
|
456
|
+
return coeffs, stat_kind
|
|
334
457
|
|
|
335
458
|
|
|
336
459
|
def _model_stats(scalars: dict[str, float | None]) -> dict[str, float | None]:
|
|
@@ -406,14 +529,30 @@ def build_estimation_from_returns(
|
|
|
406
529
|
statistic_kind: str
|
|
407
530
|
source: str
|
|
408
531
|
|
|
532
|
+
df_resid = e.scalars.get("df_r")
|
|
533
|
+
# e(level) is the CI level the command actually displayed; absent means the
|
|
534
|
+
# Stata default of 95. Both the critical value and the reported ci_level
|
|
535
|
+
# follow it, so `regress, level(90)` does not come back labelled 95.
|
|
536
|
+
level_raw = e.scalars.get("level")
|
|
537
|
+
ci_level = float(level_raw) if level_raw is not None else 95.0
|
|
538
|
+
|
|
409
539
|
table = r.matrices.get("table")
|
|
410
540
|
parsed = _from_r_table(table, resolve_matrix) if table is not None else None
|
|
411
541
|
if parsed is not None and table is not None and _r_table_matches_b(table, b, resolve_matrix):
|
|
412
542
|
coeffs, statistic_kind = parsed
|
|
413
543
|
source = "r_table"
|
|
414
544
|
else:
|
|
415
|
-
|
|
416
|
-
|
|
545
|
+
# r(table) is gone — cleared by whatever command ran after the
|
|
546
|
+
# estimation. Rebuild the table from e(b)/e(V) on the same
|
|
547
|
+
# distribution Stata used, so the numbers still match the log.
|
|
548
|
+
coeffs, statistic_kind = _from_b_v(
|
|
549
|
+
b,
|
|
550
|
+
v,
|
|
551
|
+
resolve_matrix,
|
|
552
|
+
b_values,
|
|
553
|
+
df_resid=float(df_resid) if df_resid is not None else None,
|
|
554
|
+
level=ci_level,
|
|
555
|
+
)
|
|
417
556
|
source = "e_b_v"
|
|
418
557
|
|
|
419
558
|
n_obs: int | None = None
|
|
@@ -430,9 +569,10 @@ def build_estimation_from_returns(
|
|
|
430
569
|
depvar=depvar,
|
|
431
570
|
n_obs=n_obs,
|
|
432
571
|
df_model=e.scalars.get("df_m"),
|
|
433
|
-
df_resid=
|
|
572
|
+
df_resid=df_resid,
|
|
434
573
|
statistic_kind=statistic_kind, # type: ignore[arg-type]
|
|
435
574
|
source=source, # type: ignore[arg-type]
|
|
575
|
+
ci_level=ci_level,
|
|
436
576
|
coefficients=coeffs,
|
|
437
577
|
model_stats=_model_stats(e.scalars),
|
|
438
578
|
diagnostics=_command_diagnostics(command, e.scalars),
|