stata-code 0.12.2__tar.gz → 0.13.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {stata_code-0.12.2 → stata_code-0.13.0}/CHANGELOG.md +85 -0
  2. {stata_code-0.12.2 → stata_code-0.13.0}/PKG-INFO +3 -3
  3. {stata_code-0.12.2 → stata_code-0.13.0}/README.en.md +1 -1
  4. {stata_code-0.12.2 → stata_code-0.13.0}/README.md +1 -1
  5. {stata_code-0.12.2 → stata_code-0.13.0}/SCHEMA.md +11 -1
  6. {stata_code-0.12.2 → stata_code-0.13.0}/pyproject.toml +1 -1
  7. {stata_code-0.12.2 → stata_code-0.13.0}/schema/run_result.schema.json +24 -0
  8. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/__init__.py +1 -1
  9. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/console.py +16 -0
  10. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/runner.py +30 -8
  11. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/schema.py +48 -1
  12. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/mcp/server.py +1 -1
  13. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_console.py +27 -0
  14. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_real_stata.py +34 -0
  15. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_schema.py +44 -0
  16. {stata_code-0.12.2 → stata_code-0.13.0}/.gitignore +0 -0
  17. {stata_code-0.12.2 → stata_code-0.13.0}/LICENSE +0 -0
  18. {stata_code-0.12.2 → stata_code-0.13.0}/LICENSE-POLICY.md +0 -0
  19. {stata_code-0.12.2 → stata_code-0.13.0}/PUBLISHING.md +0 -0
  20. {stata_code-0.12.2 → stata_code-0.13.0}/docs/competitive-landscape.md +0 -0
  21. {stata_code-0.12.2 → stata_code-0.13.0}/docs/design/hard_timeout.md +0 -0
  22. {stata_code-0.12.2 → stata_code-0.13.0}/docs/industry-leader-roadmap.md +0 -0
  23. {stata_code-0.12.2 → stata_code-0.13.0}/docs/quickstart.zh.md +0 -0
  24. {stata_code-0.12.2 → stata_code-0.13.0}/examples/01-basic-regression.md +0 -0
  25. {stata_code-0.12.2 → stata_code-0.13.0}/examples/02-did-card-krueger.md +0 -0
  26. {stata_code-0.12.2 → stata_code-0.13.0}/examples/03-graphs.md +0 -0
  27. {stata_code-0.12.2 → stata_code-0.13.0}/examples/04-multi-session.md +0 -0
  28. {stata_code-0.12.2 → stata_code-0.13.0}/examples/05-large-matrix.md +0 -0
  29. {stata_code-0.12.2 → stata_code-0.13.0}/examples/06-cross-stack-parity-audit.md +0 -0
  30. {stata_code-0.12.2 → stata_code-0.13.0}/examples/07-data-mcp-handoff.md +0 -0
  31. {stata_code-0.12.2 → stata_code-0.13.0}/examples/README.md +0 -0
  32. {stata_code-0.12.2 → stata_code-0.13.0}/scripts/build_skill_zip.py +0 -0
  33. {stata_code-0.12.2 → stata_code-0.13.0}/scripts/build_standalone.py +0 -0
  34. {stata_code-0.12.2 → stata_code-0.13.0}/scripts/check_github_actions.py +0 -0
  35. {stata_code-0.12.2 → stata_code-0.13.0}/scripts/check_versions.py +0 -0
  36. {stata_code-0.12.2 → stata_code-0.13.0}/scripts/export_schema.py +0 -0
  37. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/__main__.py +0 -0
  38. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/cli.py +0 -0
  39. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/__init__.py +0 -0
  40. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/_pool.py +0 -0
  41. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/_refs.py +0 -0
  42. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/_runtime.py +0 -0
  43. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/daemon.py +0 -0
  44. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/errors.py +0 -0
  45. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/estimation.py +0 -0
  46. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/handoff.py +0 -0
  47. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/jobs.py +0 -0
  48. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/lint.py +0 -0
  49. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/log_artifacts.py +0 -0
  50. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/notebook.py +0 -0
  51. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/policy.py +0 -0
  52. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/provenance.py +0 -0
  53. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/core/run_index.py +0 -0
  54. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/doctor.py +0 -0
  55. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/kernel/__init__.py +0 -0
  56. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/kernel/__main__.py +0 -0
  57. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/kernel/assets/logo-32x32.png +0 -0
  58. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/kernel/assets/logo-64x64.png +0 -0
  59. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/kernel/assets/logo-svg.svg +0 -0
  60. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/kernel/kernel.py +0 -0
  61. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/mcp/__init__.py +0 -0
  62. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/mcp/__main__.py +0 -0
  63. {stata_code-0.12.2 → stata_code-0.13.0}/stata_code/mcp_setup.py +0 -0
  64. {stata_code-0.12.2 → stata_code-0.13.0}/tests/__init__.py +0 -0
  65. {stata_code-0.12.2 → stata_code-0.13.0}/tests/conftest.py +0 -0
  66. {stata_code-0.12.2 → stata_code-0.13.0}/tests/fixtures/.gitkeep +0 -0
  67. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_agent_ergonomics.py +0 -0
  68. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_benchmark.py +0 -0
  69. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_bugfix_regressions.py +0 -0
  70. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_cancel.py +0 -0
  71. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_cli.py +0 -0
  72. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_daemon.py +0 -0
  73. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_doctor.py +0 -0
  74. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_errors.py +0 -0
  75. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_estimation.py +0 -0
  76. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_github_actions.py +0 -0
  77. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_handoff.py +0 -0
  78. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_kernel.py +0 -0
  79. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_lint.py +0 -0
  80. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_lint_mcp.py +0 -0
  81. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_log_artifacts.py +0 -0
  82. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_mcp.py +0 -0
  83. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_mcp_kernel_extra.py +0 -0
  84. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_mcp_transport.py +0 -0
  85. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_method_prompts.py +0 -0
  86. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_new_tools.py +0 -0
  87. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_notebook.py +0 -0
  88. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_notebook_phase2.py +0 -0
  89. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_policy.py +0 -0
  90. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_pool.py +0 -0
  91. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_pool_refs_extra.py +0 -0
  92. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_provenance.py +0 -0
  93. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_public_api.py +0 -0
  94. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_release_versions.py +0 -0
  95. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_run_index.py +0 -0
  96. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_runner.py +0 -0
  97. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_runner_helpers.py +0 -0
  98. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_runtime_discovery.py +0 -0
  99. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_schema_artifact.py +0 -0
  100. {stata_code-0.12.2 → stata_code-0.13.0}/tests/test_skill_package.py +0 -0
@@ -6,6 +6,91 @@ to semver-major.minor for the result schema (see `SCHEMA.md` §6).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## 0.13.0 — 2026-10-03
10
+
11
+ A data viewer for `.dta` files. Opening a dataset in VS Code used to mean
12
+ either launching Stata or converting to CSV and losing the variable labels,
13
+ value labels, formats and notes on the way. The extension now reads `.dta`
14
+ directly and shows all of it.
15
+
16
+ ### Added
17
+
18
+ - **VS Code: a Stata-free `.dta` data viewer.** Double-click any `.dta` file
19
+ and it opens in a read-only grid, registered as the default editor for
20
+ `*.dta`. The file is parsed in the extension itself, from StataCorp's
21
+ published format documentation, so it needs neither Stata, Python, nor the
22
+ MCP server. What the viewer keeps that a CSV round trip drops:
23
+ - variable labels (under each column name and in the variables panel);
24
+ - value labels, shown in place of their codes, with a *Value labels* toggle
25
+ to see the underlying numbers;
26
+ - display formats — `%td` / `%tc` / `%tm` / `%tq` / `%th` / `%tw` / `%ty`
27
+ dates including custom detail codes (`%tdCCYY-NN-DD`), `%w.df`, `%w.de`,
28
+ `%w.dg`, comma variants — printed as Stata prints them;
29
+ - the 27 missing values as `.`, `.a` … `.z`, including value labels attached
30
+ to extended missing values;
31
+ - dataset and variable notes, the dataset label, sort order and timestamp;
32
+ - `strL` variables and non-ASCII text.
33
+
34
+ Observations are fixed-width records, so the viewer reads only the rows on
35
+ screen; a 3-million-row file opens and jumps to its last row as fast as a
36
+ 5-row one. A variables panel filters by name or label and jumps to a column;
37
+ *Go to row*, arrow / Page / Home / End keys and `Cmd/Ctrl+C` work as
38
+ expected; the viewer reloads when Stata re-saves the file.
39
+
40
+ Formats covered: 113–115 (Stata 8–12), 117 (Stata 13), 118 / 119
41
+ (Stata 14+), 120 / 121 (Stata 18 alias variables), in either byte order.
42
+ The parser is tested against fixtures written by a real Stata, and its
43
+ formatted cells are asserted equal to Stata's own `list` output.
44
+ - **VS Code: `stataCode.dtaLegacyEncoding` setting** (default `auto`) for
45
+ `.dta` formats older than 118, which predate UTF-8. `auto` tries UTF-8,
46
+ then GB18030, then Windows-1252.
47
+ - **`dataset.variables[*].format` and `.value_label`.** Each variable now
48
+ reports its display format when that differs from the storage type's
49
+ default, and the name of its value label when one is attached — so a
50
+ consumer can tell that an integer is a `%td` date or that `foreign` is
51
+ labelled by `origin` without running `describe`. Both keys are *omitted*
52
+ (not `null`) when they have nothing to say, so the common variable costs no
53
+ extra tokens. Additive; `schema_version` stays `1.0`. Reported by both the
54
+ pystata and the console backend.
55
+ - **VS Code: the Data sidebar shows those two fields** next to each variable
56
+ (`long %td · Sale date`, `byte · [origin] · Car origin`).
57
+
58
+ ### Changed
59
+
60
+ - **VS Code: "View data preview" opens the data viewer instead of a text
61
+ listing.** The first `stataCode.dataPreviewObs` observations in memory are
62
+ copied through a scratch frame into a temporary `.dta` and shown in the same
63
+ grid as a file on disk, so in-memory data gets labels, formats and notes
64
+ too. The user's frame is not touched — not its data, sort order,
65
+ `c(filename)`, `c(changed)`, nor `r()` — and the temporary file is deleted
66
+ when the panel closes. One panel per session; previewing again refreshes it.
67
+ On Stata 15 or older (no frames) and with the console backend the command
68
+ falls back to the text listing below.
69
+ - **VS Code: `stataCode.dataPreviewObs` now defaults to `1000`** (was `50`)
70
+ and accepts up to `100000` (was `10000`): the grid scrolls, a text document
71
+ does not. The text fallback is capped at 200 rows regardless.
72
+ - **VS Code: the fallback text listing is shorter and no longer line-wrapped.**
73
+ `list` is run with a widened `linesize` (restored afterwards) so wide
74
+ datasets stop folding into `>` continuation lines; the command echo is
75
+ stripped from the body; the header collapses to a session line plus a
76
+ `74 obs x 12 vars - showing all 74` summary and the dataset path; and the
77
+ variable list is column-aligned instead of tab-separated.
78
+ - **VS Code: utility runs (data preview, `pwd`, `cd`) no longer hijack the
79
+ Output panel.** They used to force it visible and echo their whole log into
80
+ it. They now log one status line on success and only surface the panel on
81
+ failure.
82
+
83
+ ### Fixed
84
+
85
+ - **VS Code: "View data preview" failed on every dataset with fewer than 100
86
+ observations.** The preview listed rows with `list in 1/100`, and an `in`
87
+ range whose upper bound exceeds `_N` is an error in Stata, so the command
88
+ returned `r(198) observation numbers out of range` for `auto.dta` (74 obs)
89
+ and most other teaching datasets — while the document header still claimed
90
+ `showing: 0 of 74 observations`. The preview now selects rows with
91
+ `if _n <= N`, which lists whatever is there (including nothing at all), and
92
+ a failed preview reports its `rc` and message instead of a row count.
93
+
9
94
  ## 0.12.2 — 2026-08-11
10
95
 
11
96
  Correctness release. `results.estimation` could report confidence intervals
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: stata-code
3
- Version: 0.12.2
3
+ Version: 0.13.0
4
4
  Summary: Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)
5
5
  Project-URL: Homepage, https://github.com/brycewang-stanford/stata-code
6
6
  Project-URL: Repository, https://github.com/brycewang-stanford/stata-code
@@ -527,7 +527,7 @@ Then open Jupyter Notebook / JupyterLab (or a `.ipynb` in VS Code), pick **Stata
527
527
 
528
528
  ### As a VS Code Extension
529
529
 
530
- The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors.
530
+ The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors. It also registers a **Stata-free `.dta` data viewer**: double-click any `.dta` file to browse it in a grid with its variable labels, value labels, display formats, notes and missing-value codes intact, reading only the rows on screen so multi-gigabyte files open instantly.
531
531
 
532
532
  ```bash
533
533
  # from the VS Code CLI
@@ -488,7 +488,7 @@ Then open Jupyter Notebook / JupyterLab (or a `.ipynb` in VS Code), pick **Stata
488
488
 
489
489
  ### As a VS Code Extension
490
490
 
491
- The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors.
491
+ The companion extension is on the Marketplace as [`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode). It spawns `stata-code-mcp` as a child process and adds syntax highlighting, an Outline view for `**#` sections and `program define` blocks, code-lens "Run cell" and "Run section" actions on `.do` files, a **seven-view sidebar** (sessions / last result / **data variables** / run history / logs / graphs / **outputs**) — including an agent-native equivalent of Stata's **Variables window** and an **Outputs** panel that surfaces the `esttab` tables and `export` files each run writes to disk — status-bar indicators, completions, help lookup, conservative variable rename, and inline diagnostics from the v1.0 typed errors. It also registers a **Stata-free `.dta` data viewer**: double-click any `.dta` file to browse it in a grid with its variable labels, value labels, display formats, notes and missing-value codes intact, reading only the rows on screen so multi-gigabyte files open instantly.
492
492
 
493
493
  ```bash
494
494
  # from the VS Code CLI
@@ -456,7 +456,7 @@ jupyter kernelspec list
456
456
 
457
457
  ### 作为 VS Code 扩展
458
458
 
459
- 配套扩展已发布到 Marketplace:[`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode)。它会以子进程方式启动 `stata-code-mcp`,并提供语法高亮、`**#` section 和 `program define` 的 Outline、`.do` 文件的 code-lens "Run cell" / "Run section"、**七视图侧边栏**(sessions / last result / **data 变量浏览器** / run history / logs / graphs / **outputs**)——其中包含一个 agent-native 版的 Stata **变量窗口**,以及一个把每次运行写到磁盘的 `esttab` 表格和 `export` 文件呈现出来的 **Outputs** 面板——状态栏指示器、补全、帮助跳转、保守变量重命名,以及来自 v1.0 typed errors 的内联诊断。
459
+ 配套扩展已发布到 Marketplace:[`brycewang-stanford.stata-code-vscode`](https://marketplace.visualstudio.com/items?itemName=brycewang-stanford.stata-code-vscode)。它会以子进程方式启动 `stata-code-mcp`,并提供语法高亮、`**#` section 和 `program define` 的 Outline、`.do` 文件的 code-lens "Run cell" / "Run section"、**七视图侧边栏**(sessions / last result / **data 变量浏览器** / run history / logs / graphs / **outputs**)——其中包含一个 agent-native 版的 Stata **变量窗口**,以及一个把每次运行写到磁盘的 `esttab` 表格和 `export` 文件呈现出来的 **Outputs** 面板——状态栏指示器、补全、帮助跳转、保守变量重命名,以及来自 v1.0 typed errors 的内联诊断。扩展还注册了一个**不依赖 Stata 的 `.dta` 数据浏览器**:双击任意 `.dta` 文件即可在表格中浏览,变量标签、值标签、显示格式、notes 和缺失值代码全部保留;只读取屏幕上可见的行,几个 GB 的文件也能瞬间打开。
460
460
 
461
461
  ```bash
462
462
  # 从 VS Code 命令行
@@ -425,9 +425,19 @@ A summary of the active Stata frame *after* the command ran. Always populated.
425
425
 
426
426
  ```json
427
427
  { "name": "mpg", "type": "int", "label": "Mileage (mpg)" }
428
+ { "name": "foreign", "type": "byte", "label": "Car origin", "value_label": "origin" }
429
+ { "name": "saledate", "type": "long", "label": "Sale date", "format": "%td" }
428
430
  ```
429
431
 
430
- `type` is Stata's storage type (`byte`, `int`, `long`, `float`, `double`, `str#`, `strL`). `label` is the variable label string, or `""` if none.
432
+ | Field | Type | Notes |
433
+ | --- | --- | --- |
434
+ | `name` | `string` | Variable name. |
435
+ | `type` | `string` | Stata's storage type (`byte`, `int`, `long`, `float`, `double`, `str#`, `strL`). |
436
+ | `label` | `string` | The variable label, or `""` if none. |
437
+ | `format` | `string`, optional | The display format, **only when it differs from the storage type's default** (`%8.0g` for `byte`/`int`, `%12.0g` for `long`, `%9.0g` for `float`, `%10.0g` for `double`, `%{max(9,#)}s` for `str#`, `%9s` for `strL`). Absent means "the default". This is how a consumer learns that an integer is a date (`%td`, `%tc`, `%tm`, …). |
438
+ | `value_label` | `string`, optional | Name of the value label attached to the variable. Absent when none is attached. The mapping itself is not shipped; run `label list <name>`. |
439
+
440
+ `format` and `value_label` are **omitted, not `null`**, when they have nothing to say: the variable list rides along on every run, so a null on most variables would cost tokens to carry no information. Added in 0.13 (additive; `schema_version` stays `"1.0"`). Consumers must treat a missing key and `null` alike.
431
441
 
432
442
  When `n_vars` is large (default cap: 200), the producer truncates `variables` to the first 200 entries and emits a warning of kind `dataset_variables_truncated`. Agents wanting all variables should call `describe` directly.
433
443
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "stata-code"
7
- version = "0.12.2"
7
+ version = "0.13.0"
8
8
  description = "Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)"
9
9
  # The repo default README (README.md) is Chinese; keep the PyPI long description
10
10
  # English by pointing at README.en.md.
@@ -1145,6 +1145,18 @@
1145
1145
  "VariableInfo": {
1146
1146
  "additionalProperties": true,
1147
1147
  "properties": {
1148
+ "format": {
1149
+ "anyOf": [
1150
+ {
1151
+ "type": "string"
1152
+ },
1153
+ {
1154
+ "type": "null"
1155
+ }
1156
+ ],
1157
+ "default": null,
1158
+ "title": "Format"
1159
+ },
1148
1160
  "label": {
1149
1161
  "default": "",
1150
1162
  "title": "Label",
@@ -1157,6 +1169,18 @@
1157
1169
  "type": {
1158
1170
  "title": "Type",
1159
1171
  "type": "string"
1172
+ },
1173
+ "value_label": {
1174
+ "anyOf": [
1175
+ {
1176
+ "type": "string"
1177
+ },
1178
+ {
1179
+ "type": "null"
1180
+ }
1181
+ ],
1182
+ "default": null,
1183
+ "title": "Value Label"
1160
1184
  }
1161
1185
  },
1162
1186
  "required": [
@@ -260,7 +260,7 @@ def is_available() -> bool:
260
260
  return True
261
261
 
262
262
 
263
- __version__ = "0.12.2"
263
+ __version__ = "0.13.0"
264
264
 
265
265
  __all__ = [
266
266
  # Primary entry points
@@ -57,6 +57,7 @@ from stata_code.core.schema import (
57
57
  StataInfo,
58
58
  StataReturns,
59
59
  VariableInfo,
60
+ default_display_format,
60
61
  )
61
62
 
62
63
 
@@ -217,6 +218,7 @@ MARK_SECTION = f"{_M}|SECTION|"
217
218
  MARK_MATRIX = f"{_M}|MATRIX|"
218
219
  MARK_DS = f"{_M}|DS|"
219
220
  MARK_VAR = f"{_M}|VAR|"
221
+ MARK_VARFMT = f"{_M}|VARFMT|"
220
222
  MARK_BEGIN = f"{_M}|BEGIN"
221
223
  MARK_END = f"{_M}|END"
222
224
 
@@ -274,6 +276,9 @@ def build_wrapper_do(code: str, *, working_dir: str | None = None) -> str:
274
276
  # Stata: display "<MARK>`__v'|`: type `__v''|`: variable label `__v''"
275
277
  # Built by concatenation to avoid f-string quote/backtick collisions.
276
278
  " display \"" + MARK_VAR + "`__v'|`: type `__v''|`: variable label `__v''\"",
279
+ # A second line rather than two more fields on the first: a variable
280
+ # label may itself contain "|", so it has to stay the last field.
281
+ " display \"" + MARK_VARFMT + "`__v'|`: format `__v''|`: value label `__v''\"",
277
282
  " }",
278
283
  "}",
279
284
  f'display "{MARK_END}"',
@@ -298,6 +303,7 @@ _SCALAR_RE = re.compile(r"^\s*[re]\(([A-Za-z_][A-Za-z0-9_]*)\)\s*=\s*(.+?)\s*$")
298
303
  _MACRO_RE = re.compile(r'^\s*[re]\(([A-Za-z_][A-Za-z0-9_]*)\)\s*:\s*"?(.*?)"?\s*$')
299
304
  _DS_RE = re.compile(r"^\s*" + re.escape(MARK_DS) + r"(\w+)\|(.*)$")
300
305
  _VAR_RE = re.compile(r"^\s*" + re.escape(MARK_VAR) + r"(.+?)\|(.*?)\|(.*)$")
306
+ _VARFMT_RE = re.compile(r"^\s*" + re.escape(MARK_VARFMT) + r"(.+?)\|(.*?)\|(.*)$")
301
307
  # `matrix list` dimension header: e(b)[1,2] or symmetric e(V)[2,2]
302
308
  _MATRIX_HEADER_RE = re.compile(r"^\s*(symmetric\s+)?[A-Za-z_][A-Za-z0-9_]*\([^)]*\)\[\d+,\d+\]")
303
309
 
@@ -536,6 +542,16 @@ def _parse_dataset(block: str) -> DatasetInfo:
536
542
  label=v.group(3).strip(),
537
543
  )
538
544
  )
545
+ continue
546
+ x = _VARFMT_RE.search(raw)
547
+ if x and variables and variables[-1].name == x.group(1).strip():
548
+ # Same contract as the pystata backend: report a format only when
549
+ # it is not the storage type's default, a value label only when set.
550
+ var = variables[-1]
551
+ fmt = x.group(2).strip()
552
+ if fmt and fmt != default_display_format(var.type):
553
+ var.format = fmt
554
+ var.value_label = x.group(3).strip() or None
539
555
  return DatasetInfo(
540
556
  frame="default",
541
557
  n_obs=n_obs,
@@ -75,6 +75,7 @@ from stata_code.core.schema import (
75
75
  StataReturns,
76
76
  StataWarning,
77
77
  VariableInfo,
78
+ default_display_format,
78
79
  )
79
80
 
80
81
  # ─────────────────────────────────────────────────────────────────────────────
@@ -685,6 +686,34 @@ def _cap_macro(value: str) -> str:
685
686
  return f"{value[:MACRO_INLINE_CHAR_CAP]}… ({dropped} more chars elided)"
686
687
 
687
688
 
689
+ def _variable_info(sfi: Any, index: int) -> VariableInfo:
690
+ """Describe variable ``index``: name, storage type, label, and — only when
691
+ they carry information — its display format and attached value label."""
692
+ Data = sfi.Data
693
+ storage_type = Data.getVarType(index)
694
+ fmt: str | None = None
695
+ value_label: str | None = None
696
+ # Both lookups are best-effort: an sfi build without them must not cost
697
+ # the caller the variable list.
698
+ try:
699
+ raw = Data.getVarFormat(index) or ""
700
+ if raw and raw != default_display_format(storage_type):
701
+ fmt = raw
702
+ except Exception: # noqa: BLE001
703
+ pass
704
+ try:
705
+ value_label = sfi.ValueLabel.getVarValueLabel(index) or None
706
+ except Exception: # noqa: BLE001
707
+ pass
708
+ return VariableInfo(
709
+ name=Data.getVarName(index),
710
+ type=storage_type,
711
+ label=Data.getVarLabel(index) or "",
712
+ format=fmt,
713
+ value_label=value_label,
714
+ )
715
+
716
+
688
717
  def _collect_dataset(rt: Any, include_variables: bool) -> DatasetInfo:
689
718
  sfi = rt.sfi
690
719
  Data = sfi.Data
@@ -716,14 +745,7 @@ def _collect_dataset(rt: Any, include_variables: bool) -> DatasetInfo:
716
745
  variables: list[VariableInfo] | None
717
746
  if include_variables and n_vars > 0:
718
747
  cap = min(n_vars, _DATASET_VAR_CAP)
719
- variables = [
720
- VariableInfo(
721
- name=Data.getVarName(i),
722
- type=Data.getVarType(i),
723
- label=Data.getVarLabel(i) or "",
724
- )
725
- for i in range(cap)
726
- ]
748
+ variables = [_variable_info(sfi, i) for i in range(cap)]
727
749
  else:
728
750
  variables = None
729
751
 
@@ -6,7 +6,15 @@ import re
6
6
  from enum import Enum
7
7
  from typing import Literal
8
8
 
9
- from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
9
+ from pydantic import (
10
+ BaseModel,
11
+ ConfigDict,
12
+ Field,
13
+ SerializerFunctionWrapHandler,
14
+ field_validator,
15
+ model_serializer,
16
+ model_validator,
17
+ )
10
18
 
11
19
  # ─────────────────────────────────────────────────────────────────────────────
12
20
  # Enums (closed at v1.0; new values are minor-version additive)
@@ -247,10 +255,49 @@ class ResultsInfo(_Base):
247
255
  estimation: EstimationResult | None = None
248
256
 
249
257
 
258
+ def default_display_format(storage_type: str) -> str | None:
259
+ """The display format Stata assigns a new variable of ``storage_type``.
260
+
261
+ ``None`` for a type with no fixed default (an alias variable, or a string
262
+ the caller did not recognise).
263
+ """
264
+ numeric = {
265
+ "byte": "%8.0g",
266
+ "int": "%8.0g",
267
+ "long": "%12.0g",
268
+ "float": "%9.0g",
269
+ "double": "%10.0g",
270
+ }
271
+ if storage_type in numeric:
272
+ return numeric[storage_type]
273
+ if storage_type == "strL":
274
+ return "%9s"
275
+ m = re.fullmatch(r"str(\d+)", storage_type)
276
+ if m:
277
+ return f"%{max(9, int(m.group(1)))}s"
278
+ return None
279
+
280
+
250
281
  class VariableInfo(_Base):
251
282
  name: str
252
283
  type: str # Stata storage type: byte/int/long/float/double/str#/strL
253
284
  label: str = ""
285
+ # Both are omitted from the wire when ``None`` (see ``_drop_unset``): the
286
+ # variable list rides along on every run, so a field that is null for most
287
+ # variables would cost tokens to say nothing.
288
+ format: str | None = None
289
+ """Display format, only when it differs from the storage type's default
290
+ (so ``%td`` / ``%tc`` / ``%9.2f`` appear, ``%9.0g`` on a float does not)."""
291
+ value_label: str | None = None
292
+ """Name of the attached value label, only when one is attached."""
293
+
294
+ @model_serializer(mode="wrap")
295
+ def _drop_unset(self, handler: SerializerFunctionWrapHandler) -> dict[str, object]:
296
+ data: dict[str, object] = handler(self)
297
+ for key in ("format", "value_label"):
298
+ if data.get(key) is None:
299
+ data.pop(key, None)
300
+ return data
254
301
 
255
302
 
256
303
  class DatasetInfo(_Base):
@@ -102,7 +102,7 @@ from stata_code.core.runner import (
102
102
  )
103
103
  from stata_code.core.schema import RunResult
104
104
 
105
- __version__ = "0.12.2"
105
+ __version__ = "0.13.0"
106
106
 
107
107
  SERVER_INSTRUCTIONS = (
108
108
  "Use stata-code for running and inspecting Stata code. Prefer structuredContent "
@@ -110,6 +110,16 @@ __STATACODE__|VAR|mpg|int|Mileage (mpg)
110
110
  __STATACODE__|END
111
111
  """
112
112
 
113
+ VARFMT_LOG = REGRESS_LOG.replace(
114
+ "__STATACODE__|VAR|mpg|int|Mileage (mpg)\n",
115
+ "__STATACODE__|VAR|mpg|int|Mileage (mpg)\n"
116
+ "__STATACODE__|VARFMT|mpg|%8.0g|\n"
117
+ "__STATACODE__|VAR|foreign|byte|Car origin\n"
118
+ "__STATACODE__|VARFMT|foreign|%8.0g|origin\n"
119
+ "__STATACODE__|VAR|day|long|Sale date | first\n"
120
+ "__STATACODE__|VARFMT|day|%td|\n",
121
+ )
122
+
113
123
  ERROR_LOG = r"""
114
124
  . capture noisily {
115
125
  . regress mpg wgt
@@ -233,6 +243,23 @@ class TestSuccessParse:
233
243
  assert make.type == "str18"
234
244
  assert make.label == "Make and model"
235
245
 
246
+ def test_variable_format_and_value_label(self):
247
+ r = _build(VARFMT_LOG)
248
+ by_name = {v.name: v for v in r.dataset.variables}
249
+ # A log from before the VARFMT line existed still parses.
250
+ assert by_name["make"].format is None and by_name["make"].value_label is None
251
+ # The storage type's default format is not reported.
252
+ assert by_name["mpg"].format is None
253
+ assert by_name["foreign"].value_label == "origin"
254
+ assert by_name["foreign"].format is None
255
+ assert by_name["day"].format == "%td"
256
+ # A "|" inside a variable label does not derail the parse.
257
+ assert by_name["day"].label == "Sale date | first"
258
+
259
+ def test_wrapper_asks_for_format_and_value_label(self):
260
+ wrapper = console.build_wrapper_do("describe")
261
+ assert "__STATACODE__|VARFMT|`__v'|`: format `__v''|`: value label `__v''" in wrapper
262
+
236
263
  def test_no_marker_leakage_in_log(self):
237
264
  r = _build(REGRESS_LOG)
238
265
  assert "__STATACODE__" not in r.log.head
@@ -293,3 +293,37 @@ class TestHandoffReal:
293
293
  bad = verify_dataset(r.dataset, n_obs=100, required_vars=["nope"])
294
294
  assert bad.ok is False
295
295
  assert len(bad.issues) == 2
296
+
297
+
298
+ # ─────────────────────────────────────────────────────────────────────────────
299
+ # Variable metadata: display format and value label
300
+ # ─────────────────────────────────────────────────────────────────────────────
301
+
302
+
303
+ class TestVariableMetadataReal:
304
+ def test_format_and_value_label_only_when_informative(self):
305
+ r = _run(
306
+ "sysuse auto, clear\n"
307
+ "generate long day = 22000 + _n\n"
308
+ "format day %td\n"
309
+ "format price %9.0fc",
310
+ "rs_varmeta",
311
+ )
312
+ assert r.ok, r.error
313
+ by_name = {v.name: v for v in r.dataset.variables}
314
+
315
+ # Non-default formats are reported; they are what tells an agent that
316
+ # 22001 is a date and that price prints with thousands separators.
317
+ assert by_name["day"].format == "%td"
318
+ assert by_name["price"].format == "%9.0fc"
319
+ # auto.dta's `foreign` carries the `origin` value label.
320
+ assert by_name["foreign"].value_label == "origin"
321
+ # A storage type's default format carries no information.
322
+ assert by_name["trunk"].format is None
323
+ assert by_name["trunk"].value_label is None
324
+
325
+ wire = {v["name"]: v for v in json.loads(r.model_dump_json())["dataset"]["variables"]}
326
+ assert wire["day"] == {"name": "day", "type": "long", "label": "", "format": "%td"}
327
+ assert wire["foreign"]["value_label"] == "origin"
328
+ # Unset fields are absent from the wire, not null.
329
+ assert set(wire["trunk"]) == {"name", "type", "label"}
@@ -596,3 +596,47 @@ class TestSuggestions:
596
596
 
597
597
  def test_unknown_has_no_canonical_suggestion(self):
598
598
  assert suggestions_for(ErrorKind.UNKNOWN) == []
599
+
600
+
601
+ class TestVariableInfo:
602
+ def test_unset_format_and_value_label_are_absent_from_the_wire(self):
603
+ from stata_code.core.schema import VariableInfo
604
+
605
+ v = VariableInfo(name="mpg", type="int", label="Mileage (mpg)")
606
+ assert v.format is None and v.value_label is None
607
+ assert v.model_dump() == {"name": "mpg", "type": "int", "label": "Mileage (mpg)"}
608
+ assert json.loads(v.model_dump_json()) == v.model_dump()
609
+
610
+ def test_set_fields_roundtrip(self):
611
+ from stata_code.core.schema import VariableInfo
612
+
613
+ v = VariableInfo(name="day", type="long", format="%td", value_label="daylbl")
614
+ dumped = json.loads(v.model_dump_json())
615
+ assert dumped == {
616
+ "name": "day",
617
+ "type": "long",
618
+ "label": "",
619
+ "format": "%td",
620
+ "value_label": "daylbl",
621
+ }
622
+ assert VariableInfo.model_validate(dumped) == v
623
+
624
+ def test_pre_0_13_payload_still_validates(self):
625
+ from stata_code.core.schema import VariableInfo
626
+
627
+ v = VariableInfo.model_validate({"name": "make", "type": "str18", "label": "Make"})
628
+ assert v.format is None
629
+
630
+ def test_default_display_format_matches_stata(self):
631
+ from stata_code.core.schema import default_display_format
632
+
633
+ assert default_display_format("byte") == "%8.0g"
634
+ assert default_display_format("int") == "%8.0g"
635
+ assert default_display_format("long") == "%12.0g"
636
+ assert default_display_format("float") == "%9.0g"
637
+ assert default_display_format("double") == "%10.0g"
638
+ assert default_display_format("str5") == "%9s"
639
+ assert default_display_format("str80") == "%80s"
640
+ assert default_display_format("str2045") == "%2045s"
641
+ assert default_display_format("strL") == "%9s"
642
+ assert default_display_format("alias") is None
File without changes
File without changes
File without changes