stata-code 0.13.0__tar.gz → 0.15.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. {stata_code-0.13.0 → stata_code-0.15.0}/CHANGELOG.md +79 -0
  2. {stata_code-0.13.0 → stata_code-0.15.0}/PKG-INFO +6 -5
  3. {stata_code-0.13.0 → stata_code-0.15.0}/README.en.md +5 -4
  4. {stata_code-0.13.0 → stata_code-0.15.0}/README.md +5 -4
  5. {stata_code-0.13.0 → stata_code-0.15.0}/docs/quickstart.zh.md +1 -1
  6. {stata_code-0.13.0 → stata_code-0.15.0}/pyproject.toml +1 -1
  7. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/__init__.py +1 -1
  8. stata_code-0.15.0/stata_code/core/dta_labels.py +303 -0
  9. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/mcp/server.py +112 -1
  10. stata_code-0.15.0/tests/test_dta_labels.py +256 -0
  11. {stata_code-0.13.0 → stata_code-0.15.0}/.gitignore +0 -0
  12. {stata_code-0.13.0 → stata_code-0.15.0}/LICENSE +0 -0
  13. {stata_code-0.13.0 → stata_code-0.15.0}/LICENSE-POLICY.md +0 -0
  14. {stata_code-0.13.0 → stata_code-0.15.0}/PUBLISHING.md +0 -0
  15. {stata_code-0.13.0 → stata_code-0.15.0}/SCHEMA.md +0 -0
  16. {stata_code-0.13.0 → stata_code-0.15.0}/docs/competitive-landscape.md +0 -0
  17. {stata_code-0.13.0 → stata_code-0.15.0}/docs/design/hard_timeout.md +0 -0
  18. {stata_code-0.13.0 → stata_code-0.15.0}/docs/industry-leader-roadmap.md +0 -0
  19. {stata_code-0.13.0 → stata_code-0.15.0}/examples/01-basic-regression.md +0 -0
  20. {stata_code-0.13.0 → stata_code-0.15.0}/examples/02-did-card-krueger.md +0 -0
  21. {stata_code-0.13.0 → stata_code-0.15.0}/examples/03-graphs.md +0 -0
  22. {stata_code-0.13.0 → stata_code-0.15.0}/examples/04-multi-session.md +0 -0
  23. {stata_code-0.13.0 → stata_code-0.15.0}/examples/05-large-matrix.md +0 -0
  24. {stata_code-0.13.0 → stata_code-0.15.0}/examples/06-cross-stack-parity-audit.md +0 -0
  25. {stata_code-0.13.0 → stata_code-0.15.0}/examples/07-data-mcp-handoff.md +0 -0
  26. {stata_code-0.13.0 → stata_code-0.15.0}/examples/README.md +0 -0
  27. {stata_code-0.13.0 → stata_code-0.15.0}/schema/run_result.schema.json +0 -0
  28. {stata_code-0.13.0 → stata_code-0.15.0}/scripts/build_skill_zip.py +0 -0
  29. {stata_code-0.13.0 → stata_code-0.15.0}/scripts/build_standalone.py +0 -0
  30. {stata_code-0.13.0 → stata_code-0.15.0}/scripts/check_github_actions.py +0 -0
  31. {stata_code-0.13.0 → stata_code-0.15.0}/scripts/check_versions.py +0 -0
  32. {stata_code-0.13.0 → stata_code-0.15.0}/scripts/export_schema.py +0 -0
  33. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/__main__.py +0 -0
  34. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/cli.py +0 -0
  35. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/__init__.py +0 -0
  36. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/_pool.py +0 -0
  37. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/_refs.py +0 -0
  38. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/_runtime.py +0 -0
  39. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/console.py +0 -0
  40. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/daemon.py +0 -0
  41. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/errors.py +0 -0
  42. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/estimation.py +0 -0
  43. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/handoff.py +0 -0
  44. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/jobs.py +0 -0
  45. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/lint.py +0 -0
  46. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/log_artifacts.py +0 -0
  47. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/notebook.py +0 -0
  48. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/policy.py +0 -0
  49. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/provenance.py +0 -0
  50. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/run_index.py +0 -0
  51. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/runner.py +0 -0
  52. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/core/schema.py +0 -0
  53. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/doctor.py +0 -0
  54. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/kernel/__init__.py +0 -0
  55. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/kernel/__main__.py +0 -0
  56. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/kernel/assets/logo-32x32.png +0 -0
  57. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/kernel/assets/logo-64x64.png +0 -0
  58. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/kernel/assets/logo-svg.svg +0 -0
  59. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/kernel/kernel.py +0 -0
  60. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/mcp/__init__.py +0 -0
  61. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/mcp/__main__.py +0 -0
  62. {stata_code-0.13.0 → stata_code-0.15.0}/stata_code/mcp_setup.py +0 -0
  63. {stata_code-0.13.0 → stata_code-0.15.0}/tests/__init__.py +0 -0
  64. {stata_code-0.13.0 → stata_code-0.15.0}/tests/conftest.py +0 -0
  65. {stata_code-0.13.0 → stata_code-0.15.0}/tests/fixtures/.gitkeep +0 -0
  66. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_agent_ergonomics.py +0 -0
  67. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_benchmark.py +0 -0
  68. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_bugfix_regressions.py +0 -0
  69. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_cancel.py +0 -0
  70. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_cli.py +0 -0
  71. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_console.py +0 -0
  72. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_daemon.py +0 -0
  73. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_doctor.py +0 -0
  74. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_errors.py +0 -0
  75. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_estimation.py +0 -0
  76. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_github_actions.py +0 -0
  77. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_handoff.py +0 -0
  78. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_kernel.py +0 -0
  79. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_lint.py +0 -0
  80. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_lint_mcp.py +0 -0
  81. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_log_artifacts.py +0 -0
  82. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_mcp.py +0 -0
  83. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_mcp_kernel_extra.py +0 -0
  84. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_mcp_transport.py +0 -0
  85. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_method_prompts.py +0 -0
  86. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_new_tools.py +0 -0
  87. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_notebook.py +0 -0
  88. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_notebook_phase2.py +0 -0
  89. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_policy.py +0 -0
  90. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_pool.py +0 -0
  91. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_pool_refs_extra.py +0 -0
  92. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_provenance.py +0 -0
  93. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_public_api.py +0 -0
  94. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_real_stata.py +0 -0
  95. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_release_versions.py +0 -0
  96. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_run_index.py +0 -0
  97. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_runner.py +0 -0
  98. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_runner_helpers.py +0 -0
  99. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_runtime_discovery.py +0 -0
  100. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_schema.py +0 -0
  101. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_schema_artifact.py +0 -0
  102. {stata_code-0.13.0 → stata_code-0.15.0}/tests/test_skill_package.py +0 -0
@@ -6,6 +6,85 @@ to semver-major.minor for the result schema (see `SCHEMA.md` §6).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## 0.15.0 — 2026-10-03
10
+
11
+ Variable labels can now be edited straight in a `.dta` file, by hand in the VS
12
+ Code data viewer or in bulk by an agent, without Stata. Adds one MCP tool; the
13
+ result schema is unchanged.
14
+
15
+ ### Added
16
+
17
+ - **Edit variable labels without Stata.** A variable label is a fixed-width
18
+ field in every `.dta` format since Stata 8, so it can be replaced by
19
+ overwriting that field alone: no offset moves and the data, value labels,
20
+ notes, formats and characteristics keep their bytes. Two entry points share
21
+ that approach and the same Stata-written fixtures:
22
+ - **VS Code data viewer**: *Edit* beside a variable's label in the details
23
+ panel (or double-click the variable in the list), Enter to write it to the
24
+ file, Esc to cancel. Snapshots of in-memory data stay read-only.
25
+ - **MCP tool `set_variable_labels(path, labels?, dry_run?)`** (22 tools now):
26
+ a batch `{variable: label}` map for agents, e.g. to label an unlabeled
27
+ dataset after looking at it. Omit `labels` to read the current ones. The
28
+ reply lists `changed: [{name, before, after}]`, so `before` is the undo.
29
+ Python API: `stata_code.core.dta_labels`.
30
+
31
+ Every edit is validated before any byte is written. Limits are Stata's own
32
+ 80 characters; files in formats older than 118 (Stata 13 or earlier) take
33
+ ASCII labels only, because those formats do not record their encoding.
34
+ Checked against Stata 18: the edited fixtures load, `describe` shows the new
35
+ labels, and `notes`, `label list` and `datasignature` are unchanged. Dataset
36
+ labels and value labels are variable-length and are not editable yet.
37
+
38
+ ## 0.14.0 — 2026-10-03
39
+
40
+ The `.dta` viewer added in 0.13 could only be scrolled. It can now be queried:
41
+ filter with a Stata `if` expression, sort, summarize a variable, copy a range,
42
+ and export the view or the codebook. No changes to the Python package or the
43
+ result schema beyond the version number.
44
+
45
+ ### Added
46
+
47
+ - **VS Code data viewer: filter rows with a Stata `if` expression.** The bar
48
+ under the title accepts expressions such as `age > 60 & !missing(income)`,
49
+ `region == "South":regionlbl` or `inlist(city, "Boston", "北京")`. The
50
+ evaluator follows Stata's semantics rather than JavaScript's — missing
51
+ values sort above every number, arithmetic on a missing value is missing,
52
+ any nonzero value (missing included) is true, and a string compared with a
53
+ number is a `type mismatch` — and is tested against Stata 18's own
54
+ `count if` on 48 expressions. It covers the operators
55
+ `! ~ ^ - * / + == != ~= < <= > >= & |`, `_n` / `_N`, `"text":labelname`,
56
+ and about 30 functions (`missing`, `inlist`, `inrange`, `strpos`, `regexm`,
57
+ `strmatch`, `substr`, `round`, `mod`, `mdy`, `td()`, …). Variable
58
+ abbreviations, time-series operators and macros are not supported.
59
+ - **VS Code data viewer: sort by any column.** Click the arrow in a column
60
+ header (ascending, descending, off); Shift-click adds further keys. Order
61
+ matches Stata's `sort` / `gsort`: underlying values rather than label text,
62
+ missing values last, ties in dataset order. The gutter keeps showing each
63
+ row's observation number in the file.
64
+ - **VS Code data viewer: per-variable summary.** Selecting a variable shows
65
+ count, missing, distinct, mean, standard deviation, min, quartiles and max
66
+ over the rows in view — equal to `summarize, detail` to 12 significant
67
+ digits on the test dataset — and, for categorical variables, the most
68
+ frequent values. Clicking a value filters to it. Order statistics of a date
69
+ variable print as dates.
70
+ - **VS Code data viewer: range selection and copy.** Drag, Shift-click or
71
+ Shift-arrow selects a rectangle; `Cmd/Ctrl+C` copies it as tab-separated
72
+ text, `Cmd/Ctrl+Shift+C` with variable names, `Cmd/Ctrl+A` selects the view.
73
+ - **VS Code data viewer: export.** *Export view as CSV* writes the filtered,
74
+ sorted view with numbers at full precision (not display precision), dates as
75
+ dates, and value labels as text when the toggle is on. *Export codebook*
76
+ writes one row per variable (type, format, value label, variable label,
77
+ notes) and a second file with every value-label mapping — the metadata a CSV
78
+ cannot carry.
79
+ - **VS Code: `Stata: Load Data File in Session (use, clear)`.** On the
80
+ Explorer context menu of `.dta` files and in the viewer's menu. Asks before
81
+ discarding unsaved changes in the session.
82
+ - **VS Code data viewer: resizable columns.** Drag a header's right edge;
83
+ double-click to fit. Widths survive a reload of the same dataset.
84
+
85
+ Filtering, sorting and summaries scan the whole column and are limited to
86
+ 20 million observations; browsing has no such limit.
87
+
9
88
  ## 0.13.0 — 2026-10-03
10
89
 
11
90
  A data viewer for `.dta` files. Opening a dataset in VS Code used to mean
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: stata-code
3
- Version: 0.13.0
3
+ Version: 0.15.0
4
4
  Summary: Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)
5
5
  Project-URL: Homepage, https://github.com/brycewang-stanford/stata-code
6
6
  Project-URL: Repository, https://github.com/brycewang-stanford/stata-code
@@ -362,7 +362,7 @@ claude mcp add stata-code --scope local -- stata-code-mcp
362
362
  claude mcp add stata-code --scope project -- stata-code-mcp
363
363
  ```
364
364
 
365
- Then launch `claude` and type `/mcp` to confirm `stata-code` shows up with its 21 tools (`stata_run`, `stata_run_status`, `list_background_runs`, `stata_info`, `get_log`, `search_log`, `get_graph`, `get_matrix`, `inspect_data`, `lint_do`, `install_package`, `list_sessions`, `cancel_session`, `reset_session`, `notebook_outline`, `notebook_get_cell`, `notebook_locate`, `notebook_edit_cell`, `notebook_insert_cell`, `notebook_delete_cell`, `list_runs`).
365
+ Then launch `claude` and type `/mcp` to confirm `stata-code` shows up with its 22 tools (`stata_run`, `stata_run_status`, `list_background_runs`, `stata_info`, `get_log`, `search_log`, `get_graph`, `get_matrix`, `inspect_data`, `lint_do`, `set_variable_labels`, `install_package`, `list_sessions`, `cancel_session`, `reset_session`, `notebook_outline`, `notebook_get_cell`, `notebook_locate`, `notebook_edit_cell`, `notebook_insert_cell`, `notebook_delete_cell`, `list_runs`).
366
366
 
367
367
  #### Error Recovery in Agent Workflows
368
368
 
@@ -450,7 +450,7 @@ If an OpenAI-backed client reports `API Error: 400 Invalid schema for function
450
450
  upgrade to `stata-code>=0.6.5`, then restart the MCP client. Older server
451
451
  processes keep advertising the stale schema until they are restarted.
452
452
 
453
- The MCP server registers 21 tools:
453
+ The MCP server registers 22 tools:
454
454
 
455
455
  | Tool | Purpose |
456
456
  | --- | --- |
@@ -464,6 +464,7 @@ The MCP server registers 21 tools:
464
464
  | `get_matrix` | Fetch matrix payloads behind a `matrix://` ref |
465
465
  | `inspect_data` | Run `describe` + `codebook` and return compact dataset metadata |
466
466
  | `lint_do` | Statically check do-file source (unbalanced braces, missing `end`, dangling `///`) before spending a run |
467
+ | `set_variable_labels` | Write variable labels straight into a `.dta` file without Stata (the label fields are overwritten in place; data, value labels and notes keep their bytes); omit `labels` to read the current ones |
467
468
  | `install_package` | Install an SSC or explicit `net install` package and verify it resolves |
468
469
  | `list_sessions` | Enumerate live sessions |
469
470
  | `cancel_session` | Cancel a session; the subprocess-backed path terminates in-flight runs and short-circuits pending ones |
@@ -619,7 +620,7 @@ stata_code/
619
620
  │ ├── runner.py # in-process execute(); collects everything via sfi
620
621
  │ └── _pool.py # subprocess workers for public API / MCP hard timeouts
621
622
  ├── mcp/
622
- │ ├── server.py # MCP server (21 tools)
623
+ │ ├── server.py # MCP server (22 tools)
623
624
  │ └── ...
624
625
  ├── core/console.py # console (batch) backend — Stata 13+, no pystata
625
626
  └── kernel/
@@ -718,7 +719,7 @@ Rule of thumb:
718
719
  - Log truncation with ref store
719
720
  - Warning extraction: 5 categories + generic notes
720
721
  - 34-kind error taxonomy with canonical suggestions and a machine-readable `recovery` verdict (retriable / needs-code-change / needs-user-input)
721
- - MCP server: 21 tools, including notebook navigation / search / atomic edits, the run-bundle index (`list_runs`), log grep (`search_log`), dataset inspection (`inspect_data`), static linting (`lint_do`), and package installation (`install_package`)
722
+ - MCP server: 22 tools, including notebook navigation / search / atomic edits, the run-bundle index (`list_runs`), log grep (`search_log`), dataset inspection (`inspect_data`), static linting (`lint_do`), Stata-free variable-label editing (`set_variable_labels`), and package installation (`install_package`)
722
723
  - Command-safety guard: OS-escape / file-deletion commands (`shell`, `winexec`, `erase`, `rm`, `rmdir`, `!`) are blocked before Stata runs; configurable via `STATA_CODE_COMMAND_POLICY` / `STATA_CODE_POLICY_ALLOW` / `STATA_CODE_POLICY_BLOCK`
723
724
  - Bash / plain-terminal surface: `stata-code run` (a `.do` file, `-e` snippets, or stdin) prints the same structured `RunResult` any agent that can shell out can consume; `stata-code lint` runs the linter; `stata-code setup` writes MCP client configs
724
725
  - Console (batch) backend (`core/console.py`, `--backend console`, `run_console()`): drives the Stata command-line executable, parses the log into the same typed `RunResult`, and supports **Stata 13+ with no pystata**
@@ -323,7 +323,7 @@ claude mcp add stata-code --scope local -- stata-code-mcp
323
323
  claude mcp add stata-code --scope project -- stata-code-mcp
324
324
  ```
325
325
 
326
- Then launch `claude` and type `/mcp` to confirm `stata-code` shows up with its 21 tools (`stata_run`, `stata_run_status`, `list_background_runs`, `stata_info`, `get_log`, `search_log`, `get_graph`, `get_matrix`, `inspect_data`, `lint_do`, `install_package`, `list_sessions`, `cancel_session`, `reset_session`, `notebook_outline`, `notebook_get_cell`, `notebook_locate`, `notebook_edit_cell`, `notebook_insert_cell`, `notebook_delete_cell`, `list_runs`).
326
+ Then launch `claude` and type `/mcp` to confirm `stata-code` shows up with its 22 tools (`stata_run`, `stata_run_status`, `list_background_runs`, `stata_info`, `get_log`, `search_log`, `get_graph`, `get_matrix`, `inspect_data`, `lint_do`, `set_variable_labels`, `install_package`, `list_sessions`, `cancel_session`, `reset_session`, `notebook_outline`, `notebook_get_cell`, `notebook_locate`, `notebook_edit_cell`, `notebook_insert_cell`, `notebook_delete_cell`, `list_runs`).
327
327
 
328
328
  #### Error Recovery in Agent Workflows
329
329
 
@@ -411,7 +411,7 @@ If an OpenAI-backed client reports `API Error: 400 Invalid schema for function
411
411
  upgrade to `stata-code>=0.6.5`, then restart the MCP client. Older server
412
412
  processes keep advertising the stale schema until they are restarted.
413
413
 
414
- The MCP server registers 21 tools:
414
+ The MCP server registers 22 tools:
415
415
 
416
416
  | Tool | Purpose |
417
417
  | --- | --- |
@@ -425,6 +425,7 @@ The MCP server registers 21 tools:
425
425
  | `get_matrix` | Fetch matrix payloads behind a `matrix://` ref |
426
426
  | `inspect_data` | Run `describe` + `codebook` and return compact dataset metadata |
427
427
  | `lint_do` | Statically check do-file source (unbalanced braces, missing `end`, dangling `///`) before spending a run |
428
+ | `set_variable_labels` | Write variable labels straight into a `.dta` file without Stata (the label fields are overwritten in place; data, value labels and notes keep their bytes); omit `labels` to read the current ones |
428
429
  | `install_package` | Install an SSC or explicit `net install` package and verify it resolves |
429
430
  | `list_sessions` | Enumerate live sessions |
430
431
  | `cancel_session` | Cancel a session; the subprocess-backed path terminates in-flight runs and short-circuits pending ones |
@@ -580,7 +581,7 @@ stata_code/
580
581
  │ ├── runner.py # in-process execute(); collects everything via sfi
581
582
  │ └── _pool.py # subprocess workers for public API / MCP hard timeouts
582
583
  ├── mcp/
583
- │ ├── server.py # MCP server (21 tools)
584
+ │ ├── server.py # MCP server (22 tools)
584
585
  │ └── ...
585
586
  ├── core/console.py # console (batch) backend — Stata 13+, no pystata
586
587
  └── kernel/
@@ -679,7 +680,7 @@ Rule of thumb:
679
680
  - Log truncation with ref store
680
681
  - Warning extraction: 5 categories + generic notes
681
682
  - 34-kind error taxonomy with canonical suggestions and a machine-readable `recovery` verdict (retriable / needs-code-change / needs-user-input)
682
- - MCP server: 21 tools, including notebook navigation / search / atomic edits, the run-bundle index (`list_runs`), log grep (`search_log`), dataset inspection (`inspect_data`), static linting (`lint_do`), and package installation (`install_package`)
683
+ - MCP server: 22 tools, including notebook navigation / search / atomic edits, the run-bundle index (`list_runs`), log grep (`search_log`), dataset inspection (`inspect_data`), static linting (`lint_do`), Stata-free variable-label editing (`set_variable_labels`), and package installation (`install_package`)
683
684
  - Command-safety guard: OS-escape / file-deletion commands (`shell`, `winexec`, `erase`, `rm`, `rmdir`, `!`) are blocked before Stata runs; configurable via `STATA_CODE_COMMAND_POLICY` / `STATA_CODE_POLICY_ALLOW` / `STATA_CODE_POLICY_BLOCK`
684
685
  - Bash / plain-terminal surface: `stata-code run` (a `.do` file, `-e` snippets, or stdin) prints the same structured `RunResult` any agent that can shell out can consume; `stata-code lint` runs the linter; `stata-code setup` writes MCP client configs
685
686
  - Console (batch) backend (`core/console.py`, `--backend console`, `run_console()`): drives the Stata command-line executable, parses the log into the same typed `RunResult`, and supports **Stata 13+ with no pystata**
@@ -301,7 +301,7 @@ claude mcp add stata-code --scope local -- stata-code-mcp
301
301
  claude mcp add stata-code --scope project -- stata-code-mcp
302
302
  ```
303
303
 
304
- 接着运行 `claude`,输入 `/mcp` 确认 `stata-code` 出现并带有 21 个工具(`stata_run`, `stata_run_status`, `list_background_runs`, `stata_info`, `get_log`, `search_log`, `get_graph`, `get_matrix`, `inspect_data`, `lint_do`, `install_package`, `list_sessions`, `cancel_session`, `reset_session`, `notebook_outline`, `notebook_get_cell`, `notebook_locate`, `notebook_edit_cell`, `notebook_insert_cell`, `notebook_delete_cell`, `list_runs`)。
304
+ 接着运行 `claude`,输入 `/mcp` 确认 `stata-code` 出现并带有 22 个工具(`stata_run`, `stata_run_status`, `list_background_runs`, `stata_info`, `get_log`, `search_log`, `get_graph`, `get_matrix`, `inspect_data`, `lint_do`, `set_variable_labels`, `install_package`, `list_sessions`, `cancel_session`, `reset_session`, `notebook_outline`, `notebook_get_cell`, `notebook_locate`, `notebook_edit_cell`, `notebook_insert_cell`, `notebook_delete_cell`, `list_runs`)。
305
305
 
306
306
  #### Agent 工作流里的报错恢复
307
307
 
@@ -381,7 +381,7 @@ client 配置里写绝对路径,例如 `/abs/path/to/.venv/bin/stata-code-mcp`
381
381
  `stata-code>=0.6.5`,然后重启 MCP client。旧 server 进程在重启前仍会继续
382
382
  暴露旧 schema。
383
383
 
384
- MCP server 注册了 21 个工具:
384
+ MCP server 注册了 22 个工具:
385
385
 
386
386
  | 工具 | 用途 |
387
387
  | --- | --- |
@@ -395,6 +395,7 @@ MCP server 注册了 21 个工具:
395
395
  | `get_matrix` | 通过 `matrix://` ref 获取矩阵 `{rows, cols, values}` |
396
396
  | `inspect_data` | 运行 `describe` + `codebook`,返回紧凑的数据集元数据 |
397
397
  | `lint_do` | 在执行前静态检查 do 文件源代码(花括号不匹配、缺少 `end`、悬空 `///`) |
398
+ | `set_variable_labels` | 不经过 Stata,直接改写 `.dta` 文件里的变量标签(原地覆写标签字段,数据、值标签、notes 逐字节不变);不传 `labels` 时只读取现有标签 |
398
399
  | `install_package` | 安装 SSC 或显式 `net install` 包,并验证命令可解析 |
399
400
  | `list_sessions` | 列出 live sessions |
400
401
  | `cancel_session` | 取消某个 session;subprocess-backed 路径会终止运行中的 worker,也会短路尚未开始的运行 |
@@ -546,7 +547,7 @@ stata_code/
546
547
  │ ├── runner.py # in-process execute(); collects everything via sfi
547
548
  │ └── _pool.py # subprocess workers for public API / MCP hard timeouts
548
549
  ├── mcp/
549
- │ └── server.py # MCP server (21 tools)
550
+ │ └── server.py # MCP server (22 tools)
550
551
  └── kernel/
551
552
  └── kernel.py # Jupyter kernel
552
553
  ```
@@ -621,7 +622,7 @@ printf 'sysuse auto, clear\nsummarize mpg\nexit, clear\n' | stata-mp -q
621
622
  - 日志截断 + ref store
622
623
  - 警告抽取:5 类 + 通用 notes
623
624
  - 34 类错误分类法 + 标准化建议,以及机器可读的 `recovery` 判定(可重试 / 需改代码 / 需人工介入)
624
- - MCP server:21 个工具,覆盖执行、notebook 导航 / 检索 / 原子化编辑、运行索引(`list_runs`)、日志检索(`search_log`)、数据集检查(`inspect_data`)、静态检查(`lint_do`)和包安装(`install_package`)
625
+ - MCP server:22 个工具,覆盖执行、notebook 导航 / 检索 / 原子化编辑、运行索引(`list_runs`)、日志检索(`search_log`)、数据集检查(`inspect_data`)、静态检查(`lint_do`)、无需 Stata 的变量标签编辑(`set_variable_labels`)和包安装(`install_package`)
625
626
  - 命令安全护栏:`shell`、`winexec`、`erase`、`rm`、`rmdir`、`!` 等 OS 逃逸 / 删除文件命令在执行前被拦截;可通过 `STATA_CODE_COMMAND_POLICY` / `STATA_CODE_POLICY_ALLOW` / `STATA_CODE_POLICY_BLOCK` 配置
626
627
  - Bash / 终端入口:`stata-code run`(`.do` 文件、`-e` 片段或 stdin)打印同一套结构化 `RunResult`,任何能调用 shell 的 agent 都可消费;`stata-code lint` 运行静态检查;`stata-code setup` 写入 MCP 客户端配置
627
628
  - Console(批处理)后端(`core/console.py`、`--backend console`、`run_console()`):驱动 Stata 命令行、把日志解析成同一套带类型的 `RunResult`,支持 **Stata 13+ 且无需 pystata**
@@ -73,7 +73,7 @@ stata-code setup --claude # 写入项目级 .mcp.json(会保留其它 s
73
73
  # 或: stata-code setup --cursor / --vscode / --all
74
74
  ```
75
75
 
76
- 然后运行 `claude`,`/mcp` 里应能看到 `stata-code` 及其 21 个工具
76
+ 然后运行 `claude`,`/mcp` 里应能看到 `stata-code` 及其 22 个工具
77
77
  (`stata_run`、`lint_do`、`inspect_data`、`install_package` 等)。
78
78
 
79
79
  ---
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "stata-code"
7
- version = "0.13.0"
7
+ version = "0.15.0"
8
8
  description = "Agent-native Stata bridge — one core, multiple frontends (MCP, Jupyter, VSCode)"
9
9
  # The repo default README (README.md) is Chinese; keep the PyPI long description
10
10
  # English by pointing at README.en.md.
@@ -260,7 +260,7 @@ def is_available() -> bool:
260
260
  return True
261
261
 
262
262
 
263
- __version__ = "0.13.0"
263
+ __version__ = "0.15.0"
264
264
 
265
265
  __all__ = [
266
266
  # Primary entry points
@@ -0,0 +1,303 @@
1
+ """Read and edit variable labels in a ``.dta`` file without Stata.
2
+
3
+ A variable label is a fixed-width, zero-terminated field in every ``.dta``
4
+ format since Stata 8 (81 bytes through format 117, 321 bytes from 118 on; see
5
+ ``help dta``). Replacing one overwrites that field and nothing else: no offset
6
+ moves, and the data, value labels, notes, characteristics and strLs keep their
7
+ bytes. That is why :func:`set_variable_labels` patches the file in place
8
+ rather than loading and re-saving it, which would need Stata (or would drop
9
+ whatever the re-saving library does not model).
10
+
11
+ Written against StataCorp's published format documentation, which
12
+ LICENSE-POLICY.md §2.1 lists as an allowed reference. This is the Python
13
+ counterpart of ``vscode/src/dtaWriter.ts`` and must accept and refuse the same
14
+ labels; both are tested on the same Stata-written fixtures.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import os
20
+ import struct
21
+ from dataclasses import dataclass
22
+ from pathlib import Path
23
+ from typing import BinaryIO
24
+
25
+ __all__ = [
26
+ "MAX_VARIABLE_LABEL_CHARS",
27
+ "DtaLabelError",
28
+ "DtaLabels",
29
+ "LabelChange",
30
+ "read_variable_labels",
31
+ "set_variable_labels",
32
+ ]
33
+
34
+ #: Stata's limit on a variable label, in characters.
35
+ MAX_VARIABLE_LABEL_CHARS = 80
36
+
37
+ _VARIABLE_LABELS_TAG = b"<variable_labels>"
38
+ _VARNAMES_TAG = b"<varnames>"
39
+
40
+
41
+ class DtaLabelError(ValueError):
42
+ """The file is not a supported ``.dta`` file, or a label cannot be stored."""
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class LabelChange:
47
+ name: str
48
+ before: str
49
+ after: str
50
+
51
+ def to_dict(self) -> dict[str, str]:
52
+ return {"name": self.name, "before": self.before, "after": self.after}
53
+
54
+
55
+ @dataclass(frozen=True)
56
+ class DtaLabels:
57
+ """Where the variable labels of one file live, and what they say."""
58
+
59
+ release: int
60
+ names: list[str]
61
+ labels: list[str]
62
+ #: File offset of the first label field.
63
+ offset: int
64
+ #: Width of each label field in bytes.
65
+ width: int
66
+
67
+ def as_dict(self) -> dict[str, str]:
68
+ return dict(zip(self.names, self.labels))
69
+
70
+
71
+ def _zero_terminated(raw: bytes) -> bytes:
72
+ end = raw.find(b"\0")
73
+ return raw if end < 0 else raw[:end]
74
+
75
+
76
+ def _decode(raw: bytes, release: int) -> str:
77
+ raw = _zero_terminated(raw)
78
+ if release >= 118:
79
+ return raw.decode("utf-8", errors="replace")
80
+ # Older formats do not record their encoding; mirror the viewer's "auto".
81
+ for encoding in ("utf-8", "gb18030"):
82
+ try:
83
+ return raw.decode(encoding)
84
+ except UnicodeDecodeError:
85
+ continue
86
+ return raw.decode("cp1252", errors="replace")
87
+
88
+
89
+ def _read_exact(handle: BinaryIO, offset: int, length: int) -> bytes:
90
+ handle.seek(offset)
91
+ data = handle.read(length)
92
+ if len(data) != length:
93
+ raise DtaLabelError("malformed or truncated .dta file")
94
+ return data
95
+
96
+
97
+ def _expect(data: bytes, pos: int, tag: bytes) -> int:
98
+ if data[pos : pos + len(tag)] != tag:
99
+ raise DtaLabelError(
100
+ f"malformed .dta file: expected {tag.decode('ascii')} at byte {pos}"
101
+ )
102
+ return pos + len(tag)
103
+
104
+
105
+ def _locate_tagged(handle: BinaryIO, head: bytes, size: int) -> DtaLabels:
106
+ pos = _expect(head, 0, b"<stata_dta><header><release>")
107
+ try:
108
+ release = int(head[pos : pos + 3].decode("ascii"))
109
+ except ValueError:
110
+ raise DtaLabelError("malformed .dta file: unreadable release number") from None
111
+ if release not in (117, 118, 119, 120, 121):
112
+ raise DtaLabelError(f".dta format {release} is not supported")
113
+ pos = _expect(head, pos + 3, b"</release><byteorder>")
114
+ order = head[pos : pos + 3]
115
+ if order not in (b"LSF", b"MSF"):
116
+ raise DtaLabelError("malformed .dta file: unknown byte order")
117
+ endian = "<" if order == b"LSF" else ">"
118
+ pos = _expect(head, pos + 3, b"</byteorder><K>")
119
+ wide = release in (119, 121) # more than 32,767 variables
120
+ if wide:
121
+ (n_vars,) = struct.unpack_from(endian + "I", head, pos)
122
+ pos += 4
123
+ else:
124
+ (n_vars,) = struct.unpack_from(endian + "H", head, pos)
125
+ pos += 2
126
+ pos = _expect(head, pos, b"</K><N>")
127
+ pos += 4 if release == 117 else 8
128
+ pos = _expect(head, pos, b"</N><label>")
129
+ if release == 117:
130
+ label_length = head[pos]
131
+ pos += 1
132
+ else:
133
+ (label_length,) = struct.unpack_from(endian + "H", head, pos)
134
+ pos += 2
135
+ pos = _expect(head, pos + label_length, b"</label><timestamp>")
136
+ pos = _expect(head, pos + 1 + head[pos], b"</timestamp></header><map>")
137
+ section = struct.unpack_from(endian + "14Q", head, pos)
138
+
139
+ name_width = 33 if release == 117 else 129
140
+ label_width = 81 if release == 117 else 321
141
+ names_at, labels_at = section[3], section[7]
142
+ names_bytes = n_vars * name_width
143
+ labels_bytes = n_vars * label_width
144
+ if (
145
+ names_at + len(_VARNAMES_TAG) + names_bytes > size
146
+ or labels_at + len(_VARIABLE_LABELS_TAG) + labels_bytes > size
147
+ ):
148
+ raise DtaLabelError("malformed or truncated .dta file: section map is out of range")
149
+
150
+ if _read_exact(handle, names_at, len(_VARNAMES_TAG)) != _VARNAMES_TAG:
151
+ raise DtaLabelError("malformed .dta file: <varnames> is not where the map says")
152
+ raw_names = _read_exact(handle, names_at + len(_VARNAMES_TAG), names_bytes)
153
+ if _read_exact(handle, labels_at, len(_VARIABLE_LABELS_TAG)) != _VARIABLE_LABELS_TAG:
154
+ raise DtaLabelError(
155
+ "malformed .dta file: <variable_labels> is not where the map says"
156
+ )
157
+ offset = labels_at + len(_VARIABLE_LABELS_TAG)
158
+ raw_labels = _read_exact(handle, offset, labels_bytes)
159
+ return DtaLabels(
160
+ release=release,
161
+ names=[
162
+ _decode(raw_names[i * name_width : (i + 1) * name_width], release)
163
+ for i in range(n_vars)
164
+ ],
165
+ labels=[
166
+ _decode(raw_labels[i * label_width : (i + 1) * label_width], release)
167
+ for i in range(n_vars)
168
+ ],
169
+ offset=offset,
170
+ width=label_width,
171
+ )
172
+
173
+
174
+ def _locate_legacy(handle: BinaryIO, head: bytes, size: int) -> DtaLabels:
175
+ release = head[0]
176
+ if head[1] not in (1, 2):
177
+ raise DtaLabelError("malformed .dta file: unknown byte order")
178
+ endian = "<" if head[1] == 2 else ">"
179
+ (n_vars,) = struct.unpack_from(endian + "H", head, 4)
180
+ format_width = 12 if release == 113 else 49
181
+ names_at = 109 + n_vars
182
+ offset = names_at + n_vars * 33 + 2 * (n_vars + 1) + n_vars * (format_width + 33)
183
+ if offset + n_vars * 81 > size:
184
+ raise DtaLabelError("malformed or truncated .dta file")
185
+ raw_names = _read_exact(handle, names_at, n_vars * 33)
186
+ raw_labels = _read_exact(handle, offset, n_vars * 81)
187
+ return DtaLabels(
188
+ release=release,
189
+ names=[_decode(raw_names[i * 33 : (i + 1) * 33], release) for i in range(n_vars)],
190
+ labels=[_decode(raw_labels[i * 81 : (i + 1) * 81], release) for i in range(n_vars)],
191
+ offset=offset,
192
+ width=81,
193
+ )
194
+
195
+
196
+ def _locate(handle: BinaryIO) -> DtaLabels:
197
+ size = os.fstat(handle.fileno()).st_size
198
+ handle.seek(0)
199
+ head = handle.read(4096)
200
+ if len(head) < 4:
201
+ raise DtaLabelError("not a Stata .dta file (file is too short)")
202
+ try:
203
+ if head[:1] == b"<":
204
+ return _locate_tagged(handle, head, size)
205
+ if 113 <= head[0] <= 115:
206
+ return _locate_legacy(handle, head, size)
207
+ except (struct.error, IndexError):
208
+ raise DtaLabelError("malformed or truncated .dta file") from None
209
+ if 102 <= head[0] <= 112:
210
+ raise DtaLabelError(
211
+ f".dta format {head[0]} (Stata 7 or older) is not supported; "
212
+ "open it in Stata and save it again to convert it"
213
+ )
214
+ raise DtaLabelError("not a Stata .dta file (unrecognized header)")
215
+
216
+
217
+ def read_variable_labels(path: str | os.PathLike[str]) -> DtaLabels:
218
+ """Return the variable names and labels stored in the ``.dta`` file at ``path``."""
219
+ with Path(path).open("rb") as handle:
220
+ return _locate(handle)
221
+
222
+
223
+ def _encode(label: str, release: int, width: int) -> bytes:
224
+ if len(label) > MAX_VARIABLE_LABEL_CHARS:
225
+ raise DtaLabelError(
226
+ f"label is {len(label)} characters; Stata allows at most "
227
+ f"{MAX_VARIABLE_LABEL_CHARS}"
228
+ )
229
+ for ch in label:
230
+ code = ord(ch)
231
+ if code < 0x20 or code == 0x7F:
232
+ raise DtaLabelError(
233
+ "label contains a control character (a line break or a tab?)"
234
+ )
235
+ if 0xD800 <= code <= 0xDFFF:
236
+ raise DtaLabelError(
237
+ "label contains an unpaired surrogate and is not valid Unicode"
238
+ )
239
+ if release < 118 and code > 0x7E:
240
+ # These formats predate UTF-8 and do not record which encoding they
241
+ # use, so no non-ASCII label would read the same everywhere.
242
+ raise DtaLabelError(
243
+ f"this is a format-{release} file (Stata 13 or older), which can "
244
+ "only take plain-ASCII labels here; save it again in Stata 14 or "
245
+ "newer to use other characters"
246
+ )
247
+ encoded = label.encode("utf-8")
248
+ if len(encoded) > width - 1:
249
+ raise DtaLabelError(
250
+ f"label takes {len(encoded)} bytes; this file format has room for {width - 1}"
251
+ )
252
+ return encoded.ljust(width, b"\0")
253
+
254
+
255
+ def set_variable_labels(
256
+ path: str | os.PathLike[str],
257
+ labels: dict[str, str],
258
+ *,
259
+ dry_run: bool = False,
260
+ ) -> list[LabelChange]:
261
+ """Set variable labels in the ``.dta`` file at ``path``, in place.
262
+
263
+ ``labels`` maps variable name to the new label; ``""`` removes a label.
264
+ Every edit is validated before any byte is written, so a bad entry leaves
265
+ the file untouched. Returns the labels that actually changed (an entry that
266
+ already holds the requested text is not a change). With ``dry_run`` the
267
+ same validation runs and the same list comes back, but nothing is written.
268
+
269
+ Raises :class:`DtaLabelError` for an unsupported file, an unknown variable,
270
+ or a label Stata could not hold.
271
+ """
272
+ with Path(path).open("rb" if dry_run else "r+b") as handle:
273
+ found = _locate(handle)
274
+ index = {name: i for i, name in enumerate(found.names)}
275
+ patches: list[tuple[int, bytes, LabelChange]] = []
276
+ problems: list[str] = []
277
+ for name, label in labels.items():
278
+ i = index.get(name)
279
+ if i is None:
280
+ problems.append(f"{name}: no such variable")
281
+ continue
282
+ if not isinstance(label, str):
283
+ problems.append(f"{name}: label must be a string")
284
+ continue
285
+ if label == found.labels[i]:
286
+ continue
287
+ try:
288
+ field = _encode(label, found.release, found.width)
289
+ except DtaLabelError as exc:
290
+ problems.append(f"{name}: {exc}")
291
+ continue
292
+ patches.append(
293
+ (found.offset + i * found.width, field, LabelChange(name, found.labels[i], label))
294
+ )
295
+ if problems:
296
+ raise DtaLabelError("; ".join(problems))
297
+ if not dry_run and patches:
298
+ for offset, field, _ in patches:
299
+ handle.seek(offset)
300
+ handle.write(field)
301
+ handle.flush()
302
+ os.fsync(handle.fileno())
303
+ return [change for _, _, change in patches]