ltc-code 0.2.31__tar.gz → 0.2.33__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ltc_code-0.2.33/PKG-INFO +32 -0
- ltc_code-0.2.33/README.md +22 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/pyproject.toml +1 -1
- ltc_code-0.2.33/src/ltc_code/nsc/EXTENSIONS.md +102 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/build_nsc_outcomes_new.py +91 -20
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/directory.dta +1 -1
- ltc_code-0.2.33/src/ltc_code/paper/README.md +114 -0
- ltc_code-0.2.33/src/ltc_code/paper/build_latex.py +125 -0
- ltc_code-0.2.33/src/ltc_code/paper/config.py +27 -0
- ltc_code-0.2.33/src/ltc_code/paper/data/charter_exhibits/disclosed_results.xlsx +0 -0
- ltc_code-0.2.33/src/ltc_code/paper/data/charter_exhibits/pending_results.xlsx +0 -0
- ltc_code-0.2.33/src/ltc_code/paper/exhibit_selection.py +52 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_figures.py +82 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_requirements.txt +7 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/age25_earnings_table.tex +77 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/balance_table.tex +68 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/cmo_details_table.tex +23 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_data_comparison_table.tex +33 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_enrollment_completion_table.tex +54 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_heterogeneity_table.tex +38 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_mediation_additional_specs_table.tex +29 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_test_score_mediation_table.tex +28 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/earnings_heterogeneity_table.tex +59 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/earnings_robustness_table.tex +25 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/earnings_threshold_table.tex +50 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/intervention_calculation_table.tex +27 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/lottery_sample_selection_table.tex +28 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/other_mediators_table.tex +38 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/summary_statistics_table.tex +105 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/paper/appendix.tex +145 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/paper/preamble.tex +556 -0
- ltc_code-0.2.33/src/ltc_code/paper/latex_source/paper/tables-and-figures.tex +147 -0
- ltc_code-0.2.33/src/ltc_code/paper/run_latex.sh +11 -0
- ltc_code-0.2.33/src/ltc_code/paper/split_inputs.py +171 -0
- ltc_code-0.2.33/src/ltc_code/paper/table_bindings.json +7183 -0
- ltc_code-0.2.33/src/ltc_code/paper/verify_exclusions.py +69 -0
- ltc_code-0.2.33/src/ltc_code/paper/verify_latex.py +73 -0
- ltc_code-0.2.33/src/ltc_code/plot_colleges.py +74 -0
- ltc_code-0.2.31/PKG-INFO +0 -10
- ltc_code-0.2.31/README.md +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/__init__.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/aspire.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/check_cmo_apps.do +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/christel_house.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/green_dot.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/helpers.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june13.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june2.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june30.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june5.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june7.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/kipp_nj.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/kipp_tx.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/main.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/make_summary_stats_table.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/mappings.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/may27.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/NEW_CROSSWALK.md +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/OUTCOMES.md +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/__init__.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/build_nsc_outcomes.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/naics.csv +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/naics.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/plot_bars.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/polars_dates.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/rocketship.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/schema_mapping.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/signal_var_calcs.py +0 -0
- {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/yes_prep.py +0 -0
ltc_code-0.2.33/PKG-INFO
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: ltc-code
|
|
3
|
+
Version: 0.2.33
|
|
4
|
+
Summary: Add your description here
|
|
5
|
+
Requires-Dist: fastexcel>=0.16,<0.20
|
|
6
|
+
Requires-Dist: polars>=1.36.1,<1.42
|
|
7
|
+
Requires-Dist: polars-readstat>=0.20.2
|
|
8
|
+
Requires-Python: >=3.9
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
### College scatter plots (0.2.32)
|
|
14
|
+
|
|
15
|
+
`from ltc_code.plot_colleges import plot_colleges` provides an OI-themed
|
|
16
|
+
scatter with explicit Polars `x`/`y` aggregations, college grouping, optional
|
|
17
|
+
filters and categorical color groups. Plotting additionally requires
|
|
18
|
+
`oi-tools`, `plotnine`, and `pyarrow` in the analysis environment; they are
|
|
19
|
+
not required to run the NSC build. See [NSC extensions](src/ltc_code/nsc/EXTENSIONS.md)
|
|
20
|
+
for attendance definitions, coarsening names, and plotting examples.
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
### Paper exhibits (0.2.33)
|
|
24
|
+
|
|
25
|
+
The complete standalone LaTeX PDF workflow is bundled under
|
|
26
|
+
`src/ltc_code/paper` (installed as `ltc_code/paper`). It includes the
|
|
27
|
+
fixed disclosed and replaceable pending Excel workbooks, Python scripts,
|
|
28
|
+
LaTeX templates, and table/figure exclusion controls. See
|
|
29
|
+
[paper instructions](src/ltc_code/paper/README.md). Copy that whole directory
|
|
30
|
+
to a writable working location and run `bash run_latex.sh`. The paper
|
|
31
|
+
workflow uses its own pinned Python dependencies and requires LaTeX;
|
|
32
|
+
the existing package dependencies are unchanged.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
|
|
2
|
+
|
|
3
|
+
### College scatter plots (0.2.32)
|
|
4
|
+
|
|
5
|
+
`from ltc_code.plot_colleges import plot_colleges` provides an OI-themed
|
|
6
|
+
scatter with explicit Polars `x`/`y` aggregations, college grouping, optional
|
|
7
|
+
filters and categorical color groups. Plotting additionally requires
|
|
8
|
+
`oi-tools`, `plotnine`, and `pyarrow` in the analysis environment; they are
|
|
9
|
+
not required to run the NSC build. See [NSC extensions](src/ltc_code/nsc/EXTENSIONS.md)
|
|
10
|
+
for attendance definitions, coarsening names, and plotting examples.
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
### Paper exhibits (0.2.33)
|
|
14
|
+
|
|
15
|
+
The complete standalone LaTeX PDF workflow is bundled under
|
|
16
|
+
`src/ltc_code/paper` (installed as `ltc_code/paper`). It includes the
|
|
17
|
+
fixed disclosed and replaceable pending Excel workbooks, Python scripts,
|
|
18
|
+
LaTeX templates, and table/figure exclusion controls. See
|
|
19
|
+
[paper instructions](src/ltc_code/paper/README.md). Copy that whole directory
|
|
20
|
+
to a writable working location and run `bash run_latex.sh`. The paper
|
|
21
|
+
workflow uses its own pinned Python dependencies and requires LaTeX;
|
|
22
|
+
the existing package dependencies are unchanged.
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# NSC attendance and coarsening changes in the sandbox
|
|
2
|
+
|
|
3
|
+
The package build is `ltc_code/nsc/build_nsc_outcomes_new.py`.
|
|
4
|
+
Import the scatter helper with `from ltc_code.plot_colleges import plot_colleges`.
|
|
5
|
+
|
|
6
|
+
Version 0.2.32 changes `coarse_t12`, `coarse_t34`, and `coarse_t58` to the
|
|
7
|
+
overall four-year benchmark; their former within-group calculations are
|
|
8
|
+
retained as `coarse_t12_pool`, `coarse_t34_pool`, and `coarse_t58_pool`.
|
|
9
|
+
|
|
10
|
+
## Two attendance measures
|
|
11
|
+
|
|
12
|
+
For each age 18–26 and each grouping `any`, `4yr`, `2yr`:
|
|
13
|
+
|
|
14
|
+
| Example | Counts |
|
|
15
|
+
| --- | --- |
|
|
16
|
+
| `att_any_1098_20` | F, Q, H, plus unknown statuses under the existing policy |
|
|
17
|
+
| `att_any_1098_all_20` | The same, plus L (less than half-time) |
|
|
18
|
+
|
|
19
|
+
Both versions exclude W (withdrawn), A (leave), D (deceased), and other
|
|
20
|
+
non-enrollment statuses. The existing handling of generic part-time text is
|
|
21
|
+
retained. Q now qualifies, and written-out "less than half-time" is classified
|
|
22
|
+
before matching "half-time". Adjacent spells with different eligibility are
|
|
23
|
+
kept separate rather than assigning the highest status to the entire period.
|
|
24
|
+
|
|
25
|
+
The original names are retained. There are no duplicate `_fulltime_` columns.
|
|
26
|
+
|
|
27
|
+
Both measures use the available tax years 2011–2014 and 2016–2022. Observable
|
|
28
|
+
years without counted attendance get zero; unavailable years remain missing.
|
|
29
|
+
Four-year attendance takes priority separately within each version. These
|
|
30
|
+
measures retain the existing NSC matching, record deduplication, and weekly
|
|
31
|
+
institution selection; they cannot recover records missing from the source.
|
|
32
|
+
|
|
33
|
+
## Coarsening: preserved and additional variables
|
|
34
|
+
|
|
35
|
+
Names without `_pool` use the overall four-year benchmark only for the named
|
|
36
|
+
tiers. Names ending in `_pool` use each named group's own pooled rate.
|
|
37
|
+
All names below start with `adj_cmp_rate_4yr_coarse_`:
|
|
38
|
+
|
|
39
|
+
| Suffix | Calculation |
|
|
40
|
+
| --- | --- |
|
|
41
|
+
| `t12` | Replace only tiers 1–2 with the overall four-year average |
|
|
42
|
+
| `t34` | Replace only tiers 3–4 with the overall four-year average |
|
|
43
|
+
| `t58` | Replace only tiers 5–8 with the overall four-year average |
|
|
44
|
+
| `t12_pool` | Replace tiers 1–2 with their own pooled average |
|
|
45
|
+
| `t34_pool` | Replace tiers 3–4 with their own pooled average |
|
|
46
|
+
| `t58_pool` | Replace tiers 5–8 with their own pooled average |
|
|
47
|
+
| `t12_t34_pool` | Pool 1–2 and, separately, 3–4 |
|
|
48
|
+
| `t14_t58_pool` | Pool 1–4 and, separately, 5–8 |
|
|
49
|
+
|
|
50
|
+
Every average uses total IPEDS completers divided by the total adjusted
|
|
51
|
+
graduation cohort in the relevant group. Non-target colleges keep the detailed
|
|
52
|
+
four-year prediction. Existing zero and missing-value rules are preserved.
|
|
53
|
+
The separate `adj_cmp_rate_coarse_4yr` still replaces ALL four-year starters'
|
|
54
|
+
rates with the common four-year average.
|
|
55
|
+
|
|
56
|
+
## Simple college scatter
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
plot = plot_colleges(
|
|
60
|
+
df,
|
|
61
|
+
x=pl.col("sid_cepr").drop_nulls().n_unique(),
|
|
62
|
+
y=pl.col("completion_rate_150pct_firstinst").drop_nulls().first() * 100,
|
|
63
|
+
group_by="unitid_firstinst",
|
|
64
|
+
filters={"enrolled": 1, "tier_firstinst": [3, 4]},
|
|
65
|
+
title="First colleges of charter enrollers: tiers 3–4",
|
|
66
|
+
)
|
|
67
|
+
plots.append(plot)
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Replace `enrolled` with your charter-enrollment column. Omit `filters` for everyone,
|
|
71
|
+
or supply just one dictionary entry. Polars expressions and lists of expressions
|
|
72
|
+
also work for more general conditions. Filters apply before the x/y aggregation.
|
|
73
|
+
The title and axis labels are customizable. x and y are Polars aggregation
|
|
74
|
+
expressions; their choice determines what each college's dot represents.
|
|
75
|
+
|
|
76
|
+
`unitid_firstinst` is the matched IPEDS ID of the first qualifying institution
|
|
77
|
+
within the by-Y4 attendance window. `completion_rate_150pct_firstinst` is that
|
|
78
|
+
institution's raw graduation rate. This example describes first colleges, not
|
|
79
|
+
every college ever attended or the institution awarding the degree. Students
|
|
80
|
+
without a matched first-college ID and colleges without a rate do not appear.
|
|
81
|
+
Use a consistent institutional rate per college: `.first()` selects that rate,
|
|
82
|
+
not an average of students' predicted completion outcomes.
|
|
83
|
+
|
|
84
|
+
The helper returns a plotnine plot; `return_data=True` returns `(plot, data)`.
|
|
85
|
+
It does not save files. Call `.save(...)` on the returned plot to export it.
|
|
86
|
+
|
|
87
|
+
### OI theme and color groups
|
|
88
|
+
|
|
89
|
+
The scatter uses `oi_tools.figures.theme_oi()` (the same theme as `plot_bars`),
|
|
90
|
+
with no grid lines and clear black axes. The analysis environment needs
|
|
91
|
+
`oi-tools`, `plotnine`, and `pyarrow`.
|
|
92
|
+
|
|
93
|
+
Add `color="enrolled"` to draw separate dots for each college's enrollers and
|
|
94
|
+
non-enrollers. The function aggregates x and y separately within each
|
|
95
|
+
college/color group, after applying filters. Numeric codes become discrete
|
|
96
|
+
legend labels. For descriptive labels, pass a column containing strings such
|
|
97
|
+
as "Enrollers" and "Non-enrollers". Missing color values are labeled "Missing".
|
|
98
|
+
The OI palette supplies the colors. Omit `color` for one dot per college.
|
|
99
|
+
|
|
100
|
+
Because the institutional graduation rate is the same for both groups, their
|
|
101
|
+
dots have the same y coordinate; x differs with the number of unique students.
|
|
102
|
+
If both counts are identical, their dots overlap.
|
|
@@ -914,7 +914,7 @@ nsc = nsc.with_columns(
|
|
|
914
914
|
###########################################################
|
|
915
915
|
|
|
916
916
|
# Sarah counts half-time or more as an attended enrollment spell.
|
|
917
|
-
enrollment_status = pl.col("enrollment").str.to_uppercase().fill_null("")
|
|
917
|
+
enrollment_status = pl.col("enrollment").str.to_uppercase().str.strip_chars().fill_null("")
|
|
918
918
|
|
|
919
919
|
enroll = (
|
|
920
920
|
nsc.filter(pl.col("graduated") == "N")
|
|
@@ -935,16 +935,19 @@ enroll = (
|
|
|
935
935
|
.then(5)
|
|
936
936
|
.when(enrollment_status.str.contains("PART", literal=False))
|
|
937
937
|
.then(4)
|
|
938
|
+
.when(enrollment_status.str.contains("LESS|\\bL\\b", literal=False))
|
|
939
|
+
.then(1)
|
|
938
940
|
.when(enrollment_status.str.contains("HALF|\\bH\\b", literal=False))
|
|
939
941
|
.then(3)
|
|
940
942
|
.when(enrollment_status.str.contains("QUARTER|\\bQ\\b", literal=False))
|
|
941
|
-
.then(
|
|
942
|
-
.when(enrollment_status.str.contains("LESS|\\bL\\b", literal=False))
|
|
943
|
-
.then(1)
|
|
943
|
+
.then(4) # Three-quarter time qualifies as at least half-time.
|
|
944
944
|
.otherwise(0)
|
|
945
945
|
.alias("_enrollment_score")
|
|
946
946
|
)
|
|
947
|
-
.with_columns(
|
|
947
|
+
.with_columns(
|
|
948
|
+
(pl.col("_enrollment_score") >= 3).cast(pl.Int8).alias("_term_att"),
|
|
949
|
+
(pl.col("_enrollment_score") >= 1).cast(pl.Int8).alias("_term_att_all"),
|
|
950
|
+
)
|
|
948
951
|
.with_columns(
|
|
949
952
|
((pl.col("term_start_date").cast(pl.Int64) // 7).cast(pl.Int64)).alias(
|
|
950
953
|
"week_start"
|
|
@@ -1035,6 +1038,7 @@ enroll_weeks = (
|
|
|
1035
1038
|
"week_start",
|
|
1036
1039
|
"week_end",
|
|
1037
1040
|
"_term_att",
|
|
1041
|
+
"_term_att_all",
|
|
1038
1042
|
"_enrollment_score",
|
|
1039
1043
|
"_enrollment_stem_sarah",
|
|
1040
1044
|
"_enrollment_stem_dhs",
|
|
@@ -1068,6 +1072,8 @@ enroll_weeks = (
|
|
|
1068
1072
|
(pl.col("week") != pl.col("week").shift(1) + 1)
|
|
1069
1073
|
| (pl.col("sid_cepr") != pl.col("sid_cepr").shift(1))
|
|
1070
1074
|
| (pl.col("ID_FSC") != pl.col("ID_FSC").shift(1))
|
|
1075
|
+
| (pl.col("_term_att") != pl.col("_term_att").shift(1))
|
|
1076
|
+
| (pl.col("_term_att_all") != pl.col("_term_att_all").shift(1))
|
|
1071
1077
|
| (
|
|
1072
1078
|
pl.col("_enrollment_stem_sarah")
|
|
1073
1079
|
!= pl.col("_enrollment_stem_sarah").shift(1)
|
|
@@ -1102,6 +1108,7 @@ enroll = (
|
|
|
1102
1108
|
pl.col("k_mean").drop_nulls().first(),
|
|
1103
1109
|
pl.col("completion_rate_150pct_ip").drop_nulls().first(),
|
|
1104
1110
|
pl.col("_term_att").max(),
|
|
1111
|
+
pl.col("_term_att_all").max(),
|
|
1105
1112
|
pl.col("_enrollment_stem_sarah").max(),
|
|
1106
1113
|
pl.col("_enrollment_stem_dhs").max(),
|
|
1107
1114
|
pl.col("week").min().alias("week_start"),
|
|
@@ -1216,10 +1223,11 @@ for age in AGE_ATTENDANCE_RANGE:
|
|
|
1216
1223
|
(
|
|
1217
1224
|
(pl.col("term_start_date") <= pl.col("_age_window_end"))
|
|
1218
1225
|
& (pl.col("term_end_date") >= pl.col("_age_window_start"))
|
|
1219
|
-
& (pl.col("_term_att") == 1)
|
|
1220
1226
|
)
|
|
1221
1227
|
.cast(pl.Int8)
|
|
1222
|
-
.alias("
|
|
1228
|
+
.alias("_overlaps_all")
|
|
1229
|
+
).with_columns(
|
|
1230
|
+
(pl.col("_overlaps_all") * pl.col("_term_att")).alias("_overlaps")
|
|
1223
1231
|
)
|
|
1224
1232
|
|
|
1225
1233
|
enrollment_pieces.append(
|
|
@@ -1236,6 +1244,19 @@ for age in AGE_ATTENDANCE_RANGE:
|
|
|
1236
1244
|
.alias(f"att_{college_type}_{age}")
|
|
1237
1245
|
for college_type in AGE_ATTENDANCE_TYPES
|
|
1238
1246
|
]
|
|
1247
|
+
+ [
|
|
1248
|
+
(
|
|
1249
|
+
pl.col("_overlaps_all")
|
|
1250
|
+
* pl.col("_term_att_all")
|
|
1251
|
+
* (
|
|
1252
|
+
pl.lit(1) if college_type == "any"
|
|
1253
|
+
else pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
|
|
1254
|
+
)
|
|
1255
|
+
)
|
|
1256
|
+
.max()
|
|
1257
|
+
.alias(f"att_{college_type}_all_{age}")
|
|
1258
|
+
for college_type in ["any", "4yr", "2yr"]
|
|
1259
|
+
]
|
|
1239
1260
|
+ [
|
|
1240
1261
|
expression
|
|
1241
1262
|
for definition in STEM_DEFINITIONS
|
|
@@ -1942,7 +1963,7 @@ ipeds_tier_completion = ipeds_completion_with_sector.join(
|
|
|
1942
1963
|
).filter(pl.col("college_sector") == "4yr")
|
|
1943
1964
|
|
|
1944
1965
|
# Coarsen one bucket at a time, always starting from the detailed prediction.
|
|
1945
|
-
tier_buckets = {"
|
|
1966
|
+
tier_buckets = {"t12_pool": [1, 2], "t34_pool": [3, 4], "t58_pool": [5, 6, 7, 8]}
|
|
1946
1967
|
for label, tiers in tier_buckets.items():
|
|
1947
1968
|
bucket_rate = (
|
|
1948
1969
|
ipeds_tier_completion.filter(pl.col("tier").is_in(tiers))
|
|
@@ -1964,18 +1985,66 @@ for label, tiers in tier_buckets.items():
|
|
|
1964
1985
|
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
1965
1986
|
)
|
|
1966
1987
|
|
|
1967
|
-
#
|
|
1968
|
-
|
|
1969
|
-
|
|
1970
|
-
[
|
|
1971
|
-
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
|
|
1975
|
-
|
|
1976
|
-
|
|
1988
|
+
# Additional coarsening scenarios. Each listed group gets its own pooled
|
|
1989
|
+
# IPEDS rate; institutions outside those groups keep their detailed prediction.
|
|
1990
|
+
tier_coarsening_scenarios = {
|
|
1991
|
+
"t12_t34_pool": [[1, 2], [3, 4]],
|
|
1992
|
+
"t14_t58_pool": [[1, 2, 3, 4], [5, 6, 7, 8]],
|
|
1993
|
+
}
|
|
1994
|
+
for label, groups in tier_coarsening_scenarios.items():
|
|
1995
|
+
prediction = pl.col("adj_cmp_rate_4yr")
|
|
1996
|
+
for tiers in groups:
|
|
1997
|
+
group_rate = (
|
|
1998
|
+
ipeds_tier_completion.filter(pl.col("tier").is_in(tiers))
|
|
1999
|
+
.select(pl.col("completers_150pct_ip").sum() / pl.col("cohort_adj_150pct_ip").sum())
|
|
2000
|
+
.item()
|
|
2001
|
+
)
|
|
2002
|
+
if group_rate is None or not 0 <= group_rate <= 1:
|
|
2003
|
+
raise ValueError(f"Invalid pooled completion rate for tiers {tiers}: {group_rate}")
|
|
2004
|
+
prediction = pl.when(
|
|
2005
|
+
(pl.col("college_years_firstinst") == 4)
|
|
2006
|
+
& pl.col("tier_firstinst").is_in(tiers)
|
|
2007
|
+
& (pl.col("att_any_byY4") == 1)
|
|
2008
|
+
& pl.col("adj_cmp_rate_4yr").is_not_null()
|
|
2009
|
+
).then(group_rate).otherwise(prediction)
|
|
2010
|
+
nsc_outcomes = nsc_outcomes.with_columns(
|
|
2011
|
+
prediction.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
2012
|
+
)
|
|
1977
2013
|
|
|
1978
|
-
|
|
2014
|
+
# Replace only the named tiers with the common overall four-year average.
|
|
2015
|
+
# The _pool versions above instead use each group's own IPEDS average.
|
|
2016
|
+
benchmark_tier_buckets = {"t12": [1, 2], "t34": [3, 4], "t58": [5, 6, 7, 8]}
|
|
2017
|
+
for label, tiers in benchmark_tier_buckets.items():
|
|
2018
|
+
nsc_outcomes = nsc_outcomes.with_columns(
|
|
2019
|
+
pl.when(
|
|
2020
|
+
(pl.col("college_years_firstinst") == 4)
|
|
2021
|
+
& pl.col("tier_firstinst").is_in(tiers)
|
|
2022
|
+
& (pl.col("att_any_byY4") == 1)
|
|
2023
|
+
& pl.col("adj_cmp_rate_4yr").is_not_null()
|
|
2024
|
+
).then(cmp_rate_4yr_coarse).otherwise(pl.col("adj_cmp_rate_4yr"))
|
|
2025
|
+
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
2026
|
+
)
|
|
2027
|
+
|
|
2028
|
+
# Match available 1098-T calendar years, including the missing 2015 year.
|
|
2029
|
+
# Existing names retain AT LEAST HALF-TIME (including Q), with the existing
|
|
2030
|
+
# unknown-status policy. "all" adds less-than-half-time enrollment, excluding W/A/D and
|
|
2031
|
+
# other non-enrollment statuses. Unknown statuses count in both versions.
|
|
2032
|
+
tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
|
|
2033
|
+
tax_attendance_columns = []
|
|
2034
|
+
for college_type in ["any", "4yr", "2yr"]:
|
|
2035
|
+
for age in AGE_ATTENDANCE_RANGE:
|
|
2036
|
+
for label, source in [
|
|
2037
|
+
("", f"att_{college_type}_{age}"),
|
|
2038
|
+
("_all", f"att_{college_type}_all_{age}"),
|
|
2039
|
+
]:
|
|
2040
|
+
column = f"att_{college_type}_1098{label}_{age}"
|
|
2041
|
+
tax_attendance_columns.append(column)
|
|
2042
|
+
nsc_outcomes = nsc_outcomes.with_columns(
|
|
2043
|
+
pl.when((pl.col("cohort_lottery") + age).is_in(tax_years))
|
|
2044
|
+
.then(pl.col(source))
|
|
2045
|
+
.otherwise(None)
|
|
2046
|
+
.alias(column)
|
|
2047
|
+
)
|
|
1979
2048
|
|
|
1980
2049
|
|
|
1981
2050
|
###########################################################
|
|
@@ -2066,6 +2135,8 @@ keep_columns = [
|
|
|
2066
2135
|
"adj_cmp_rate_coarse_2yr",
|
|
2067
2136
|
"adj_cmp_rate_coarse_4yr",
|
|
2068
2137
|
*[f"adj_cmp_rate_4yr_coarse_{label}" for label in tier_buckets],
|
|
2138
|
+
*[f"adj_cmp_rate_4yr_coarse_{label}" for label in tier_coarsening_scenarios],
|
|
2139
|
+
*[f"adj_cmp_rate_4yr_coarse_{label}" for label in benchmark_tier_buckets],
|
|
2069
2140
|
"adj_cmp_rate_coarse",
|
|
2070
2141
|
"adj_cmp_rate_coarse_sample",
|
|
2071
2142
|
"adj_cmp_rate_coarsen_4yr",
|
|
@@ -2076,7 +2147,7 @@ keep_columns = [
|
|
|
2076
2147
|
"college_years_firstinst",
|
|
2077
2148
|
"tier_firstinst",
|
|
2078
2149
|
"completion_rate_150pct_firstinst",
|
|
2079
|
-
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
|
|
2150
|
+
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns + tax_attendance_columns
|
|
2080
2151
|
keep_columns = list(dict.fromkeys(keep_columns))
|
|
2081
2152
|
|
|
2082
2153
|
nsc_outcomes_final = nsc_outcomes.select(keep_columns)
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# Charter exhibits: two Excel inputs to the original LaTeX PDF
|
|
2
|
+
|
|
3
|
+
The build reads two independent workbooks. It produces the approved 19-page PDF, with the same tables, red/black numbers, figures, captions and notes.
|
|
4
|
+
|
|
5
|
+
## Run in Census
|
|
6
|
+
|
|
7
|
+
1. Keep `data/charter_exhibits/disclosed_results.xlsx` fixed.
|
|
8
|
+
2. Replace `data/charter_exhibits/pending_results.xlsx` with your actual pending results, retaining its sheets, headers and record keys. You can instead change `PENDING_INPUT` in `build_latex.py` to your new file path.
|
|
9
|
+
3. Set `synthetic` to `false` in the pending workbook's `metadata` tab after replacing its synthetic values and chart inputs. Keep `role=pending` and `schema_version=2`. This only removes the synthetic footer; **it never switches off the fixed workbook**.
|
|
10
|
+
4. Run `bash run_latex.sh` in the extracted portable folder. Output: `output/charter_exhibits_latex.pdf`.
|
|
11
|
+
|
|
12
|
+
In the sandbox, run `bash code/explore_codex/charter_exhibits/run_latex.sh` from the sandbox root. Input files are under `outputs/charter_exhibits/inputs/` and the PDF is under `outputs/charter_exhibits/`.
|
|
13
|
+
|
|
14
|
+
Python dependencies are listed in `latex_requirements.txt`; the runner uses uv. LaTeX requires `latexmk` and `pdflatex` (TeX Live or MacTeX). With dependencies already installed, `python build_latex.py` also runs the build. No network, live Overleaf connection, or live charter repository is needed. The sandbox charter repository is not modified.
|
|
15
|
+
|
|
16
|
+
## Exclude tables or figures
|
|
17
|
+
|
|
18
|
+
Edit the two lists near the top of `build_latex.py`. They default to empty, so the complete deck still builds:
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
EXCLUDE_TABLES = []
|
|
22
|
+
EXCLUDE_FIGURES = []
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
For example:
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
EXCLUDE_TABLES = ["Table A.4", "Table A.6"]
|
|
29
|
+
EXCLUDE_FIGURES = ["Figure A.1"]
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
You may use either the printed label or its identifier below. Exclusions skip the complete exhibit, its generation, and inputs used only by that exhibit. Those unused rows or whole tabs may be absent from the workbooks. Shared inputs remain required by any included exhibit. Both workbooks and their `metadata` tabs remain required.
|
|
33
|
+
|
|
34
|
+
Remaining tables and figures retain their original numbers (e.g., Table 4 stays Table 4); page numbers run consecutively. References in the unchanged draft notes retain the original exhibit numbers, even when referring to an omitted exhibit. Clearing both lists restores everything. Invalid names and excluding every exhibit produce a clear error. The audit JSON records the included and excluded exhibits. These switches control whole tables/figures, not sample columns within an exhibit.
|
|
35
|
+
|
|
36
|
+
| Printed label | Identifier |
|
|
37
|
+
| --- | --- |
|
|
38
|
+
| Table 1 | `summary_statistics_table` |
|
|
39
|
+
| Table 2 | `balance_table` |
|
|
40
|
+
| Table 3 | `age25_earnings_table` |
|
|
41
|
+
| Table 4 | `earnings_heterogeneity_table` |
|
|
42
|
+
| Table 5 | `college_enrollment_completion_table` |
|
|
43
|
+
| Table 6 | `college_test_score_mediation_table` |
|
|
44
|
+
| Table 7 | `other_mediators_table` |
|
|
45
|
+
| Figure A.1 | `appendix-credo-distributions` |
|
|
46
|
+
| Figure A.2 | `appendix-college-by-age-placeholder` |
|
|
47
|
+
| Figure A.3 | `appendix-cmo-leave-one-out` |
|
|
48
|
+
| Figure A.4 | `appendix-earnings-by-age-firm-chars` |
|
|
49
|
+
| Table A.1 | `earnings_robustness_table` |
|
|
50
|
+
| Table A.2 | `earnings_threshold_table` |
|
|
51
|
+
| Table A.3 | `college_data_comparison_table` |
|
|
52
|
+
| Table A.4 | `college_heterogeneity_table` |
|
|
53
|
+
| Table A.5 | `college_mediation_additional_specs_table` |
|
|
54
|
+
| Table A.6 | `lottery_sample_selection_table` |
|
|
55
|
+
| Table A.7 | `cmo_details_table` |
|
|
56
|
+
| Table A.8 | `intervention_calculation_table` |
|
|
57
|
+
|
|
58
|
+
## Fixed disclosed workbook
|
|
59
|
+
|
|
60
|
+
`disclosed_results.xlsx` contains only existing manuscript values from synced Overleaf commit `6ebdd8c1a84bcfd72cabf3c85a6b2a986cc5c416`, using numeric Excel cells rather than display strings for estimates.
|
|
61
|
+
|
|
62
|
+
| Tab | Input columns / purpose |
|
|
63
|
+
| --- | --- |
|
|
64
|
+
| `summary_stats` | `sample, variable, mean, sd, count, unit` |
|
|
65
|
+
| `TOT` | `sample, depvar, spec, coef, se, ccm, N_ppl, unit, coef_stars` |
|
|
66
|
+
| `ITT`, `first_stage` | Regression keys, `coef, se, control_mean, N_ppl, unit, coef_stars` |
|
|
67
|
+
| `OLS` | Regression keys, `coef, se, comparison_mean, N_ppl, unit, coef_stars` |
|
|
68
|
+
| `balance` | `sample, depvar, coef, se, control_mean, N_ppl, unit, coef_stars` |
|
|
69
|
+
| `mediation_conversions` | `depvar, earnings_gain, unit`; exact conversion values on the manuscript's existing scale |
|
|
70
|
+
| `interventions` | `program, estimated_effect, intervention, notes, years, unit`; source effects, durations and calculation notes |
|
|
71
|
+
| `metadata` | Input role, version and source snapshot |
|
|
72
|
+
|
|
73
|
+
Unreported sample sizes are blank. The summary table's `variable=n_ppl` rows store reported total people in `count`; no source sample sizes were invented or copied from synthetic data. Statistical stars supplied by the manuscript are retained. The previously corrected first-stage/ITT SE pairing is retained.
|
|
74
|
+
|
|
75
|
+
## Pending workbook
|
|
76
|
+
|
|
77
|
+
`pending_results.xlsx` contains the red table entries and synthetic chart data. It currently reproduces the approved demonstration. Replace its data with your actual estimates. No old synthetic workbook or hidden numerical fallback is used.
|
|
78
|
+
|
|
79
|
+
- `summary_stats`, `TOT`, `OLS`, `balance`: the same keyed estimate format as the fixed file, for pending fields only.
|
|
80
|
+
- `predictions`: `sample, depvar, spec, coef, predicted_gain, unit`. Supply precomputed lognormal threshold effects in `coef` and mediation predicted earnings in `predicted_gain`. These are direct inputs, not calculations from the demonstration's old synthetic conversion factors. The sample workbook preserves the exact red predictions you approved.
|
|
81
|
+
- `sample_selection`: `step, applications, N_ppl`.
|
|
82
|
+
- `cmo_details`: `cmo, geographic_area, grades, lottery_years`. Grade and year ranges are text.
|
|
83
|
+
- `credo`: one CMO per row, with `math_effect`, `reading_effect`, `in_study` (0 or 1). Effects are in standard deviations.
|
|
84
|
+
- `TOT_by_cmo`: network regression estimates, SEs and CCMs; supplies the leave-one-network-out plot.
|
|
85
|
+
- `figure_reference`: the separately supplied pooled estimate used for the leave-one-network-out reference line. This preserves the approved synthetic chart independently of the fixed earnings table. Replace it with the reference estimate appropriate to your real chart.
|
|
86
|
+
- Age profiles remain in `TOT`, with `sample=age_profile` and `spec=observed` or `ein_fes`.
|
|
87
|
+
- `metadata`: keep `role=pending`, `schema_version=2`; update `synthetic` and `source` for real inputs.
|
|
88
|
+
|
|
89
|
+
The pending file is a template for the fields that remain to be filled. **A blank field alongside a populated field can belong to the fixed workbook.** For example, a subgroup row may contain only its pending `ccm`, while `coef` and `se` are fixed in the disclosed file. Leave those fixed fields blank in the pending file. The loader rejects conflicting nonblank fields instead of silently overwriting them. It also rejects duplicate keys and missing required rows or columns. Rows can be sorted freely.
|
|
90
|
+
|
|
91
|
+
## Units, missingness and stars
|
|
92
|
+
|
|
93
|
+
- `unit=proportion`: store probabilities and probability effects on the 0-1 scale. For example, `0.056` displays as `5.6` percentage points. Means, SEs and CCMs use the same scale.
|
|
94
|
+
- `unit=usd`: unscaled dollars. `percentile`: rank points. `years` and `year`: years. `count`: whole people or applications.
|
|
95
|
+
- Threshold prediction inputs are percentage points (`3.3` means 3.3 pp); mediation prediction inputs are dollars (`834` means $834). The fixed mediation conversions retain the manuscript's exact source scale.
|
|
96
|
+
- Keep unknown numerical results empty. Zero is a real result and displays as zero. Missing values display as dashes.
|
|
97
|
+
- `coef_stars` accepts `*`, `**`, `***`, `none`, or `auto`. The supplied demo keeps the exact approved stars. When replacing estimates, supply your corresponding stars or set `auto` to calculate normal-approximation stars from `coef/se`. Table 6 deliberately displays its mediator coefficients without stars, as before.
|
|
98
|
+
- `N_ppl` and summary counts are retained as inputs; they appear only where the original paper layout includes them. Blank fixed Ns do not borrow pending counts.
|
|
99
|
+
|
|
100
|
+
## Code and verification
|
|
101
|
+
|
|
102
|
+
- `build_latex.py`: two input paths, LaTeX document wrapper and compilation.
|
|
103
|
+
- `split_inputs.py`: keyed input validation for included exhibits, fixed/pending field separation, formatting and table rendering.
|
|
104
|
+
- `exhibit_selection.py`: exclusion names, source-block selection, original numbering and omitted-exhibit references.
|
|
105
|
+
- `table_bindings.json`: stable sample/outcome/specification-to-table-cell mappings and formatting; no fitted numerical estimates.
|
|
106
|
+
- `latex_figures.py`: workbook-driven charts, with the previously approved composition.
|
|
107
|
+
- `latex_source/`: original manuscript LaTeX templates, captions and draft notes.
|
|
108
|
+
- `verify_latex.py`: verifies row-order independence, rejects duplicate/overlapping inputs, and simulates a real pending replacement while checking all fixed cells remain unchanged.
|
|
109
|
+
|
|
110
|
+
`verify_exclusions.py` additionally tests missing excluded input sheets, shared dependencies, table-only and figure-only decks, original numbering, and restoring the full deck.
|
|
111
|
+
|
|
112
|
+
The build writes an audit JSON with both input paths and SHA-256 hashes, consumed keys, and layout hashes. The refactor was checked against the approved PDF: all 19 pages have identical text and the chart PNGs are byte-identical. Replacement testing leaves the fixed workbook byte-identical.
|
|
113
|
+
|
|
114
|
+
Only the two XLSX files and templates are needed at runtime. The earlier single-workbook `overleaf_values` mechanism and prototype synthetic calculation scripts are obsolete and excluded from the portable package. Existing draft caption/figure inconsistencies remain as approved; this refactor changes input ownership and code structure, not the paper's substantive content.
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# --- Import necessary packages ---
|
|
2
|
+
import hashlib
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import re
|
|
6
|
+
import shutil
|
|
7
|
+
import subprocess
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
sys.path.append('/Users/williamratnavale/Projects/codex_sandbox/code')
|
|
12
|
+
from config import HERE, OUTPUT_DIR
|
|
13
|
+
os.environ.setdefault('MPLCONFIGDIR',str(OUTPUT_DIR/'.matplotlib'))
|
|
14
|
+
from split_inputs import SplitInputs, build_tables
|
|
15
|
+
from latex_figures import make_figures
|
|
16
|
+
from pypdf import PdfReader
|
|
17
|
+
from exhibit_selection import select_exhibits
|
|
18
|
+
|
|
19
|
+
###########################################################
|
|
20
|
+
# Define paths: this build never reads or writes the live charter repository
|
|
21
|
+
###########################################################
|
|
22
|
+
|
|
23
|
+
LOCAL_INPUTS=HERE/'data/charter_exhibits'
|
|
24
|
+
INPUT_DIR=LOCAL_INPUTS if (LOCAL_INPUTS/'disclosed_results.xlsx').exists() else OUTPUT_DIR/'inputs'
|
|
25
|
+
# Keep DISCLOSED_INPUT fixed. Replace PENDING_INPUT with your pending actual results.
|
|
26
|
+
DISCLOSED_INPUT=INPUT_DIR/'disclosed_results.xlsx'
|
|
27
|
+
PENDING_INPUT=INPUT_DIR/'pending_results.xlsx'
|
|
28
|
+
LATEX_OUTPUT=(HERE/'output' if INPUT_DIR==LOCAL_INPUTS else OUTPUT_DIR)/'charter_exhibits_latex.pdf'
|
|
29
|
+
SOURCE=HERE/'latex_source'
|
|
30
|
+
BUILD=LATEX_OUTPUT.parent/'latex_build'
|
|
31
|
+
|
|
32
|
+
# Toggle whole exhibits here. Empty lists include everything.
|
|
33
|
+
# Examples: EXCLUDE_TABLES = ['Table A.4', 'lottery_sample_selection_table']
|
|
34
|
+
# EXCLUDE_FIGURES = ['Figure A.1', 'appendix-cmo-leave-one-out']
|
|
35
|
+
# All accepted labels/identifiers are listed in README_LATEX.md (README.md in ZIP).
|
|
36
|
+
EXCLUDE_TABLES = []
|
|
37
|
+
EXCLUDE_FIGURES = []
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def build_latex(pending_path=PENDING_INPUT, output_path=LATEX_OUTPUT, build_dir=BUILD, disclosed_path=DISCLOSED_INPUT,
|
|
41
|
+
exclude_tables=None, exclude_figures=None):
|
|
42
|
+
main,appendix,included,excluded,references=select_exhibits(
|
|
43
|
+
SOURCE, EXCLUDE_TABLES if exclude_tables is None else exclude_tables,
|
|
44
|
+
EXCLUDE_FIGURES if exclude_figures is None else exclude_figures)
|
|
45
|
+
tables=[e['name'] for e in included if e['kind']=='table']
|
|
46
|
+
figures=[e['name'] for e in included if e['kind']=='figure']
|
|
47
|
+
data=SplitInputs(disclosed_path,pending_path,selected_tables=tables,selected_figures=figures)
|
|
48
|
+
build_dir=Path(build_dir);output_path=Path(output_path)
|
|
49
|
+
build_dir.mkdir(parents=True,exist_ok=True)
|
|
50
|
+
shutil.copytree(SOURCE/'paper',build_dir/'paper',dirs_exist_ok=True)
|
|
51
|
+
(build_dir/'outputs/scalars').mkdir(parents=True,exist_ok=True)
|
|
52
|
+
# The exhibit-only document does not consume paper prose scalars.
|
|
53
|
+
(build_dir/'outputs/scalars/scalars.sty').write_text('% No prose scalars are used by these exhibits.\n')
|
|
54
|
+
for folder in ['tables','figures']:
|
|
55
|
+
shutil.rmtree(build_dir/'outputs'/folder,ignore_errors=True)
|
|
56
|
+
derived=build_tables(data,SOURCE,build_dir)
|
|
57
|
+
make_figures(data,build_dir/'outputs/figures',selected_figures=figures)
|
|
58
|
+
(build_dir/'paper/main-tables.tex').write_text(main)
|
|
59
|
+
(build_dir/'paper/appendix.tex').write_text(appendix)
|
|
60
|
+
# Preserve the original captions, notes, float/page geometry and appendix numbering.
|
|
61
|
+
# Reference-only labels resolve references to the preceding (excluded) figures.
|
|
62
|
+
wrapper=r'''\documentclass[11pt,english]{article}
|
|
63
|
+
\makeatletter
|
|
64
|
+
\def\input@path{{paper/}{outputs/scalars/}}
|
|
65
|
+
\makeatother
|
|
66
|
+
\input{preamble}
|
|
67
|
+
\makeatletter
|
|
68
|
+
\AtBeginDocument{%
|
|
69
|
+
\newlabel{fig:earnings-early-adulthood}{{1}{1}{}{figure.1}{}}%
|
|
70
|
+
\newlabel{fig:earnings-distribution}{{2}{1}{}{figure.2}{}}%
|
|
71
|
+
\newlabel{fig:childhood-program-earnings}{{4}{1}{}{figure.4}{}}%
|
|
72
|
+
\bibcite{chetty2011star}{{2011}{Chetty et~al.}{{Chetty et~al.}}{}}%
|
|
73
|
+
\bibcite{credoCMO}{{2017}{CREDO}{{CREDO}}{}}%
|
|
74
|
+
}
|
|
75
|
+
\makeatother
|
|
76
|
+
\begin{document}
|
|
77
|
+
\input{main-tables}
|
|
78
|
+
\begin{appendices}
|
|
79
|
+
\input{appendix}
|
|
80
|
+
\end{appendices}
|
|
81
|
+
\end{document}
|
|
82
|
+
'''
|
|
83
|
+
if references:
|
|
84
|
+
wrapper=wrapper.replace(r'\AtBeginDocument{%',r'\AtBeginDocument{%'+'\n'+references)
|
|
85
|
+
if not any(e['section']=='appendix' for e in included):
|
|
86
|
+
wrapper=wrapper.replace('\\begin{appendices}\n\\input{appendix}\n\\end{appendices}', '')
|
|
87
|
+
if data.metadata['synthetic'].lower()=='true':
|
|
88
|
+
wrapper=wrapper.replace(r'\begin{document}',r'''\AddToShipoutPictureFG{\AtPageLowerLeft{\put(72,18){\makebox(0,0)[l]{\sffamily\fontsize{7}{8}\selectfont Synthetic inputs -- layout reproduced from manuscript}}}}
|
|
89
|
+
\begin{document}''')
|
|
90
|
+
mixed=data.metadata.get('synthetic','').lower()=='true'
|
|
91
|
+
if mixed:
|
|
92
|
+
wrapper=wrapper.replace('Synthetic inputs -- layout reproduced from manuscript',
|
|
93
|
+
'Black table values: manuscript; red: synthetic placeholders. Plots: synthetic.')
|
|
94
|
+
(build_dir/'main.tex').write_text(wrapper)
|
|
95
|
+
engine=shutil.which('latexmk') or '/Library/TeX/texbin/latexmk'
|
|
96
|
+
command=[engine,'-pdf','-interaction=nonstopmode','-halt-on-error','-file-line-error','main.tex']
|
|
97
|
+
env=os.environ.copy();env['PATH']='/Library/TeX/texbin:'+env.get('PATH','')
|
|
98
|
+
result=subprocess.run(command,cwd=build_dir,env=env,capture_output=True,text=True)
|
|
99
|
+
(build_dir/'compile-output.txt').write_text(result.stdout+'\n'+result.stderr)
|
|
100
|
+
if result.returncode:
|
|
101
|
+
raise RuntimeError('LaTeX compilation failed:\n'+(result.stdout+result.stderr)[-4500:])
|
|
102
|
+
pdf=PdfReader(build_dir/'main.pdf')
|
|
103
|
+
text='\n'.join(p.extract_text() for p in pdf.pages)
|
|
104
|
+
if len(pdf.pages)!=len(included):
|
|
105
|
+
raise AssertionError(f'Expected {len(included)} exhibit pages, got {len(pdf.pages)}')
|
|
106
|
+
for page,exhibit in zip(pdf.pages,included):
|
|
107
|
+
if exhibit['title'].upper() not in page.extract_text():
|
|
108
|
+
raise AssertionError(f'Missing or renumbered exhibit: {exhibit["title"]}')
|
|
109
|
+
output_path.parent.mkdir(parents=True,exist_ok=True)
|
|
110
|
+
shutil.copyfile(build_dir/'main.pdf',output_path)
|
|
111
|
+
audit={'pages':len(pdf.pages),'input_sha256':{role:hashlib.sha256(path.read_bytes()).hexdigest() for role,path in data.paths.items()},
|
|
112
|
+
'source_hashes':{str(p.relative_to(SOURCE)):hashlib.sha256(p.read_bytes()).hexdigest() for p in SOURCE.rglob('*.tex')},
|
|
113
|
+
'derived':derived,'used_keys':[list(k) for k in sorted(data.used)],
|
|
114
|
+
'input_paths':{role:str(path) for role,path in data.paths.items()},
|
|
115
|
+
'binding_sha256':hashlib.sha256((HERE/'table_bindings.json').read_bytes()).hexdigest(),
|
|
116
|
+
'derived_scope':'Fixed disclosed fields and separately supplied pending results',
|
|
117
|
+
'included_exhibits':[e['title'] for e in included],
|
|
118
|
+
'excluded_exhibits':[e['title'] for e in excluded]}
|
|
119
|
+
output_path.with_suffix('.audit.json').write_text(json.dumps(audit,indent=2)+'\n')
|
|
120
|
+
print(f'Created {output_path}: {len(pdf.pages)} pages using source LaTeX layouts')
|
|
121
|
+
return audit
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
if __name__=='__main__':
|
|
125
|
+
build_latex()
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# --- Import necessary packages ---
|
|
2
|
+
import sys
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
sys.path.append("/Users/williamratnavale/Projects/codex_sandbox/code")
|
|
6
|
+
try:
|
|
7
|
+
from paths import CODEX_SYNTHETIC, OUTPUTS
|
|
8
|
+
except ModuleNotFoundError:
|
|
9
|
+
# A copied folder also works on the inside without the sandbox paths module.
|
|
10
|
+
CODEX_SYNTHETIC = Path(__file__).resolve().parent / "data"
|
|
11
|
+
OUTPUTS = Path(__file__).resolve().parent / "output"
|
|
12
|
+
|
|
13
|
+
###########################################################
|
|
14
|
+
# Define necessary paths
|
|
15
|
+
###########################################################
|
|
16
|
+
|
|
17
|
+
HERE = Path(__file__).resolve().parent
|
|
18
|
+
# Replace this one path with your real workbook. The renderer never generates data.
|
|
19
|
+
BUNDLED_INPUT = HERE / "data" / "charter_exhibits" / "synthetic_estimates.xlsx"
|
|
20
|
+
INPUT_XLSX = BUNDLED_INPUT if BUNDLED_INPUT.exists() else CODEX_SYNTHETIC / "charter_exhibits" / "synthetic_estimates.xlsx"
|
|
21
|
+
OUTPUT_DIR = HERE / "output" if BUNDLED_INPUT.exists() else OUTPUTS / "charter_exhibits"
|
|
22
|
+
OUTPUT_PDF = OUTPUT_DIR / "charter_exhibits.pdf"
|
|
23
|
+
SOURCE_SNAPSHOT = HERE / "source_snapshot.json"
|
|
24
|
+
|
|
25
|
+
# Blank values require an explicit status, rather than becoming zero.
|
|
26
|
+
STATUSES = {"available", "unavailable", "not_applicable"}
|
|
27
|
+
PAGE_SIZE = (960, 540)
|