ltc-code 0.2.31__tar.gz → 0.2.33__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. ltc_code-0.2.33/PKG-INFO +32 -0
  2. ltc_code-0.2.33/README.md +22 -0
  3. {ltc_code-0.2.31 → ltc_code-0.2.33}/pyproject.toml +1 -1
  4. ltc_code-0.2.33/src/ltc_code/nsc/EXTENSIONS.md +102 -0
  5. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/build_nsc_outcomes_new.py +91 -20
  6. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/directory.dta +1 -1
  7. ltc_code-0.2.33/src/ltc_code/paper/README.md +114 -0
  8. ltc_code-0.2.33/src/ltc_code/paper/build_latex.py +125 -0
  9. ltc_code-0.2.33/src/ltc_code/paper/config.py +27 -0
  10. ltc_code-0.2.33/src/ltc_code/paper/data/charter_exhibits/disclosed_results.xlsx +0 -0
  11. ltc_code-0.2.33/src/ltc_code/paper/data/charter_exhibits/pending_results.xlsx +0 -0
  12. ltc_code-0.2.33/src/ltc_code/paper/exhibit_selection.py +52 -0
  13. ltc_code-0.2.33/src/ltc_code/paper/latex_figures.py +82 -0
  14. ltc_code-0.2.33/src/ltc_code/paper/latex_requirements.txt +7 -0
  15. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/age25_earnings_table.tex +77 -0
  16. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/balance_table.tex +68 -0
  17. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/cmo_details_table.tex +23 -0
  18. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_data_comparison_table.tex +33 -0
  19. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_enrollment_completion_table.tex +54 -0
  20. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_heterogeneity_table.tex +38 -0
  21. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_mediation_additional_specs_table.tex +29 -0
  22. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/college_test_score_mediation_table.tex +28 -0
  23. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/earnings_heterogeneity_table.tex +59 -0
  24. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/earnings_robustness_table.tex +25 -0
  25. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/earnings_threshold_table.tex +50 -0
  26. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/intervention_calculation_table.tex +27 -0
  27. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/lottery_sample_selection_table.tex +28 -0
  28. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/other_mediators_table.tex +38 -0
  29. ltc_code-0.2.33/src/ltc_code/paper/latex_source/outputs/tables/output/summary_statistics_table.tex +105 -0
  30. ltc_code-0.2.33/src/ltc_code/paper/latex_source/paper/appendix.tex +145 -0
  31. ltc_code-0.2.33/src/ltc_code/paper/latex_source/paper/preamble.tex +556 -0
  32. ltc_code-0.2.33/src/ltc_code/paper/latex_source/paper/tables-and-figures.tex +147 -0
  33. ltc_code-0.2.33/src/ltc_code/paper/run_latex.sh +11 -0
  34. ltc_code-0.2.33/src/ltc_code/paper/split_inputs.py +171 -0
  35. ltc_code-0.2.33/src/ltc_code/paper/table_bindings.json +7183 -0
  36. ltc_code-0.2.33/src/ltc_code/paper/verify_exclusions.py +69 -0
  37. ltc_code-0.2.33/src/ltc_code/paper/verify_latex.py +73 -0
  38. ltc_code-0.2.33/src/ltc_code/plot_colleges.py +74 -0
  39. ltc_code-0.2.31/PKG-INFO +0 -10
  40. ltc_code-0.2.31/README.md +0 -0
  41. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
  42. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
  43. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
  44. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
  45. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
  46. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
  47. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
  48. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
  49. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
  50. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
  51. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
  52. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
  53. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
  54. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
  55. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
  56. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
  57. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
  58. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
  59. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
  60. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
  61. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
  62. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
  63. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
  64. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
  65. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
  66. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
  67. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
  68. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
  69. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
  70. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
  71. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
  72. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/__init__.py +0 -0
  73. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/aspire.py +0 -0
  74. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/check_cmo_apps.do +0 -0
  75. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/christel_house.py +0 -0
  76. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/green_dot.py +0 -0
  77. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/helpers.py +0 -0
  78. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june13.py +0 -0
  79. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june2.py +0 -0
  80. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june30.py +0 -0
  81. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june5.py +0 -0
  82. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/june7.py +0 -0
  83. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/kipp_nj.py +0 -0
  84. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/kipp_tx.py +0 -0
  85. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/main.py +0 -0
  86. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/make_summary_stats_table.py +0 -0
  87. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/mappings.py +0 -0
  88. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/may27.py +0 -0
  89. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/NEW_CROSSWALK.md +0 -0
  90. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/OUTCOMES.md +0 -0
  91. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/__init__.py +0 -0
  92. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/build_nsc_outcomes.py +0 -0
  93. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
  94. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
  95. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
  96. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/naics.csv +0 -0
  97. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/naics.py +0 -0
  98. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
  99. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
  100. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
  101. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
  102. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
  103. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
  104. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
  105. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
  106. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
  107. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
  108. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/plot_bars.py +0 -0
  109. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/polars_dates.py +0 -0
  110. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/rocketship.py +0 -0
  111. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/schema_mapping.py +0 -0
  112. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
  113. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
  114. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
  115. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/signal_var_calcs.py +0 -0
  116. {ltc_code-0.2.31 → ltc_code-0.2.33}/src/ltc_code/yes_prep.py +0 -0
@@ -0,0 +1,32 @@
1
+ Metadata-Version: 2.3
2
+ Name: ltc-code
3
+ Version: 0.2.33
4
+ Summary: Add your description here
5
+ Requires-Dist: fastexcel>=0.16,<0.20
6
+ Requires-Dist: polars>=1.36.1,<1.42
7
+ Requires-Dist: polars-readstat>=0.20.2
8
+ Requires-Python: >=3.9
9
+ Description-Content-Type: text/markdown
10
+
11
+
12
+
13
+ ### College scatter plots (0.2.32)
14
+
15
+ `from ltc_code.plot_colleges import plot_colleges` provides an OI-themed
16
+ scatter with explicit Polars `x`/`y` aggregations, college grouping, optional
17
+ filters and categorical color groups. Plotting additionally requires
18
+ `oi-tools`, `plotnine`, and `pyarrow` in the analysis environment; they are
19
+ not required to run the NSC build. See [NSC extensions](src/ltc_code/nsc/EXTENSIONS.md)
20
+ for attendance definitions, coarsening names, and plotting examples.
21
+
22
+
23
+ ### Paper exhibits (0.2.33)
24
+
25
+ The complete standalone LaTeX PDF workflow is bundled under
26
+ `src/ltc_code/paper` (installed as `ltc_code/paper`). It includes the
27
+ fixed disclosed and replaceable pending Excel workbooks, Python scripts,
28
+ LaTeX templates, and table/figure exclusion controls. See
29
+ [paper instructions](src/ltc_code/paper/README.md). Copy that whole directory
30
+ to a writable working location and run `bash run_latex.sh`. The paper
31
+ workflow uses its own pinned Python dependencies and requires LaTeX;
32
+ the existing package dependencies are unchanged.
@@ -0,0 +1,22 @@
1
+
2
+
3
+ ### College scatter plots (0.2.32)
4
+
5
+ `from ltc_code.plot_colleges import plot_colleges` provides an OI-themed
6
+ scatter with explicit Polars `x`/`y` aggregations, college grouping, optional
7
+ filters and categorical color groups. Plotting additionally requires
8
+ `oi-tools`, `plotnine`, and `pyarrow` in the analysis environment; they are
9
+ not required to run the NSC build. See [NSC extensions](src/ltc_code/nsc/EXTENSIONS.md)
10
+ for attendance definitions, coarsening names, and plotting examples.
11
+
12
+
13
+ ### Paper exhibits (0.2.33)
14
+
15
+ The complete standalone LaTeX PDF workflow is bundled under
16
+ `src/ltc_code/paper` (installed as `ltc_code/paper`). It includes the
17
+ fixed disclosed and replaceable pending Excel workbooks, Python scripts,
18
+ LaTeX templates, and table/figure exclusion controls. See
19
+ [paper instructions](src/ltc_code/paper/README.md). Copy that whole directory
20
+ to a writable working location and run `bash run_latex.sh`. The paper
21
+ workflow uses its own pinned Python dependencies and requires LaTeX;
22
+ the existing package dependencies are unchanged.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ltc-code"
3
- version = "0.2.31"
3
+ version = "0.2.33"
4
4
  description = "Add your description here"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.9"
@@ -0,0 +1,102 @@
1
+ # NSC attendance and coarsening changes in the sandbox
2
+
3
+ The package build is `ltc_code/nsc/build_nsc_outcomes_new.py`.
4
+ Import the scatter helper with `from ltc_code.plot_colleges import plot_colleges`.
5
+
6
+ Version 0.2.32 changes `coarse_t12`, `coarse_t34`, and `coarse_t58` to the
7
+ overall four-year benchmark; their former within-group calculations are
8
+ retained as `coarse_t12_pool`, `coarse_t34_pool`, and `coarse_t58_pool`.
9
+
10
+ ## Two attendance measures
11
+
12
+ For each age 18–26 and each grouping `any`, `4yr`, `2yr`:
13
+
14
+ | Example | Counts |
15
+ | --- | --- |
16
+ | `att_any_1098_20` | F, Q, H, plus unknown statuses under the existing policy |
17
+ | `att_any_1098_all_20` | The same, plus L (less than half-time) |
18
+
19
+ Both versions exclude W (withdrawn), A (leave), D (deceased), and other
20
+ non-enrollment statuses. The existing handling of generic part-time text is
21
+ retained. Q now qualifies, and written-out "less than half-time" is classified
22
+ before matching "half-time". Adjacent spells with different eligibility are
23
+ kept separate rather than assigning the highest status to the entire period.
24
+
25
+ The original names are retained. There are no duplicate `_fulltime_` columns.
26
+
27
+ Both measures use the available tax years 2011–2014 and 2016–2022. Observable
28
+ years without counted attendance get zero; unavailable years remain missing.
29
+ Four-year attendance takes priority separately within each version. These
30
+ measures retain the existing NSC matching, record deduplication, and weekly
31
+ institution selection; they cannot recover records missing from the source.
32
+
33
+ ## Coarsening: preserved and additional variables
34
+
35
+ Names without `_pool` use the overall four-year benchmark only for the named
36
+ tiers. Names ending in `_pool` use each named group's own pooled rate.
37
+ All names below start with `adj_cmp_rate_4yr_coarse_`:
38
+
39
+ | Suffix | Calculation |
40
+ | --- | --- |
41
+ | `t12` | Replace only tiers 1–2 with the overall four-year average |
42
+ | `t34` | Replace only tiers 3–4 with the overall four-year average |
43
+ | `t58` | Replace only tiers 5–8 with the overall four-year average |
44
+ | `t12_pool` | Replace tiers 1–2 with their own pooled average |
45
+ | `t34_pool` | Replace tiers 3–4 with their own pooled average |
46
+ | `t58_pool` | Replace tiers 5–8 with their own pooled average |
47
+ | `t12_t34_pool` | Pool 1–2 and, separately, 3–4 |
48
+ | `t14_t58_pool` | Pool 1–4 and, separately, 5–8 |
49
+
50
+ Every average uses total IPEDS completers divided by the total adjusted
51
+ graduation cohort in the relevant group. Non-target colleges keep the detailed
52
+ four-year prediction. Existing zero and missing-value rules are preserved.
53
+ The separate `adj_cmp_rate_coarse_4yr` still replaces ALL four-year starters'
54
+ rates with the common four-year average.
55
+
56
+ ## Simple college scatter
57
+
58
+ ```python
59
+ plot = plot_colleges(
60
+ df,
61
+ x=pl.col("sid_cepr").drop_nulls().n_unique(),
62
+ y=pl.col("completion_rate_150pct_firstinst").drop_nulls().first() * 100,
63
+ group_by="unitid_firstinst",
64
+ filters={"enrolled": 1, "tier_firstinst": [3, 4]},
65
+ title="First colleges of charter enrollers: tiers 3–4",
66
+ )
67
+ plots.append(plot)
68
+ ```
69
+
70
+ Replace `enrolled` with your charter-enrollment column. Omit `filters` for everyone,
71
+ or supply just one dictionary entry. Polars expressions and lists of expressions
72
+ also work for more general conditions. Filters apply before the x/y aggregation.
73
+ The title and axis labels are customizable. x and y are Polars aggregation
74
+ expressions; their choice determines what each college's dot represents.
75
+
76
+ `unitid_firstinst` is the matched IPEDS ID of the first qualifying institution
77
+ within the by-Y4 attendance window. `completion_rate_150pct_firstinst` is that
78
+ institution's raw graduation rate. This example describes first colleges, not
79
+ every college ever attended or the institution awarding the degree. Students
80
+ without a matched first-college ID and colleges without a rate do not appear.
81
+ Use a consistent institutional rate per college: `.first()` selects that rate,
82
+ not an average of students' predicted completion outcomes.
83
+
84
+ The helper returns a plotnine plot; `return_data=True` returns `(plot, data)`.
85
+ It does not save files. Call `.save(...)` on the returned plot to export it.
86
+
87
+ ### OI theme and color groups
88
+
89
+ The scatter uses `oi_tools.figures.theme_oi()` (the same theme as `plot_bars`),
90
+ with no grid lines and clear black axes. The analysis environment needs
91
+ `oi-tools`, `plotnine`, and `pyarrow`.
92
+
93
+ Add `color="enrolled"` to draw separate dots for each college's enrollers and
94
+ non-enrollers. The function aggregates x and y separately within each
95
+ college/color group, after applying filters. Numeric codes become discrete
96
+ legend labels. For descriptive labels, pass a column containing strings such
97
+ as "Enrollers" and "Non-enrollers". Missing color values are labeled "Missing".
98
+ The OI palette supplies the colors. Omit `color` for one dot per college.
99
+
100
+ Because the institutional graduation rate is the same for both groups, their
101
+ dots have the same y coordinate; x differs with the number of unique students.
102
+ If both counts are identical, their dots overlap.
@@ -914,7 +914,7 @@ nsc = nsc.with_columns(
914
914
  ###########################################################
915
915
 
916
916
  # Sarah counts half-time or more as an attended enrollment spell.
917
- enrollment_status = pl.col("enrollment").str.to_uppercase().fill_null("")
917
+ enrollment_status = pl.col("enrollment").str.to_uppercase().str.strip_chars().fill_null("")
918
918
 
919
919
  enroll = (
920
920
  nsc.filter(pl.col("graduated") == "N")
@@ -935,16 +935,19 @@ enroll = (
935
935
  .then(5)
936
936
  .when(enrollment_status.str.contains("PART", literal=False))
937
937
  .then(4)
938
+ .when(enrollment_status.str.contains("LESS|\\bL\\b", literal=False))
939
+ .then(1)
938
940
  .when(enrollment_status.str.contains("HALF|\\bH\\b", literal=False))
939
941
  .then(3)
940
942
  .when(enrollment_status.str.contains("QUARTER|\\bQ\\b", literal=False))
941
- .then(2)
942
- .when(enrollment_status.str.contains("LESS|\\bL\\b", literal=False))
943
- .then(1)
943
+ .then(4) # Three-quarter time qualifies as at least half-time.
944
944
  .otherwise(0)
945
945
  .alias("_enrollment_score")
946
946
  )
947
- .with_columns((pl.col("_enrollment_score") >= 3).cast(pl.Int8).alias("_term_att"))
947
+ .with_columns(
948
+ (pl.col("_enrollment_score") >= 3).cast(pl.Int8).alias("_term_att"),
949
+ (pl.col("_enrollment_score") >= 1).cast(pl.Int8).alias("_term_att_all"),
950
+ )
948
951
  .with_columns(
949
952
  ((pl.col("term_start_date").cast(pl.Int64) // 7).cast(pl.Int64)).alias(
950
953
  "week_start"
@@ -1035,6 +1038,7 @@ enroll_weeks = (
1035
1038
  "week_start",
1036
1039
  "week_end",
1037
1040
  "_term_att",
1041
+ "_term_att_all",
1038
1042
  "_enrollment_score",
1039
1043
  "_enrollment_stem_sarah",
1040
1044
  "_enrollment_stem_dhs",
@@ -1068,6 +1072,8 @@ enroll_weeks = (
1068
1072
  (pl.col("week") != pl.col("week").shift(1) + 1)
1069
1073
  | (pl.col("sid_cepr") != pl.col("sid_cepr").shift(1))
1070
1074
  | (pl.col("ID_FSC") != pl.col("ID_FSC").shift(1))
1075
+ | (pl.col("_term_att") != pl.col("_term_att").shift(1))
1076
+ | (pl.col("_term_att_all") != pl.col("_term_att_all").shift(1))
1071
1077
  | (
1072
1078
  pl.col("_enrollment_stem_sarah")
1073
1079
  != pl.col("_enrollment_stem_sarah").shift(1)
@@ -1102,6 +1108,7 @@ enroll = (
1102
1108
  pl.col("k_mean").drop_nulls().first(),
1103
1109
  pl.col("completion_rate_150pct_ip").drop_nulls().first(),
1104
1110
  pl.col("_term_att").max(),
1111
+ pl.col("_term_att_all").max(),
1105
1112
  pl.col("_enrollment_stem_sarah").max(),
1106
1113
  pl.col("_enrollment_stem_dhs").max(),
1107
1114
  pl.col("week").min().alias("week_start"),
@@ -1216,10 +1223,11 @@ for age in AGE_ATTENDANCE_RANGE:
1216
1223
  (
1217
1224
  (pl.col("term_start_date") <= pl.col("_age_window_end"))
1218
1225
  & (pl.col("term_end_date") >= pl.col("_age_window_start"))
1219
- & (pl.col("_term_att") == 1)
1220
1226
  )
1221
1227
  .cast(pl.Int8)
1222
- .alias("_overlaps")
1228
+ .alias("_overlaps_all")
1229
+ ).with_columns(
1230
+ (pl.col("_overlaps_all") * pl.col("_term_att")).alias("_overlaps")
1223
1231
  )
1224
1232
 
1225
1233
  enrollment_pieces.append(
@@ -1236,6 +1244,19 @@ for age in AGE_ATTENDANCE_RANGE:
1236
1244
  .alias(f"att_{college_type}_{age}")
1237
1245
  for college_type in AGE_ATTENDANCE_TYPES
1238
1246
  ]
1247
+ + [
1248
+ (
1249
+ pl.col("_overlaps_all")
1250
+ * pl.col("_term_att_all")
1251
+ * (
1252
+ pl.lit(1) if college_type == "any"
1253
+ else pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
1254
+ )
1255
+ )
1256
+ .max()
1257
+ .alias(f"att_{college_type}_all_{age}")
1258
+ for college_type in ["any", "4yr", "2yr"]
1259
+ ]
1239
1260
  + [
1240
1261
  expression
1241
1262
  for definition in STEM_DEFINITIONS
@@ -1942,7 +1963,7 @@ ipeds_tier_completion = ipeds_completion_with_sector.join(
1942
1963
  ).filter(pl.col("college_sector") == "4yr")
1943
1964
 
1944
1965
  # Coarsen one bucket at a time, always starting from the detailed prediction.
1945
- tier_buckets = {"t12": [1, 2], "t34": [3, 4], "t58": [5, 6, 7, 8]}
1966
+ tier_buckets = {"t12_pool": [1, 2], "t34_pool": [3, 4], "t58_pool": [5, 6, 7, 8]}
1946
1967
  for label, tiers in tier_buckets.items():
1947
1968
  bucket_rate = (
1948
1969
  ipeds_tier_completion.filter(pl.col("tier").is_in(tiers))
@@ -1964,18 +1985,66 @@ for label, tiers in tier_buckets.items():
1964
1985
  .alias(f"adj_cmp_rate_4yr_coarse_{label}")
1965
1986
  )
1966
1987
 
1967
- # Match the available 1098-T calendar years, including the missing 2015 year.
1968
- tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
1969
- nsc_outcomes = nsc_outcomes.with_columns(
1970
- [
1971
- pl.when((pl.col("cohort_lottery") + age).is_in(tax_years))
1972
- .then(pl.col(f"att_any_{age}"))
1973
- .otherwise(None)
1974
- .alias(f"att_any_1098_{age}")
1975
- for age in AGE_ATTENDANCE_RANGE
1976
- ]
1988
+ # Additional coarsening scenarios. Each listed group gets its own pooled
1989
+ # IPEDS rate; institutions outside those groups keep their detailed prediction.
1990
+ tier_coarsening_scenarios = {
1991
+ "t12_t34_pool": [[1, 2], [3, 4]],
1992
+ "t14_t58_pool": [[1, 2, 3, 4], [5, 6, 7, 8]],
1993
+ }
1994
+ for label, groups in tier_coarsening_scenarios.items():
1995
+ prediction = pl.col("adj_cmp_rate_4yr")
1996
+ for tiers in groups:
1997
+ group_rate = (
1998
+ ipeds_tier_completion.filter(pl.col("tier").is_in(tiers))
1999
+ .select(pl.col("completers_150pct_ip").sum() / pl.col("cohort_adj_150pct_ip").sum())
2000
+ .item()
2001
+ )
2002
+ if group_rate is None or not 0 <= group_rate <= 1:
2003
+ raise ValueError(f"Invalid pooled completion rate for tiers {tiers}: {group_rate}")
2004
+ prediction = pl.when(
2005
+ (pl.col("college_years_firstinst") == 4)
2006
+ & pl.col("tier_firstinst").is_in(tiers)
2007
+ & (pl.col("att_any_byY4") == 1)
2008
+ & pl.col("adj_cmp_rate_4yr").is_not_null()
2009
+ ).then(group_rate).otherwise(prediction)
2010
+ nsc_outcomes = nsc_outcomes.with_columns(
2011
+ prediction.alias(f"adj_cmp_rate_4yr_coarse_{label}")
2012
+ )
1977
2013
 
1978
- )
2014
+ # Replace only the named tiers with the common overall four-year average.
2015
+ # The _pool versions above instead use each group's own IPEDS average.
2016
+ benchmark_tier_buckets = {"t12": [1, 2], "t34": [3, 4], "t58": [5, 6, 7, 8]}
2017
+ for label, tiers in benchmark_tier_buckets.items():
2018
+ nsc_outcomes = nsc_outcomes.with_columns(
2019
+ pl.when(
2020
+ (pl.col("college_years_firstinst") == 4)
2021
+ & pl.col("tier_firstinst").is_in(tiers)
2022
+ & (pl.col("att_any_byY4") == 1)
2023
+ & pl.col("adj_cmp_rate_4yr").is_not_null()
2024
+ ).then(cmp_rate_4yr_coarse).otherwise(pl.col("adj_cmp_rate_4yr"))
2025
+ .alias(f"adj_cmp_rate_4yr_coarse_{label}")
2026
+ )
2027
+
2028
+ # Match available 1098-T calendar years, including the missing 2015 year.
2029
+ # Existing names retain AT LEAST HALF-TIME (including Q), with the existing
2030
+ # unknown-status policy. "all" adds less-than-half-time enrollment, excluding W/A/D and
2031
+ # other non-enrollment statuses. Unknown statuses count in both versions.
2032
+ tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
2033
+ tax_attendance_columns = []
2034
+ for college_type in ["any", "4yr", "2yr"]:
2035
+ for age in AGE_ATTENDANCE_RANGE:
2036
+ for label, source in [
2037
+ ("", f"att_{college_type}_{age}"),
2038
+ ("_all", f"att_{college_type}_all_{age}"),
2039
+ ]:
2040
+ column = f"att_{college_type}_1098{label}_{age}"
2041
+ tax_attendance_columns.append(column)
2042
+ nsc_outcomes = nsc_outcomes.with_columns(
2043
+ pl.when((pl.col("cohort_lottery") + age).is_in(tax_years))
2044
+ .then(pl.col(source))
2045
+ .otherwise(None)
2046
+ .alias(column)
2047
+ )
1979
2048
 
1980
2049
 
1981
2050
  ###########################################################
@@ -2066,6 +2135,8 @@ keep_columns = [
2066
2135
  "adj_cmp_rate_coarse_2yr",
2067
2136
  "adj_cmp_rate_coarse_4yr",
2068
2137
  *[f"adj_cmp_rate_4yr_coarse_{label}" for label in tier_buckets],
2138
+ *[f"adj_cmp_rate_4yr_coarse_{label}" for label in tier_coarsening_scenarios],
2139
+ *[f"adj_cmp_rate_4yr_coarse_{label}" for label in benchmark_tier_buckets],
2069
2140
  "adj_cmp_rate_coarse",
2070
2141
  "adj_cmp_rate_coarse_sample",
2071
2142
  "adj_cmp_rate_coarsen_4yr",
@@ -2076,7 +2147,7 @@ keep_columns = [
2076
2147
  "college_years_firstinst",
2077
2148
  "tier_firstinst",
2078
2149
  "completion_rate_150pct_firstinst",
2079
- ] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
2150
+ ] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns + tax_attendance_columns
2080
2151
  keep_columns = list(dict.fromkeys(keep_columns))
2081
2152
 
2082
2153
  nsc_outcomes_final = nsc_outcomes.select(keep_columns)
@@ -1,4 +1,4 @@
1
1
  [diffend] Oversized file quarantined before diffing.
2
- name: ltc_code-0.2.31/src/ltc_code/nsc/raw/directory.dta
2
+ name: ltc_code-0.2.33/src/ltc_code/nsc/raw/directory.dta
3
3
  size: 256195841 bytes
4
4
  sha256: 9fe2a62ad77a3a17f5c8f45df08cc2e8daeb46425eeb062a576353aa2bd463d1
@@ -0,0 +1,114 @@
1
+ # Charter exhibits: two Excel inputs to the original LaTeX PDF
2
+
3
+ The build reads two independent workbooks. It produces the approved 19-page PDF, with the same tables, red/black numbers, figures, captions and notes.
4
+
5
+ ## Run in Census
6
+
7
+ 1. Keep `data/charter_exhibits/disclosed_results.xlsx` fixed.
8
+ 2. Replace `data/charter_exhibits/pending_results.xlsx` with your actual pending results, retaining its sheets, headers and record keys. You can instead change `PENDING_INPUT` in `build_latex.py` to your new file path.
9
+ 3. Set `synthetic` to `false` in the pending workbook's `metadata` tab after replacing its synthetic values and chart inputs. Keep `role=pending` and `schema_version=2`. This only removes the synthetic footer; **it never switches off the fixed workbook**.
10
+ 4. Run `bash run_latex.sh` in the extracted portable folder. Output: `output/charter_exhibits_latex.pdf`.
11
+
12
+ In the sandbox, run `bash code/explore_codex/charter_exhibits/run_latex.sh` from the sandbox root. Input files are under `outputs/charter_exhibits/inputs/` and the PDF is under `outputs/charter_exhibits/`.
13
+
14
+ Python dependencies are listed in `latex_requirements.txt`; the runner uses uv. LaTeX requires `latexmk` and `pdflatex` (TeX Live or MacTeX). With dependencies already installed, `python build_latex.py` also runs the build. No network, live Overleaf connection, or live charter repository is needed. The sandbox charter repository is not modified.
15
+
16
+ ## Exclude tables or figures
17
+
18
+ Edit the two lists near the top of `build_latex.py`. They default to empty, so the complete deck still builds:
19
+
20
+ ```python
21
+ EXCLUDE_TABLES = []
22
+ EXCLUDE_FIGURES = []
23
+ ```
24
+
25
+ For example:
26
+
27
+ ```python
28
+ EXCLUDE_TABLES = ["Table A.4", "Table A.6"]
29
+ EXCLUDE_FIGURES = ["Figure A.1"]
30
+ ```
31
+
32
+ You may use either the printed label or its identifier below. Exclusions skip the complete exhibit, its generation, and inputs used only by that exhibit. Those unused rows or whole tabs may be absent from the workbooks. Shared inputs remain required by any included exhibit. Both workbooks and their `metadata` tabs remain required.
33
+
34
+ Remaining tables and figures retain their original numbers (e.g., Table 4 stays Table 4); page numbers run consecutively. References in the unchanged draft notes retain the original exhibit numbers, even when referring to an omitted exhibit. Clearing both lists restores everything. Invalid names and excluding every exhibit produce a clear error. The audit JSON records the included and excluded exhibits. These switches control whole tables/figures, not sample columns within an exhibit.
35
+
36
+ | Printed label | Identifier |
37
+ | --- | --- |
38
+ | Table 1 | `summary_statistics_table` |
39
+ | Table 2 | `balance_table` |
40
+ | Table 3 | `age25_earnings_table` |
41
+ | Table 4 | `earnings_heterogeneity_table` |
42
+ | Table 5 | `college_enrollment_completion_table` |
43
+ | Table 6 | `college_test_score_mediation_table` |
44
+ | Table 7 | `other_mediators_table` |
45
+ | Figure A.1 | `appendix-credo-distributions` |
46
+ | Figure A.2 | `appendix-college-by-age-placeholder` |
47
+ | Figure A.3 | `appendix-cmo-leave-one-out` |
48
+ | Figure A.4 | `appendix-earnings-by-age-firm-chars` |
49
+ | Table A.1 | `earnings_robustness_table` |
50
+ | Table A.2 | `earnings_threshold_table` |
51
+ | Table A.3 | `college_data_comparison_table` |
52
+ | Table A.4 | `college_heterogeneity_table` |
53
+ | Table A.5 | `college_mediation_additional_specs_table` |
54
+ | Table A.6 | `lottery_sample_selection_table` |
55
+ | Table A.7 | `cmo_details_table` |
56
+ | Table A.8 | `intervention_calculation_table` |
57
+
58
+ ## Fixed disclosed workbook
59
+
60
+ `disclosed_results.xlsx` contains only existing manuscript values from synced Overleaf commit `6ebdd8c1a84bcfd72cabf3c85a6b2a986cc5c416`, using numeric Excel cells rather than display strings for estimates.
61
+
62
+ | Tab | Input columns / purpose |
63
+ | --- | --- |
64
+ | `summary_stats` | `sample, variable, mean, sd, count, unit` |
65
+ | `TOT` | `sample, depvar, spec, coef, se, ccm, N_ppl, unit, coef_stars` |
66
+ | `ITT`, `first_stage` | Regression keys, `coef, se, control_mean, N_ppl, unit, coef_stars` |
67
+ | `OLS` | Regression keys, `coef, se, comparison_mean, N_ppl, unit, coef_stars` |
68
+ | `balance` | `sample, depvar, coef, se, control_mean, N_ppl, unit, coef_stars` |
69
+ | `mediation_conversions` | `depvar, earnings_gain, unit`; exact conversion values on the manuscript's existing scale |
70
+ | `interventions` | `program, estimated_effect, intervention, notes, years, unit`; source effects, durations and calculation notes |
71
+ | `metadata` | Input role, version and source snapshot |
72
+
73
+ Unreported sample sizes are blank. The summary table's `variable=n_ppl` rows store reported total people in `count`; no source sample sizes were invented or copied from synthetic data. Statistical stars supplied by the manuscript are retained. The previously corrected first-stage/ITT SE pairing is retained.
74
+
75
+ ## Pending workbook
76
+
77
+ `pending_results.xlsx` contains the red table entries and synthetic chart data. It currently reproduces the approved demonstration. Replace its data with your actual estimates. No old synthetic workbook or hidden numerical fallback is used.
78
+
79
+ - `summary_stats`, `TOT`, `OLS`, `balance`: the same keyed estimate format as the fixed file, for pending fields only.
80
+ - `predictions`: `sample, depvar, spec, coef, predicted_gain, unit`. Supply precomputed lognormal threshold effects in `coef` and mediation predicted earnings in `predicted_gain`. These are direct inputs, not calculations from the demonstration's old synthetic conversion factors. The sample workbook preserves the exact red predictions you approved.
81
+ - `sample_selection`: `step, applications, N_ppl`.
82
+ - `cmo_details`: `cmo, geographic_area, grades, lottery_years`. Grade and year ranges are text.
83
+ - `credo`: one CMO per row, with `math_effect`, `reading_effect`, `in_study` (0 or 1). Effects are in standard deviations.
84
+ - `TOT_by_cmo`: network regression estimates, SEs and CCMs; supplies the leave-one-network-out plot.
85
+ - `figure_reference`: the separately supplied pooled estimate used for the leave-one-network-out reference line. This preserves the approved synthetic chart independently of the fixed earnings table. Replace it with the reference estimate appropriate to your real chart.
86
+ - Age profiles remain in `TOT`, with `sample=age_profile` and `spec=observed` or `ein_fes`.
87
+ - `metadata`: keep `role=pending`, `schema_version=2`; update `synthetic` and `source` for real inputs.
88
+
89
+ The pending file is a template for the fields that remain to be filled. **A blank field alongside a populated field can belong to the fixed workbook.** For example, a subgroup row may contain only its pending `ccm`, while `coef` and `se` are fixed in the disclosed file. Leave those fixed fields blank in the pending file. The loader rejects conflicting nonblank fields instead of silently overwriting them. It also rejects duplicate keys and missing required rows or columns. Rows can be sorted freely.
90
+
91
+ ## Units, missingness and stars
92
+
93
+ - `unit=proportion`: store probabilities and probability effects on the 0-1 scale. For example, `0.056` displays as `5.6` percentage points. Means, SEs and CCMs use the same scale.
94
+ - `unit=usd`: unscaled dollars. `percentile`: rank points. `years` and `year`: years. `count`: whole people or applications.
95
+ - Threshold prediction inputs are percentage points (`3.3` means 3.3 pp); mediation prediction inputs are dollars (`834` means $834). The fixed mediation conversions retain the manuscript's exact source scale.
96
+ - Keep unknown numerical results empty. Zero is a real result and displays as zero. Missing values display as dashes.
97
+ - `coef_stars` accepts `*`, `**`, `***`, `none`, or `auto`. The supplied demo keeps the exact approved stars. When replacing estimates, supply your corresponding stars or set `auto` to calculate normal-approximation stars from `coef/se`. Table 6 deliberately displays its mediator coefficients without stars, as before.
98
+ - `N_ppl` and summary counts are retained as inputs; they appear only where the original paper layout includes them. Blank fixed Ns do not borrow pending counts.
99
+
100
+ ## Code and verification
101
+
102
+ - `build_latex.py`: two input paths, LaTeX document wrapper and compilation.
103
+ - `split_inputs.py`: keyed input validation for included exhibits, fixed/pending field separation, formatting and table rendering.
104
+ - `exhibit_selection.py`: exclusion names, source-block selection, original numbering and omitted-exhibit references.
105
+ - `table_bindings.json`: stable sample/outcome/specification-to-table-cell mappings and formatting; no fitted numerical estimates.
106
+ - `latex_figures.py`: workbook-driven charts, with the previously approved composition.
107
+ - `latex_source/`: original manuscript LaTeX templates, captions and draft notes.
108
+ - `verify_latex.py`: verifies row-order independence, rejects duplicate/overlapping inputs, and simulates a real pending replacement while checking all fixed cells remain unchanged.
109
+
110
+ `verify_exclusions.py` additionally tests missing excluded input sheets, shared dependencies, table-only and figure-only decks, original numbering, and restoring the full deck.
111
+
112
+ The build writes an audit JSON with both input paths and SHA-256 hashes, consumed keys, and layout hashes. The refactor was checked against the approved PDF: all 19 pages have identical text and the chart PNGs are byte-identical. Replacement testing leaves the fixed workbook byte-identical.
113
+
114
+ Only the two XLSX files and templates are needed at runtime. The earlier single-workbook `overleaf_values` mechanism and prototype synthetic calculation scripts are obsolete and excluded from the portable package. Existing draft caption/figure inconsistencies remain as approved; this refactor changes input ownership and code structure, not the paper's substantive content.
@@ -0,0 +1,125 @@
1
+ # --- Import necessary packages ---
2
+ import hashlib
3
+ import json
4
+ import os
5
+ import re
6
+ import shutil
7
+ import subprocess
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ sys.path.append('/Users/williamratnavale/Projects/codex_sandbox/code')
12
+ from config import HERE, OUTPUT_DIR
13
+ os.environ.setdefault('MPLCONFIGDIR',str(OUTPUT_DIR/'.matplotlib'))
14
+ from split_inputs import SplitInputs, build_tables
15
+ from latex_figures import make_figures
16
+ from pypdf import PdfReader
17
+ from exhibit_selection import select_exhibits
18
+
19
+ ###########################################################
20
+ # Define paths: this build never reads or writes the live charter repository
21
+ ###########################################################
22
+
23
+ LOCAL_INPUTS=HERE/'data/charter_exhibits'
24
+ INPUT_DIR=LOCAL_INPUTS if (LOCAL_INPUTS/'disclosed_results.xlsx').exists() else OUTPUT_DIR/'inputs'
25
+ # Keep DISCLOSED_INPUT fixed. Replace PENDING_INPUT with your pending actual results.
26
+ DISCLOSED_INPUT=INPUT_DIR/'disclosed_results.xlsx'
27
+ PENDING_INPUT=INPUT_DIR/'pending_results.xlsx'
28
+ LATEX_OUTPUT=(HERE/'output' if INPUT_DIR==LOCAL_INPUTS else OUTPUT_DIR)/'charter_exhibits_latex.pdf'
29
+ SOURCE=HERE/'latex_source'
30
+ BUILD=LATEX_OUTPUT.parent/'latex_build'
31
+
32
+ # Toggle whole exhibits here. Empty lists include everything.
33
+ # Examples: EXCLUDE_TABLES = ['Table A.4', 'lottery_sample_selection_table']
34
+ # EXCLUDE_FIGURES = ['Figure A.1', 'appendix-cmo-leave-one-out']
35
+ # All accepted labels/identifiers are listed in README_LATEX.md (README.md in ZIP).
36
+ EXCLUDE_TABLES = []
37
+ EXCLUDE_FIGURES = []
38
+
39
+
40
+ def build_latex(pending_path=PENDING_INPUT, output_path=LATEX_OUTPUT, build_dir=BUILD, disclosed_path=DISCLOSED_INPUT,
41
+ exclude_tables=None, exclude_figures=None):
42
+ main,appendix,included,excluded,references=select_exhibits(
43
+ SOURCE, EXCLUDE_TABLES if exclude_tables is None else exclude_tables,
44
+ EXCLUDE_FIGURES if exclude_figures is None else exclude_figures)
45
+ tables=[e['name'] for e in included if e['kind']=='table']
46
+ figures=[e['name'] for e in included if e['kind']=='figure']
47
+ data=SplitInputs(disclosed_path,pending_path,selected_tables=tables,selected_figures=figures)
48
+ build_dir=Path(build_dir);output_path=Path(output_path)
49
+ build_dir.mkdir(parents=True,exist_ok=True)
50
+ shutil.copytree(SOURCE/'paper',build_dir/'paper',dirs_exist_ok=True)
51
+ (build_dir/'outputs/scalars').mkdir(parents=True,exist_ok=True)
52
+ # The exhibit-only document does not consume paper prose scalars.
53
+ (build_dir/'outputs/scalars/scalars.sty').write_text('% No prose scalars are used by these exhibits.\n')
54
+ for folder in ['tables','figures']:
55
+ shutil.rmtree(build_dir/'outputs'/folder,ignore_errors=True)
56
+ derived=build_tables(data,SOURCE,build_dir)
57
+ make_figures(data,build_dir/'outputs/figures',selected_figures=figures)
58
+ (build_dir/'paper/main-tables.tex').write_text(main)
59
+ (build_dir/'paper/appendix.tex').write_text(appendix)
60
+ # Preserve the original captions, notes, float/page geometry and appendix numbering.
61
+ # Reference-only labels resolve references to the preceding (excluded) figures.
62
+ wrapper=r'''\documentclass[11pt,english]{article}
63
+ \makeatletter
64
+ \def\input@path{{paper/}{outputs/scalars/}}
65
+ \makeatother
66
+ \input{preamble}
67
+ \makeatletter
68
+ \AtBeginDocument{%
69
+ \newlabel{fig:earnings-early-adulthood}{{1}{1}{}{figure.1}{}}%
70
+ \newlabel{fig:earnings-distribution}{{2}{1}{}{figure.2}{}}%
71
+ \newlabel{fig:childhood-program-earnings}{{4}{1}{}{figure.4}{}}%
72
+ \bibcite{chetty2011star}{{2011}{Chetty et~al.}{{Chetty et~al.}}{}}%
73
+ \bibcite{credoCMO}{{2017}{CREDO}{{CREDO}}{}}%
74
+ }
75
+ \makeatother
76
+ \begin{document}
77
+ \input{main-tables}
78
+ \begin{appendices}
79
+ \input{appendix}
80
+ \end{appendices}
81
+ \end{document}
82
+ '''
83
+ if references:
84
+ wrapper=wrapper.replace(r'\AtBeginDocument{%',r'\AtBeginDocument{%'+'\n'+references)
85
+ if not any(e['section']=='appendix' for e in included):
86
+ wrapper=wrapper.replace('\\begin{appendices}\n\\input{appendix}\n\\end{appendices}', '')
87
+ if data.metadata['synthetic'].lower()=='true':
88
+ wrapper=wrapper.replace(r'\begin{document}',r'''\AddToShipoutPictureFG{\AtPageLowerLeft{\put(72,18){\makebox(0,0)[l]{\sffamily\fontsize{7}{8}\selectfont Synthetic inputs -- layout reproduced from manuscript}}}}
89
+ \begin{document}''')
90
+ mixed=data.metadata.get('synthetic','').lower()=='true'
91
+ if mixed:
92
+ wrapper=wrapper.replace('Synthetic inputs -- layout reproduced from manuscript',
93
+ 'Black table values: manuscript; red: synthetic placeholders. Plots: synthetic.')
94
+ (build_dir/'main.tex').write_text(wrapper)
95
+ engine=shutil.which('latexmk') or '/Library/TeX/texbin/latexmk'
96
+ command=[engine,'-pdf','-interaction=nonstopmode','-halt-on-error','-file-line-error','main.tex']
97
+ env=os.environ.copy();env['PATH']='/Library/TeX/texbin:'+env.get('PATH','')
98
+ result=subprocess.run(command,cwd=build_dir,env=env,capture_output=True,text=True)
99
+ (build_dir/'compile-output.txt').write_text(result.stdout+'\n'+result.stderr)
100
+ if result.returncode:
101
+ raise RuntimeError('LaTeX compilation failed:\n'+(result.stdout+result.stderr)[-4500:])
102
+ pdf=PdfReader(build_dir/'main.pdf')
103
+ text='\n'.join(p.extract_text() for p in pdf.pages)
104
+ if len(pdf.pages)!=len(included):
105
+ raise AssertionError(f'Expected {len(included)} exhibit pages, got {len(pdf.pages)}')
106
+ for page,exhibit in zip(pdf.pages,included):
107
+ if exhibit['title'].upper() not in page.extract_text():
108
+ raise AssertionError(f'Missing or renumbered exhibit: {exhibit["title"]}')
109
+ output_path.parent.mkdir(parents=True,exist_ok=True)
110
+ shutil.copyfile(build_dir/'main.pdf',output_path)
111
+ audit={'pages':len(pdf.pages),'input_sha256':{role:hashlib.sha256(path.read_bytes()).hexdigest() for role,path in data.paths.items()},
112
+ 'source_hashes':{str(p.relative_to(SOURCE)):hashlib.sha256(p.read_bytes()).hexdigest() for p in SOURCE.rglob('*.tex')},
113
+ 'derived':derived,'used_keys':[list(k) for k in sorted(data.used)],
114
+ 'input_paths':{role:str(path) for role,path in data.paths.items()},
115
+ 'binding_sha256':hashlib.sha256((HERE/'table_bindings.json').read_bytes()).hexdigest(),
116
+ 'derived_scope':'Fixed disclosed fields and separately supplied pending results',
117
+ 'included_exhibits':[e['title'] for e in included],
118
+ 'excluded_exhibits':[e['title'] for e in excluded]}
119
+ output_path.with_suffix('.audit.json').write_text(json.dumps(audit,indent=2)+'\n')
120
+ print(f'Created {output_path}: {len(pdf.pages)} pages using source LaTeX layouts')
121
+ return audit
122
+
123
+
124
+ if __name__=='__main__':
125
+ build_latex()
@@ -0,0 +1,27 @@
1
+ # --- Import necessary packages ---
2
+ import sys
3
+ from pathlib import Path
4
+
5
+ sys.path.append("/Users/williamratnavale/Projects/codex_sandbox/code")
6
+ try:
7
+ from paths import CODEX_SYNTHETIC, OUTPUTS
8
+ except ModuleNotFoundError:
9
+ # A copied folder also works on the inside without the sandbox paths module.
10
+ CODEX_SYNTHETIC = Path(__file__).resolve().parent / "data"
11
+ OUTPUTS = Path(__file__).resolve().parent / "output"
12
+
13
+ ###########################################################
14
+ # Define necessary paths
15
+ ###########################################################
16
+
17
+ HERE = Path(__file__).resolve().parent
18
+ # Replace this one path with your real workbook. The renderer never generates data.
19
+ BUNDLED_INPUT = HERE / "data" / "charter_exhibits" / "synthetic_estimates.xlsx"
20
+ INPUT_XLSX = BUNDLED_INPUT if BUNDLED_INPUT.exists() else CODEX_SYNTHETIC / "charter_exhibits" / "synthetic_estimates.xlsx"
21
+ OUTPUT_DIR = HERE / "output" if BUNDLED_INPUT.exists() else OUTPUTS / "charter_exhibits"
22
+ OUTPUT_PDF = OUTPUT_DIR / "charter_exhibits.pdf"
23
+ SOURCE_SNAPSHOT = HERE / "source_snapshot.json"
24
+
25
+ # Blank values require an explicit status, rather than becoming zero.
26
+ STATUSES = {"available", "unavailable", "not_applicable"}
27
+ PAGE_SIZE = (960, 540)