ltc-code 0.2.31__tar.gz → 0.2.32__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. ltc_code-0.2.32/PKG-INFO +20 -0
  2. ltc_code-0.2.32/README.md +10 -0
  3. {ltc_code-0.2.31 → ltc_code-0.2.32}/pyproject.toml +1 -1
  4. ltc_code-0.2.32/src/ltc_code/nsc/EXTENSIONS.md +102 -0
  5. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/build_nsc_outcomes_new.py +91 -20
  6. ltc_code-0.2.32/src/ltc_code/plot_colleges.py +74 -0
  7. ltc_code-0.2.31/PKG-INFO +0 -10
  8. ltc_code-0.2.31/README.md +0 -0
  9. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
  10. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
  11. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
  12. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
  13. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
  14. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
  15. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
  16. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
  17. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
  18. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
  19. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
  20. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
  21. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
  22. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
  23. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
  24. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
  25. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
  26. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
  27. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
  28. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
  29. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
  30. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
  31. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
  32. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
  33. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
  34. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
  35. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
  36. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
  37. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
  38. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
  39. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
  40. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/__init__.py +0 -0
  41. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/aspire.py +0 -0
  42. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/check_cmo_apps.do +0 -0
  43. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/christel_house.py +0 -0
  44. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/green_dot.py +0 -0
  45. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/helpers.py +0 -0
  46. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/june13.py +0 -0
  47. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/june2.py +0 -0
  48. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/june30.py +0 -0
  49. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/june5.py +0 -0
  50. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/june7.py +0 -0
  51. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/kipp_nj.py +0 -0
  52. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/kipp_tx.py +0 -0
  53. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/main.py +0 -0
  54. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/make_summary_stats_table.py +0 -0
  55. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/mappings.py +0 -0
  56. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/may27.py +0 -0
  57. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/NEW_CROSSWALK.md +0 -0
  58. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/OUTCOMES.md +0 -0
  59. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/__init__.py +0 -0
  60. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/build_nsc_outcomes.py +0 -0
  61. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
  62. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
  63. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
  64. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/naics.csv +0 -0
  65. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/naics.py +0 -0
  66. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
  67. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
  68. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
  69. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
  70. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
  71. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
  72. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
  73. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
  74. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/directory.dta +0 -0
  75. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
  76. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
  77. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/plot_bars.py +0 -0
  78. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/polars_dates.py +0 -0
  79. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/rocketship.py +0 -0
  80. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/schema_mapping.py +0 -0
  81. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
  82. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
  83. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
  84. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/signal_var_calcs.py +0 -0
  85. {ltc_code-0.2.31 → ltc_code-0.2.32}/src/ltc_code/yes_prep.py +0 -0
@@ -0,0 +1,20 @@
1
+ Metadata-Version: 2.3
2
+ Name: ltc-code
3
+ Version: 0.2.32
4
+ Summary: Add your description here
5
+ Requires-Dist: fastexcel>=0.16,<0.20
6
+ Requires-Dist: polars>=1.36.1,<1.42
7
+ Requires-Dist: polars-readstat>=0.20.2
8
+ Requires-Python: >=3.9
9
+ Description-Content-Type: text/markdown
10
+
11
+
12
+
13
+ ### College scatter plots (0.2.32)
14
+
15
+ `from ltc_code.plot_colleges import plot_colleges` provides an OI-themed
16
+ scatter with explicit Polars `x`/`y` aggregations, college grouping, optional
17
+ filters and categorical color groups. Plotting additionally requires
18
+ `oi-tools`, `plotnine`, and `pyarrow` in the analysis environment; they are
19
+ not required to run the NSC build. See [NSC extensions](src/ltc_code/nsc/EXTENSIONS.md)
20
+ for attendance definitions, coarsening names, and plotting examples.
@@ -0,0 +1,10 @@
1
+
2
+
3
+ ### College scatter plots (0.2.32)
4
+
5
+ `from ltc_code.plot_colleges import plot_colleges` provides an OI-themed
6
+ scatter with explicit Polars `x`/`y` aggregations, college grouping, optional
7
+ filters and categorical color groups. Plotting additionally requires
8
+ `oi-tools`, `plotnine`, and `pyarrow` in the analysis environment; they are
9
+ not required to run the NSC build. See [NSC extensions](src/ltc_code/nsc/EXTENSIONS.md)
10
+ for attendance definitions, coarsening names, and plotting examples.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ltc-code"
3
- version = "0.2.31"
3
+ version = "0.2.32"
4
4
  description = "Add your description here"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.9"
@@ -0,0 +1,102 @@
1
+ # NSC attendance and coarsening changes in the sandbox
2
+
3
+ The package build is `ltc_code/nsc/build_nsc_outcomes_new.py`.
4
+ Import the scatter helper with `from ltc_code.plot_colleges import plot_colleges`.
5
+
6
+ Version 0.2.32 changes `coarse_t12`, `coarse_t34`, and `coarse_t58` to the
7
+ overall four-year benchmark; their former within-group calculations are
8
+ retained as `coarse_t12_pool`, `coarse_t34_pool`, and `coarse_t58_pool`.
9
+
10
+ ## Two attendance measures
11
+
12
+ For each age 18–26 and each grouping `any`, `4yr`, `2yr`:
13
+
14
+ | Example | Counts |
15
+ | --- | --- |
16
+ | `att_any_1098_20` | F, Q, H, plus unknown statuses under the existing policy |
17
+ | `att_any_1098_all_20` | The same, plus L (less than half-time) |
18
+
19
+ Both versions exclude W (withdrawn), A (leave), D (deceased), and other
20
+ non-enrollment statuses. The existing handling of generic part-time text is
21
+ retained. Q now qualifies, and written-out "less than half-time" is classified
22
+ before matching "half-time". Adjacent spells with different eligibility are
23
+ kept separate rather than assigning the highest status to the entire period.
24
+
25
+ The original names are retained. There are no duplicate `_fulltime_` columns.
26
+
27
+ Both measures use the available tax years 2011–2014 and 2016–2022. Observable
28
+ years without counted attendance get zero; unavailable years remain missing.
29
+ Four-year attendance takes priority separately within each version. These
30
+ measures retain the existing NSC matching, record deduplication, and weekly
31
+ institution selection; they cannot recover records missing from the source.
32
+
33
+ ## Coarsening: preserved and additional variables
34
+
35
+ Names without `_pool` use the overall four-year benchmark only for the named
36
+ tiers. Names ending in `_pool` use each named group's own pooled rate.
37
+ All names below start with `adj_cmp_rate_4yr_coarse_`:
38
+
39
+ | Suffix | Calculation |
40
+ | --- | --- |
41
+ | `t12` | Replace only tiers 1–2 with the overall four-year average |
42
+ | `t34` | Replace only tiers 3–4 with the overall four-year average |
43
+ | `t58` | Replace only tiers 5–8 with the overall four-year average |
44
+ | `t12_pool` | Replace tiers 1–2 with their own pooled average |
45
+ | `t34_pool` | Replace tiers 3–4 with their own pooled average |
46
+ | `t58_pool` | Replace tiers 5–8 with their own pooled average |
47
+ | `t12_t34_pool` | Pool 1–2 and, separately, 3–4 |
48
+ | `t14_t58_pool` | Pool 1–4 and, separately, 5–8 |
49
+
50
+ Every average uses total IPEDS completers divided by the total adjusted
51
+ graduation cohort in the relevant group. Non-target colleges keep the detailed
52
+ four-year prediction. Existing zero and missing-value rules are preserved.
53
+ The separate `adj_cmp_rate_coarse_4yr` still replaces ALL four-year starters'
54
+ rates with the common four-year average.
55
+
56
+ ## Simple college scatter
57
+
58
+ ```python
59
+ plot = plot_colleges(
60
+ df,
61
+ x=pl.col("sid_cepr").drop_nulls().n_unique(),
62
+ y=pl.col("completion_rate_150pct_firstinst").drop_nulls().first() * 100,
63
+ group_by="unitid_firstinst",
64
+ filters={"enrolled": 1, "tier_firstinst": [3, 4]},
65
+ title="First colleges of charter enrollers: tiers 3–4",
66
+ )
67
+ plots.append(plot)
68
+ ```
69
+
70
+ Replace `enrolled` with your charter-enrollment column. Omit `filters` for everyone,
71
+ or supply just one dictionary entry. Polars expressions and lists of expressions
72
+ also work for more general conditions. Filters apply before the x/y aggregation.
73
+ The title and axis labels are customizable. x and y are Polars aggregation
74
+ expressions; their choice determines what each college's dot represents.
75
+
76
+ `unitid_firstinst` is the matched IPEDS ID of the first qualifying institution
77
+ within the by-Y4 attendance window. `completion_rate_150pct_firstinst` is that
78
+ institution's raw graduation rate. This example describes first colleges, not
79
+ every college ever attended or the institution awarding the degree. Students
80
+ without a matched first-college ID and colleges without a rate do not appear.
81
+ Use a consistent institutional rate per college: `.first()` selects that rate,
82
+ not an average of students' predicted completion outcomes.
83
+
84
+ The helper returns a plotnine plot; `return_data=True` returns `(plot, data)`.
85
+ It does not save files. Call `.save(...)` on the returned plot to export it.
86
+
87
+ ### OI theme and color groups
88
+
89
+ The scatter uses `oi_tools.figures.theme_oi()` (the same theme as `plot_bars`),
90
+ with no grid lines and clear black axes. The analysis environment needs
91
+ `oi-tools`, `plotnine`, and `pyarrow`.
92
+
93
+ Add `color="enrolled"` to draw separate dots for each college's enrollers and
94
+ non-enrollers. The function aggregates x and y separately within each
95
+ college/color group, after applying filters. Numeric codes become discrete
96
+ legend labels. For descriptive labels, pass a column containing strings such
97
+ as "Enrollers" and "Non-enrollers". Missing color values are labeled "Missing".
98
+ The OI palette supplies the colors. Omit `color` for one dot per college.
99
+
100
+ Because the institutional graduation rate is the same for both groups, their
101
+ dots have the same y coordinate; x differs with the number of unique students.
102
+ If both counts are identical, their dots overlap.
@@ -914,7 +914,7 @@ nsc = nsc.with_columns(
914
914
  ###########################################################
915
915
 
916
916
  # Sarah counts half-time or more as an attended enrollment spell.
917
- enrollment_status = pl.col("enrollment").str.to_uppercase().fill_null("")
917
+ enrollment_status = pl.col("enrollment").str.to_uppercase().str.strip_chars().fill_null("")
918
918
 
919
919
  enroll = (
920
920
  nsc.filter(pl.col("graduated") == "N")
@@ -935,16 +935,19 @@ enroll = (
935
935
  .then(5)
936
936
  .when(enrollment_status.str.contains("PART", literal=False))
937
937
  .then(4)
938
+ .when(enrollment_status.str.contains("LESS|\\bL\\b", literal=False))
939
+ .then(1)
938
940
  .when(enrollment_status.str.contains("HALF|\\bH\\b", literal=False))
939
941
  .then(3)
940
942
  .when(enrollment_status.str.contains("QUARTER|\\bQ\\b", literal=False))
941
- .then(2)
942
- .when(enrollment_status.str.contains("LESS|\\bL\\b", literal=False))
943
- .then(1)
943
+ .then(4) # Three-quarter time qualifies as at least half-time.
944
944
  .otherwise(0)
945
945
  .alias("_enrollment_score")
946
946
  )
947
- .with_columns((pl.col("_enrollment_score") >= 3).cast(pl.Int8).alias("_term_att"))
947
+ .with_columns(
948
+ (pl.col("_enrollment_score") >= 3).cast(pl.Int8).alias("_term_att"),
949
+ (pl.col("_enrollment_score") >= 1).cast(pl.Int8).alias("_term_att_all"),
950
+ )
948
951
  .with_columns(
949
952
  ((pl.col("term_start_date").cast(pl.Int64) // 7).cast(pl.Int64)).alias(
950
953
  "week_start"
@@ -1035,6 +1038,7 @@ enroll_weeks = (
1035
1038
  "week_start",
1036
1039
  "week_end",
1037
1040
  "_term_att",
1041
+ "_term_att_all",
1038
1042
  "_enrollment_score",
1039
1043
  "_enrollment_stem_sarah",
1040
1044
  "_enrollment_stem_dhs",
@@ -1068,6 +1072,8 @@ enroll_weeks = (
1068
1072
  (pl.col("week") != pl.col("week").shift(1) + 1)
1069
1073
  | (pl.col("sid_cepr") != pl.col("sid_cepr").shift(1))
1070
1074
  | (pl.col("ID_FSC") != pl.col("ID_FSC").shift(1))
1075
+ | (pl.col("_term_att") != pl.col("_term_att").shift(1))
1076
+ | (pl.col("_term_att_all") != pl.col("_term_att_all").shift(1))
1071
1077
  | (
1072
1078
  pl.col("_enrollment_stem_sarah")
1073
1079
  != pl.col("_enrollment_stem_sarah").shift(1)
@@ -1102,6 +1108,7 @@ enroll = (
1102
1108
  pl.col("k_mean").drop_nulls().first(),
1103
1109
  pl.col("completion_rate_150pct_ip").drop_nulls().first(),
1104
1110
  pl.col("_term_att").max(),
1111
+ pl.col("_term_att_all").max(),
1105
1112
  pl.col("_enrollment_stem_sarah").max(),
1106
1113
  pl.col("_enrollment_stem_dhs").max(),
1107
1114
  pl.col("week").min().alias("week_start"),
@@ -1216,10 +1223,11 @@ for age in AGE_ATTENDANCE_RANGE:
1216
1223
  (
1217
1224
  (pl.col("term_start_date") <= pl.col("_age_window_end"))
1218
1225
  & (pl.col("term_end_date") >= pl.col("_age_window_start"))
1219
- & (pl.col("_term_att") == 1)
1220
1226
  )
1221
1227
  .cast(pl.Int8)
1222
- .alias("_overlaps")
1228
+ .alias("_overlaps_all")
1229
+ ).with_columns(
1230
+ (pl.col("_overlaps_all") * pl.col("_term_att")).alias("_overlaps")
1223
1231
  )
1224
1232
 
1225
1233
  enrollment_pieces.append(
@@ -1236,6 +1244,19 @@ for age in AGE_ATTENDANCE_RANGE:
1236
1244
  .alias(f"att_{college_type}_{age}")
1237
1245
  for college_type in AGE_ATTENDANCE_TYPES
1238
1246
  ]
1247
+ + [
1248
+ (
1249
+ pl.col("_overlaps_all")
1250
+ * pl.col("_term_att_all")
1251
+ * (
1252
+ pl.lit(1) if college_type == "any"
1253
+ else pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
1254
+ )
1255
+ )
1256
+ .max()
1257
+ .alias(f"att_{college_type}_all_{age}")
1258
+ for college_type in ["any", "4yr", "2yr"]
1259
+ ]
1239
1260
  + [
1240
1261
  expression
1241
1262
  for definition in STEM_DEFINITIONS
@@ -1942,7 +1963,7 @@ ipeds_tier_completion = ipeds_completion_with_sector.join(
1942
1963
  ).filter(pl.col("college_sector") == "4yr")
1943
1964
 
1944
1965
  # Coarsen one bucket at a time, always starting from the detailed prediction.
1945
- tier_buckets = {"t12": [1, 2], "t34": [3, 4], "t58": [5, 6, 7, 8]}
1966
+ tier_buckets = {"t12_pool": [1, 2], "t34_pool": [3, 4], "t58_pool": [5, 6, 7, 8]}
1946
1967
  for label, tiers in tier_buckets.items():
1947
1968
  bucket_rate = (
1948
1969
  ipeds_tier_completion.filter(pl.col("tier").is_in(tiers))
@@ -1964,18 +1985,66 @@ for label, tiers in tier_buckets.items():
1964
1985
  .alias(f"adj_cmp_rate_4yr_coarse_{label}")
1965
1986
  )
1966
1987
 
1967
- # Match the available 1098-T calendar years, including the missing 2015 year.
1968
- tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
1969
- nsc_outcomes = nsc_outcomes.with_columns(
1970
- [
1971
- pl.when((pl.col("cohort_lottery") + age).is_in(tax_years))
1972
- .then(pl.col(f"att_any_{age}"))
1973
- .otherwise(None)
1974
- .alias(f"att_any_1098_{age}")
1975
- for age in AGE_ATTENDANCE_RANGE
1976
- ]
1988
+ # Additional coarsening scenarios. Each listed group gets its own pooled
1989
+ # IPEDS rate; institutions outside those groups keep their detailed prediction.
1990
+ tier_coarsening_scenarios = {
1991
+ "t12_t34_pool": [[1, 2], [3, 4]],
1992
+ "t14_t58_pool": [[1, 2, 3, 4], [5, 6, 7, 8]],
1993
+ }
1994
+ for label, groups in tier_coarsening_scenarios.items():
1995
+ prediction = pl.col("adj_cmp_rate_4yr")
1996
+ for tiers in groups:
1997
+ group_rate = (
1998
+ ipeds_tier_completion.filter(pl.col("tier").is_in(tiers))
1999
+ .select(pl.col("completers_150pct_ip").sum() / pl.col("cohort_adj_150pct_ip").sum())
2000
+ .item()
2001
+ )
2002
+ if group_rate is None or not 0 <= group_rate <= 1:
2003
+ raise ValueError(f"Invalid pooled completion rate for tiers {tiers}: {group_rate}")
2004
+ prediction = pl.when(
2005
+ (pl.col("college_years_firstinst") == 4)
2006
+ & pl.col("tier_firstinst").is_in(tiers)
2007
+ & (pl.col("att_any_byY4") == 1)
2008
+ & pl.col("adj_cmp_rate_4yr").is_not_null()
2009
+ ).then(group_rate).otherwise(prediction)
2010
+ nsc_outcomes = nsc_outcomes.with_columns(
2011
+ prediction.alias(f"adj_cmp_rate_4yr_coarse_{label}")
2012
+ )
1977
2013
 
1978
- )
2014
+ # Replace only the named tiers with the common overall four-year average.
2015
+ # The _pool versions above instead use each group's own IPEDS average.
2016
+ benchmark_tier_buckets = {"t12": [1, 2], "t34": [3, 4], "t58": [5, 6, 7, 8]}
2017
+ for label, tiers in benchmark_tier_buckets.items():
2018
+ nsc_outcomes = nsc_outcomes.with_columns(
2019
+ pl.when(
2020
+ (pl.col("college_years_firstinst") == 4)
2021
+ & pl.col("tier_firstinst").is_in(tiers)
2022
+ & (pl.col("att_any_byY4") == 1)
2023
+ & pl.col("adj_cmp_rate_4yr").is_not_null()
2024
+ ).then(cmp_rate_4yr_coarse).otherwise(pl.col("adj_cmp_rate_4yr"))
2025
+ .alias(f"adj_cmp_rate_4yr_coarse_{label}")
2026
+ )
2027
+
2028
+ # Match available 1098-T calendar years, including the missing 2015 year.
2029
+ # Existing names retain AT LEAST HALF-TIME (including Q), with the existing
2030
+ # unknown-status policy. "all" adds less-than-half-time enrollment, excluding W/A/D and
2031
+ # other non-enrollment statuses. Unknown statuses count in both versions.
2032
+ tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
2033
+ tax_attendance_columns = []
2034
+ for college_type in ["any", "4yr", "2yr"]:
2035
+ for age in AGE_ATTENDANCE_RANGE:
2036
+ for label, source in [
2037
+ ("", f"att_{college_type}_{age}"),
2038
+ ("_all", f"att_{college_type}_all_{age}"),
2039
+ ]:
2040
+ column = f"att_{college_type}_1098{label}_{age}"
2041
+ tax_attendance_columns.append(column)
2042
+ nsc_outcomes = nsc_outcomes.with_columns(
2043
+ pl.when((pl.col("cohort_lottery") + age).is_in(tax_years))
2044
+ .then(pl.col(source))
2045
+ .otherwise(None)
2046
+ .alias(column)
2047
+ )
1979
2048
 
1980
2049
 
1981
2050
  ###########################################################
@@ -2066,6 +2135,8 @@ keep_columns = [
2066
2135
  "adj_cmp_rate_coarse_2yr",
2067
2136
  "adj_cmp_rate_coarse_4yr",
2068
2137
  *[f"adj_cmp_rate_4yr_coarse_{label}" for label in tier_buckets],
2138
+ *[f"adj_cmp_rate_4yr_coarse_{label}" for label in tier_coarsening_scenarios],
2139
+ *[f"adj_cmp_rate_4yr_coarse_{label}" for label in benchmark_tier_buckets],
2069
2140
  "adj_cmp_rate_coarse",
2070
2141
  "adj_cmp_rate_coarse_sample",
2071
2142
  "adj_cmp_rate_coarsen_4yr",
@@ -2076,7 +2147,7 @@ keep_columns = [
2076
2147
  "college_years_firstinst",
2077
2148
  "tier_firstinst",
2078
2149
  "completion_rate_150pct_firstinst",
2079
- ] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
2150
+ ] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns + tax_attendance_columns
2080
2151
  keep_columns = list(dict.fromkeys(keep_columns))
2081
2152
 
2082
2153
  nsc_outcomes_final = nsc_outcomes.select(keep_columns)
@@ -0,0 +1,74 @@
1
+ # --- Import necessary packages ---
2
+ import polars as pl
3
+
4
+
5
+ def plot_colleges(
6
+ df: pl.DataFrame,
7
+ *,
8
+ x: pl.Expr,
9
+ y: pl.Expr,
10
+ group_by: str,
11
+ filters=None,
12
+ color: str = None,
13
+ title: str = "College graduation rates",
14
+ x_label: str = "Unique students",
15
+ y_label: str = "150% graduation rate (%)",
16
+ return_data: bool = False,
17
+ ):
18
+ """Filter rows and aggregate x/y by college, optionally split by color.
19
+
20
+ x and y are Polars aggregation expressions, e.g. col('sid_cepr').n_unique()
21
+ and col('rate').drop_nulls().first() * 100. Supply a raw institution rate
22
+ with a consistent value per college, not a student-level prediction.
23
+ filters accepts a dictionary such as {'enrolled': 1, 'tier': [3, 4]},
24
+ a Polars expression, or a list of expressions. Filtering precedes counting.
25
+ color names a categorical column: each college/color group gets its own
26
+ dot and its own x/y aggregation. Numeric group codes are treated as labels.
27
+ Missing college IDs and missing/nonfinite x or y are omitted. No files
28
+ are saved unless the caller explicitly saves the returned plot.
29
+ """
30
+ from plotnine import aes, element_blank, element_line, geom_point, ggplot, labs, theme
31
+ from oi_tools.figures import OIColors, scale_color_oi, theme_oi
32
+
33
+ if isinstance(filters, dict):
34
+ filters = [
35
+ pl.col(column).is_in(value) if isinstance(value, (list, tuple, set))
36
+ else pl.col(column).is_null() if value is None
37
+ else pl.col(column) == value
38
+ for column, value in filters.items()
39
+ ]
40
+ if filters is not None:
41
+ df = df.filter(*filters) if isinstance(filters, (list, tuple)) else df.filter(filters)
42
+ groups = list(dict.fromkeys([group_by] + ([color] if color else [])))
43
+ colleges = (
44
+ df.drop_nulls(group_by)
45
+ .group_by(groups)
46
+ .agg(x.alias('_x'), y.alias('_y'))
47
+ .filter(pl.col('_x').is_finite() & pl.col('_y').is_finite())
48
+ .sort(groups)
49
+ )
50
+ if colleges.is_empty():
51
+ raise ValueError('No colleges with both x and y remain after filtering.')
52
+ mapping = aes(x='_x', y='_y')
53
+ points = geom_point(alpha=0.8, size=2.5, color=OIColors.BLUE)
54
+ if color is not None:
55
+ colleges = colleges.with_columns(
56
+ pl.col(color).cast(pl.String).fill_null('Missing').alias('_color')
57
+ )
58
+ mapping = aes(x='_x', y='_y', color='_color')
59
+ points = geom_point(alpha=0.8, size=2.5)
60
+ plot = (
61
+ ggplot(colleges, mapping)
62
+ + points
63
+ + theme_oi()
64
+ + labs(title=title, x=x_label, y=y_label,
65
+ caption=f'{colleges[group_by].n_unique():,} colleges with observed values')
66
+ + theme(
67
+ panel_grid=element_blank(),
68
+ axis_line=element_line(color='black', size=0.7),
69
+ figure_size=(7, 5),
70
+ )
71
+ )
72
+ if color is not None:
73
+ plot += scale_color_oi()
74
+ return (plot, colleges) if return_data else plot
ltc_code-0.2.31/PKG-INFO DELETED
@@ -1,10 +0,0 @@
1
- Metadata-Version: 2.3
2
- Name: ltc-code
3
- Version: 0.2.31
4
- Summary: Add your description here
5
- Requires-Dist: fastexcel>=0.16,<0.20
6
- Requires-Dist: polars>=1.36.1,<1.42
7
- Requires-Dist: polars-readstat>=0.20.2
8
- Requires-Python: >=3.9
9
- Description-Content-Type: text/markdown
10
-
ltc_code-0.2.31/README.md DELETED
File without changes