ltc-code 0.2.20__tar.gz → 0.2.22__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ltc_code-0.2.20 → ltc_code-0.2.22}/PKG-INFO +1 -1
- {ltc_code-0.2.20 → ltc_code-0.2.22}/pyproject.toml +1 -1
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/build_nsc_outcomes.py +256 -11
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/.DS_Store +0 -0
- ltc_code-0.2.22/src/ltc_code/signal_var_calcs.py +41 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/README.md +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/.DS_Store +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/__init__.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/aspire.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/check_cmo_apps.do +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/christel_house.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/green_dot.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/helpers.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/june13.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/june2.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/june30.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/june5.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/june7.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/kipp_nj.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/kipp_tx.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/main.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/make_summary_stats_table.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/mappings.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/may27.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/.DS_Store +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/__init__.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/input/.gitkeep +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/naics.csv +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/naics.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/output/.gitkeep +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/directory.dta +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/polars_dates.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/rocketship.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/schema_mapping.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
- {ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/yes_prep.py +0 -0
|
@@ -86,6 +86,8 @@ NSC_CUTOFF_DATE = date(2026, 12, 31)
|
|
|
86
86
|
RANDOM_SEED = 3852804
|
|
87
87
|
STEM_CIP_FAMILIES = {11, 14, 15, 26, 27, 40, 41}
|
|
88
88
|
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
|
|
89
|
+
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
|
|
90
|
+
AGE_ATTENDANCE_RANGE = range(18, 27)
|
|
89
91
|
|
|
90
92
|
|
|
91
93
|
###########################################################
|
|
@@ -235,11 +237,13 @@ chetty_college_outcomes = (
|
|
|
235
237
|
missing_string_as_null=True,
|
|
236
238
|
value_labels_as_strings=False,
|
|
237
239
|
)
|
|
238
|
-
.select("super_opeid", "tier", "tier_name", "k_mean")
|
|
240
|
+
.select("super_opeid", "tier", "tier_name", "iclevel", "count", "k_mean")
|
|
239
241
|
.collect()
|
|
240
242
|
.with_columns(
|
|
241
243
|
pl.col("super_opeid").cast(pl.Int64, strict=False),
|
|
242
244
|
pl.col("tier").cast(pl.Int8, strict=False),
|
|
245
|
+
pl.col("iclevel").cast(pl.Int8, strict=False),
|
|
246
|
+
pl.col("count").cast(pl.Float64, strict=False),
|
|
243
247
|
pl.col("k_mean").cast(pl.Float64, strict=False),
|
|
244
248
|
)
|
|
245
249
|
.drop_nulls("super_opeid")
|
|
@@ -253,6 +257,38 @@ k_mean_insuffdata = chetty_college_outcomes.filter(
|
|
|
253
257
|
pl.col("super_opeid") == -1
|
|
254
258
|
).item(0, "k_mean")
|
|
255
259
|
|
|
260
|
+
# Collapse MRC college-specific earnings to national student-weighted means.
|
|
261
|
+
# MRC count is the mean number of children per cohort represented by each row.
|
|
262
|
+
mrc_sector_means = (
|
|
263
|
+
chetty_college_outcomes
|
|
264
|
+
.filter(
|
|
265
|
+
(pl.col("super_opeid") > 0)
|
|
266
|
+
& pl.col("iclevel").is_in([1, 2, 3])
|
|
267
|
+
& (pl.col("count") > 0)
|
|
268
|
+
& pl.col("k_mean").is_not_null()
|
|
269
|
+
)
|
|
270
|
+
.with_columns(
|
|
271
|
+
pl.when(pl.col("iclevel") == 1)
|
|
272
|
+
.then(pl.lit("4yr"))
|
|
273
|
+
.otherwise(pl.lit("2yr_or_less"))
|
|
274
|
+
.alias("college_sector")
|
|
275
|
+
)
|
|
276
|
+
.group_by("college_sector")
|
|
277
|
+
.agg(
|
|
278
|
+
(
|
|
279
|
+
(pl.col("k_mean") * pl.col("count")).sum()
|
|
280
|
+
/ pl.col("count").sum()
|
|
281
|
+
).alias("k_mean_coarse")
|
|
282
|
+
)
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
k_mean_4yr_coarse = mrc_sector_means.filter(
|
|
286
|
+
pl.col("college_sector") == "4yr"
|
|
287
|
+
).item(0, "k_mean_coarse")
|
|
288
|
+
k_mean_2yr_coarse = mrc_sector_means.filter(
|
|
289
|
+
pl.col("college_sector") == "2yr_or_less"
|
|
290
|
+
).item(0, "k_mean_coarse")
|
|
291
|
+
|
|
256
292
|
chetty_ope_crosswalk = (
|
|
257
293
|
scan_readstat(
|
|
258
294
|
CHETTY_MRC_TABLE11,
|
|
@@ -275,24 +311,102 @@ chetty_by_opeid = (
|
|
|
275
311
|
.unique("opeid", keep="first")
|
|
276
312
|
)
|
|
277
313
|
|
|
278
|
-
# IPEDS
|
|
279
|
-
#
|
|
314
|
+
# The IPEDS 150%-of-normal-time rate is six-year completion at four-year
|
|
315
|
+
# institutions and three-year completion at two-year institutions.
|
|
280
316
|
ipeds_completion = (
|
|
281
317
|
scan_readstat(
|
|
282
318
|
IPEDS_DATA,
|
|
283
319
|
missing_string_as_null=True,
|
|
284
320
|
value_labels_as_strings=False,
|
|
285
321
|
)
|
|
286
|
-
.select(
|
|
322
|
+
.select(
|
|
323
|
+
"unitid",
|
|
324
|
+
"completion_rate_150pct_ip",
|
|
325
|
+
"cohort_adj_150pct_ip",
|
|
326
|
+
"completers_150pct_ip",
|
|
327
|
+
)
|
|
287
328
|
.collect()
|
|
288
329
|
.with_columns(
|
|
289
330
|
pl.col("unitid").cast(pl.Int64, strict=False),
|
|
290
331
|
pl.col("completion_rate_150pct_ip").cast(pl.Float64, strict=False),
|
|
332
|
+
pl.col("cohort_adj_150pct_ip").cast(pl.Float64, strict=False),
|
|
333
|
+
pl.col("completers_150pct_ip").cast(pl.Float64, strict=False),
|
|
291
334
|
)
|
|
292
335
|
.drop_nulls("unitid")
|
|
293
336
|
.unique("unitid", keep="first")
|
|
294
337
|
)
|
|
295
338
|
|
|
339
|
+
# Use the graduation cohort itself as the student weight. Summing completers and
|
|
340
|
+
# cohorts is equivalent to weighting each college's rate by its IPEDS denominator.
|
|
341
|
+
ipeds_sector_rates = (
|
|
342
|
+
ipeds_completion
|
|
343
|
+
.join(
|
|
344
|
+
ipeds_2013.select("unitid", "years_ipeds_2013"),
|
|
345
|
+
on="unitid",
|
|
346
|
+
how="left",
|
|
347
|
+
validate="1:1",
|
|
348
|
+
)
|
|
349
|
+
.join(
|
|
350
|
+
ipeds_manual.select("unitid", "years_manual"),
|
|
351
|
+
on="unitid",
|
|
352
|
+
how="left",
|
|
353
|
+
validate="1:1",
|
|
354
|
+
)
|
|
355
|
+
.join(
|
|
356
|
+
directory.select("unitid", "years_directory"),
|
|
357
|
+
on="unitid",
|
|
358
|
+
how="left",
|
|
359
|
+
validate="1:1",
|
|
360
|
+
)
|
|
361
|
+
.with_columns(
|
|
362
|
+
pl.coalesce(
|
|
363
|
+
[
|
|
364
|
+
pl.col("years_manual"),
|
|
365
|
+
pl.col("years_ipeds_2013"),
|
|
366
|
+
pl.col("years_directory"),
|
|
367
|
+
]
|
|
368
|
+
).alias("college_years")
|
|
369
|
+
)
|
|
370
|
+
.filter(
|
|
371
|
+
pl.col("college_years").is_in([1, 2, 4])
|
|
372
|
+
& (pl.col("cohort_adj_150pct_ip") > 0)
|
|
373
|
+
& pl.col("completers_150pct_ip").is_not_null()
|
|
374
|
+
)
|
|
375
|
+
.with_columns(
|
|
376
|
+
pl.when(pl.col("college_years") == 4)
|
|
377
|
+
.then(pl.lit("4yr"))
|
|
378
|
+
.otherwise(pl.lit("2yr_or_less"))
|
|
379
|
+
.alias("college_sector")
|
|
380
|
+
)
|
|
381
|
+
.group_by("college_sector")
|
|
382
|
+
.agg(
|
|
383
|
+
pl.col("completers_150pct_ip").sum().alias("completers"),
|
|
384
|
+
pl.col("cohort_adj_150pct_ip").sum().alias("cohort"),
|
|
385
|
+
)
|
|
386
|
+
.with_columns((pl.col("completers") / pl.col("cohort")).alias("cmp_rate_coarse"))
|
|
387
|
+
)
|
|
388
|
+
|
|
389
|
+
cmp_rate_4yr_coarse = ipeds_sector_rates.filter(
|
|
390
|
+
pl.col("college_sector") == "4yr"
|
|
391
|
+
).item(0, "cmp_rate_coarse")
|
|
392
|
+
cmp_rate_2yr_coarse = ipeds_sector_rates.filter(
|
|
393
|
+
pl.col("college_sector") == "2yr_or_less"
|
|
394
|
+
).item(0, "cmp_rate_coarse")
|
|
395
|
+
|
|
396
|
+
for label, value in {
|
|
397
|
+
"four-year k_mean": k_mean_4yr_coarse,
|
|
398
|
+
"two-year-or-less k_mean": k_mean_2yr_coarse,
|
|
399
|
+
}.items():
|
|
400
|
+
if value is None or value <= 0:
|
|
401
|
+
raise ValueError(f"Invalid national student-weighted {label}: {value}")
|
|
402
|
+
|
|
403
|
+
for label, value in {
|
|
404
|
+
"four-year completion rate": cmp_rate_4yr_coarse,
|
|
405
|
+
"two-year-or-less completion rate": cmp_rate_2yr_coarse,
|
|
406
|
+
}.items():
|
|
407
|
+
if value is None or not 0 <= value <= 1:
|
|
408
|
+
raise ValueError(f"Invalid national student-weighted {label}: {value}")
|
|
409
|
+
|
|
296
410
|
|
|
297
411
|
###########################################################
|
|
298
412
|
# Build college reference data
|
|
@@ -331,7 +445,7 @@ college_ref = (
|
|
|
331
445
|
)
|
|
332
446
|
.with_columns(
|
|
333
447
|
(pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
|
|
334
|
-
|
|
448
|
+
pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
|
|
335
449
|
pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
|
|
336
450
|
)
|
|
337
451
|
.with_columns(
|
|
@@ -908,6 +1022,40 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
908
1022
|
)
|
|
909
1023
|
)
|
|
910
1024
|
|
|
1025
|
+
# Build calendar-year attendance indicators for each age. These use birth year
|
|
1026
|
+
# plus age, rather than the July-to-June school-year windows above.
|
|
1027
|
+
for age in AGE_ATTENDANCE_RANGE:
|
|
1028
|
+
bounded = enroll.with_columns(
|
|
1029
|
+
(pl.col("cohort_lottery") + age).alias("_age_window_year")
|
|
1030
|
+
).with_columns(
|
|
1031
|
+
pl.date(pl.col("_age_window_year"), 1, 1).alias("_age_window_start"),
|
|
1032
|
+
pl.date(pl.col("_age_window_year"), 12, 31).alias("_age_window_end"),
|
|
1033
|
+
).with_columns(
|
|
1034
|
+
(
|
|
1035
|
+
(pl.col("term_start_date") <= pl.col("_age_window_end"))
|
|
1036
|
+
& (pl.col("term_end_date") >= pl.col("_age_window_start"))
|
|
1037
|
+
& (pl.col("_term_att") == 1)
|
|
1038
|
+
)
|
|
1039
|
+
.cast(pl.Int8)
|
|
1040
|
+
.alias("_overlaps")
|
|
1041
|
+
)
|
|
1042
|
+
|
|
1043
|
+
enrollment_pieces.append(
|
|
1044
|
+
bounded.group_by("sid_cepr").agg(
|
|
1045
|
+
[
|
|
1046
|
+
(
|
|
1047
|
+
pl.col("_overlaps")
|
|
1048
|
+
* pl.col(f"college_{college_type}")
|
|
1049
|
+
.fill_null(0)
|
|
1050
|
+
.cast(pl.Int8)
|
|
1051
|
+
)
|
|
1052
|
+
.max()
|
|
1053
|
+
.alias(f"att_{college_type}_{age}")
|
|
1054
|
+
for college_type in AGE_ATTENDANCE_TYPES
|
|
1055
|
+
]
|
|
1056
|
+
)
|
|
1057
|
+
)
|
|
1058
|
+
|
|
911
1059
|
enrollment_outcomes = enrollment_pieces[0]
|
|
912
1060
|
for piece in enrollment_pieces[1:]:
|
|
913
1061
|
enrollment_outcomes = enrollment_outcomes.join(
|
|
@@ -939,7 +1087,6 @@ first_institution = (
|
|
|
939
1087
|
.first()
|
|
940
1088
|
.alias("college_name_firstinst"),
|
|
941
1089
|
pl.col("college_years").first().alias("college_years_firstinst"),
|
|
942
|
-
pl.col("k_mean").first().alias("k_mean_firstinst"),
|
|
943
1090
|
pl.col("completion_rate_150pct_ip")
|
|
944
1091
|
.first()
|
|
945
1092
|
.alias("completion_rate_150pct_firstinst"),
|
|
@@ -1370,6 +1517,14 @@ nsc_outcomes = (
|
|
|
1370
1517
|
.then(k_mean_neverattend)
|
|
1371
1518
|
.otherwise(pl.col("k_mean"))
|
|
1372
1519
|
.alias("k_mean"),
|
|
1520
|
+
pl.when(pl.col("ID_FSC_firstinst").is_null())
|
|
1521
|
+
.then(k_mean_neverattend)
|
|
1522
|
+
.when(pl.col("college_years_firstinst") == 4)
|
|
1523
|
+
.then(k_mean_4yr_coarse)
|
|
1524
|
+
.when(pl.col("college_years_firstinst").is_in([1, 2]))
|
|
1525
|
+
.then(k_mean_2yr_coarse)
|
|
1526
|
+
.otherwise(k_mean_insuffdata)
|
|
1527
|
+
.alias("k_mean_coarse"),
|
|
1373
1528
|
)
|
|
1374
1529
|
)
|
|
1375
1530
|
|
|
@@ -1466,6 +1621,27 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1466
1621
|
]
|
|
1467
1622
|
)
|
|
1468
1623
|
|
|
1624
|
+
for age in AGE_ATTENDANCE_RANGE:
|
|
1625
|
+
age_start = pl.date(pl.col("cohort_lottery") + age, 1, 1)
|
|
1626
|
+
age_columns = [
|
|
1627
|
+
column
|
|
1628
|
+
for column in nsc_outcomes.columns
|
|
1629
|
+
if column.startswith("att_") and column.endswith(f"_{age}")
|
|
1630
|
+
]
|
|
1631
|
+
|
|
1632
|
+
nsc_outcomes = nsc_outcomes.with_columns(
|
|
1633
|
+
[
|
|
1634
|
+
pl.when(age_start > NSC_CUTOFF_DATE)
|
|
1635
|
+
.then(None)
|
|
1636
|
+
.when(pl.col(column).is_null())
|
|
1637
|
+
.then(0)
|
|
1638
|
+
.otherwise(pl.col(column))
|
|
1639
|
+
.cast(pl.Int8)
|
|
1640
|
+
.alias(column)
|
|
1641
|
+
for column in age_columns
|
|
1642
|
+
]
|
|
1643
|
+
)
|
|
1644
|
+
|
|
1469
1645
|
|
|
1470
1646
|
# Sarah removes the raw school-specific rate for students who were not observed
|
|
1471
1647
|
# attending by Y2 before creating cmp_rate.
|
|
@@ -1479,10 +1655,18 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
1479
1655
|
# Sarah creates cmp_rate as a copy of the restricted raw completion rate, then
|
|
1480
1656
|
# fills missing cmp_rate values with the late-enrollee/control-group mean.
|
|
1481
1657
|
nsc_outcomes = nsc_outcomes.with_columns(
|
|
1482
|
-
pl.col("completion_rate_150pct_ip").alias("cmp_rate")
|
|
1658
|
+
pl.col("completion_rate_150pct_ip").alias("cmp_rate"),
|
|
1659
|
+
pl.when(pl.col("att_any_byY2") == 0)
|
|
1660
|
+
.then(None)
|
|
1661
|
+
.when(pl.col("college_years_firstinst") == 4)
|
|
1662
|
+
.then(cmp_rate_4yr_coarse)
|
|
1663
|
+
.when(pl.col("college_years_firstinst").is_in([1, 2]))
|
|
1664
|
+
.then(cmp_rate_2yr_coarse)
|
|
1665
|
+
.otherwise(None)
|
|
1666
|
+
.alias("cmp_rate_coarse"),
|
|
1483
1667
|
)
|
|
1484
1668
|
|
|
1485
|
-
|
|
1669
|
+
late_enrollee_means = (
|
|
1486
1670
|
nsc_outcomes
|
|
1487
1671
|
.filter(
|
|
1488
1672
|
(pl.col("att_4yr_inY2") == 0)
|
|
@@ -1494,15 +1678,29 @@ late_enrollee_mean = (
|
|
|
1494
1678
|
)
|
|
1495
1679
|
& (pl.col("offer") == 0)
|
|
1496
1680
|
)
|
|
1497
|
-
.select(
|
|
1498
|
-
|
|
1681
|
+
.select(
|
|
1682
|
+
pl.col("cmp_rate").mean().alias("cmp_rate"),
|
|
1683
|
+
pl.col("cmp_rate_coarse").mean().alias("cmp_rate_coarse"),
|
|
1684
|
+
)
|
|
1499
1685
|
)
|
|
1500
1686
|
|
|
1687
|
+
late_enrollee_mean = late_enrollee_means.item(0, "cmp_rate")
|
|
1688
|
+
late_enrollee_coarse_mean = late_enrollee_means.item(0, "cmp_rate_coarse")
|
|
1689
|
+
|
|
1690
|
+
if late_enrollee_mean is None or late_enrollee_coarse_mean is None:
|
|
1691
|
+
raise ValueError(
|
|
1692
|
+
"The late-enrollee control donor group has no usable completion-rate values."
|
|
1693
|
+
)
|
|
1694
|
+
|
|
1501
1695
|
nsc_outcomes = nsc_outcomes.with_columns(
|
|
1502
1696
|
pl.when(pl.col("cmp_rate").is_null())
|
|
1503
1697
|
.then(pl.lit(late_enrollee_mean, dtype=pl.Float64))
|
|
1504
1698
|
.otherwise(pl.col("cmp_rate"))
|
|
1505
|
-
.alias("adj_cmp_rate")
|
|
1699
|
+
.alias("adj_cmp_rate"),
|
|
1700
|
+
pl.when(pl.col("cmp_rate_coarse").is_null())
|
|
1701
|
+
.then(pl.lit(late_enrollee_coarse_mean, dtype=pl.Float64))
|
|
1702
|
+
.otherwise(pl.col("cmp_rate_coarse"))
|
|
1703
|
+
.alias("adj_cmp_rate_coarse"),
|
|
1506
1704
|
).with_columns(
|
|
1507
1705
|
pl.when(pl.col("att_4yr_byY2").is_null())
|
|
1508
1706
|
.then(None)
|
|
@@ -1512,6 +1710,14 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
1512
1710
|
.then(None)
|
|
1513
1711
|
.otherwise(pl.col("adj_cmp_rate"))
|
|
1514
1712
|
.alias("adj_cmp_rate"),
|
|
1713
|
+
pl.when(pl.col("att_4yr_byY2").is_null())
|
|
1714
|
+
.then(None)
|
|
1715
|
+
.otherwise(pl.col("cmp_rate_coarse"))
|
|
1716
|
+
.alias("cmp_rate_coarse"),
|
|
1717
|
+
pl.when(pl.col("att_4yr_byY2").is_null())
|
|
1718
|
+
.then(None)
|
|
1719
|
+
.otherwise(pl.col("adj_cmp_rate_coarse"))
|
|
1720
|
+
.alias("adj_cmp_rate_coarse"),
|
|
1515
1721
|
# Preserve the package's existing name as an alias for Sarah's adjusted rate.
|
|
1516
1722
|
pl.when(pl.col("att_4yr_byY2").is_null())
|
|
1517
1723
|
.then(None)
|
|
@@ -1535,6 +1741,10 @@ outcome_columns = [
|
|
|
1535
1741
|
"att_2yr_byY2", "att_2yr_byY3", "att_2yr_byY4",
|
|
1536
1742
|
"att_elite_byY2", "att_elite_byY3", "att_elite_byY4", "att_elite_byY5", "att_elite_byY6",
|
|
1537
1743
|
|
|
1744
|
+
*[f"att_any_{age}" for age in AGE_ATTENDANCE_RANGE],
|
|
1745
|
+
*[f"att_4yr_{age}" for age in AGE_ATTENDANCE_RANGE],
|
|
1746
|
+
*[f"att_2yr_{age}" for age in AGE_ATTENDANCE_RANGE],
|
|
1747
|
+
|
|
1538
1748
|
"cmp_any_byY4", "cmp_any_byY5", "cmp_any_byY6", "cmp_any_byY7", "cmp_any_byY8",
|
|
1539
1749
|
"cmp_AA_byY4", "cmp_AA_byY5", "cmp_AA_byY6", "cmp_AA_byY7", "cmp_AA_byY8",
|
|
1540
1750
|
"cmp_BA_byY4", "cmp_BA_byY5", "cmp_BA_byY6", "cmp_BA_byY7", "cmp_BA_byY8",
|
|
@@ -1551,9 +1761,12 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
1551
1761
|
keep_columns = [
|
|
1552
1762
|
"sid_cepr",
|
|
1553
1763
|
"k_mean",
|
|
1764
|
+
"k_mean_coarse",
|
|
1554
1765
|
"completion_rate_150pct_ip",
|
|
1555
1766
|
"cmp_rate",
|
|
1556
1767
|
"adj_cmp_rate",
|
|
1768
|
+
"cmp_rate_coarse",
|
|
1769
|
+
"adj_cmp_rate_coarse",
|
|
1557
1770
|
"predicted_completion",
|
|
1558
1771
|
"ID_FSC_firstinst",
|
|
1559
1772
|
"college_name_firstinst",
|
|
@@ -1591,6 +1804,38 @@ print(
|
|
|
1591
1804
|
f"{first_college_coverage['has_completion_rate'].sum()}/"
|
|
1592
1805
|
f"{first_college_coverage.height} with IPEDS completion rate"
|
|
1593
1806
|
)
|
|
1807
|
+
print(
|
|
1808
|
+
"National student-weighted coarse values: "
|
|
1809
|
+
f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
|
|
1810
|
+
f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
|
|
1811
|
+
f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
|
|
1812
|
+
f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}, "
|
|
1813
|
+
f"nonattender completion imputation={late_enrollee_mean:.4f}, "
|
|
1814
|
+
f"coarse nonattender imputation={late_enrollee_coarse_mean:.4f}"
|
|
1815
|
+
)
|
|
1816
|
+
|
|
1817
|
+
observable_y2 = nsc_outcomes.filter(pl.col("att_4yr_byY2").is_not_null())
|
|
1818
|
+
coarse_audit = observable_y2.select(
|
|
1819
|
+
pl.len().alias("students"),
|
|
1820
|
+
pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
|
|
1821
|
+
pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
|
|
1822
|
+
pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
|
|
1823
|
+
pl.col("adj_cmp_rate_coarse")
|
|
1824
|
+
.null_count()
|
|
1825
|
+
.alias("adj_cmp_rate_coarse_missing"),
|
|
1826
|
+
pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
|
|
1827
|
+
pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
|
|
1828
|
+
)
|
|
1829
|
+
|
|
1830
|
+
if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
|
|
1831
|
+
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
1832
|
+
if coarse_audit.item(0, "adj_cmp_rate_coarse_missing") > 0:
|
|
1833
|
+
raise ValueError(
|
|
1834
|
+
"adj_cmp_rate_coarse is unexpectedly missing for observable students."
|
|
1835
|
+
)
|
|
1836
|
+
|
|
1837
|
+
print("Coarse outcome audit:")
|
|
1838
|
+
print(coarse_audit)
|
|
1594
1839
|
print(
|
|
1595
1840
|
nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
|
|
1596
1841
|
"sid_cepr"
|
|
Binary file
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# --- Import necessary packages ---
|
|
2
|
+
import polars as pl
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
SIGNAL_DEP_VARS = ["wages25", "college20"]
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def calculate_signal_variance(results: pl.DataFrame) -> pl.DataFrame:
|
|
9
|
+
"""Calculate signal variance and its implied 90-10 spread by outcome."""
|
|
10
|
+
return (
|
|
11
|
+
results
|
|
12
|
+
.filter(pl.col("depvar").is_in(SIGNAL_DEP_VARS))
|
|
13
|
+
.drop_nulls(["sample", "depvar", "coef", "se"])
|
|
14
|
+
.group_by("depvar")
|
|
15
|
+
.agg(
|
|
16
|
+
pl.col("sample").n_unique().alias("n_cmos"),
|
|
17
|
+
pl.col("coef").mean().alias("mean_te"),
|
|
18
|
+
pl.col("coef").var().alias("raw_variance"),
|
|
19
|
+
(pl.col("se") ** 2).mean().alias("noise_variance"),
|
|
20
|
+
)
|
|
21
|
+
.with_columns(
|
|
22
|
+
(
|
|
23
|
+
pl.col("raw_variance")
|
|
24
|
+
- pl.col("noise_variance")
|
|
25
|
+
).alias("signal_variance_raw")
|
|
26
|
+
)
|
|
27
|
+
# Retain the raw estimate, but use a nonnegative value for the spread.
|
|
28
|
+
.with_columns(
|
|
29
|
+
pl.col("signal_variance_raw")
|
|
30
|
+
.clip(lower_bound=0)
|
|
31
|
+
.alias("signal_variance")
|
|
32
|
+
)
|
|
33
|
+
.with_columns(
|
|
34
|
+
pl.col("signal_variance").sqrt().alias("signal_sd")
|
|
35
|
+
)
|
|
36
|
+
# Under normality, P90 - P10 equals 2.563103 standard deviations.
|
|
37
|
+
.with_columns(
|
|
38
|
+
(2.563103 * pl.col("signal_sd")).alias("spread_90_10")
|
|
39
|
+
)
|
|
40
|
+
.sort("depvar")
|
|
41
|
+
)
|
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/aspire.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/helpers.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/mappings.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/christel_house.py
RENAMED
|
File without changes
|
{ltc_code-0.2.20 → ltc_code-0.2.22}/src/ltc_code/20260630_census_disclosure/democracy_prep.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|