ltc-code 0.2.13__tar.gz → 0.2.14__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. {ltc_code-0.2.13 → ltc_code-0.2.14}/PKG-INFO +2 -2
  2. {ltc_code-0.2.13 → ltc_code-0.2.14}/pyproject.toml +2 -2
  3. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/.DS_Store +0 -0
  4. ltc_code-0.2.14/src/ltc_code/nsc/.DS_Store +0 -0
  5. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/build_nsc_outcomes.py +195 -68
  6. ltc_code-0.2.14/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
  7. {ltc_code-0.2.13 → ltc_code-0.2.14}/README.md +0 -0
  8. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
  9. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
  10. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
  11. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
  12. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
  13. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
  14. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
  15. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
  16. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
  17. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
  18. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
  19. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
  20. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
  21. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
  22. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
  23. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
  24. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
  25. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
  26. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
  27. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
  28. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
  29. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
  30. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
  31. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
  32. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
  33. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
  34. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
  35. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
  36. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
  37. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
  38. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
  39. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/__init__.py +0 -0
  40. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/aspire.py +0 -0
  41. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/check_cmo_apps.do +0 -0
  42. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/christel_house.py +0 -0
  43. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/green_dot.py +0 -0
  44. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/helpers.py +0 -0
  45. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/june13.py +0 -0
  46. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/june2.py +0 -0
  47. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/june30.py +0 -0
  48. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/june5.py +0 -0
  49. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/june7.py +0 -0
  50. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/kipp_nj.py +0 -0
  51. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/kipp_tx.py +0 -0
  52. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/main.py +0 -0
  53. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/mappings.py +0 -0
  54. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/may27.py +0 -0
  55. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/__init__.py +0 -0
  56. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/input/.gitkeep +0 -0
  57. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/naics.csv +0 -0
  58. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/naics.py +0 -0
  59. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
  60. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/output/.gitkeep +0 -0
  61. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
  62. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
  63. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
  64. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
  65. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/raw/directory.dta +0 -0
  66. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
  67. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/polars_dates.py +0 -0
  68. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/rocketship.py +0 -0
  69. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/schema_mapping.py +0 -0
  70. {ltc_code-0.2.13 → ltc_code-0.2.14}/src/ltc_code/yes_prep.py +0 -0
@@ -1,10 +1,10 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: ltc-code
3
- Version: 0.2.13
3
+ Version: 0.2.14
4
4
  Summary: Add your description here
5
5
  Requires-Dist: fastexcel>=0.20.2
6
6
  Requires-Dist: polars>=1.42.1
7
7
  Requires-Dist: polars-readstat>=0.20.2
8
- Requires-Python: >=3.10
8
+ Requires-Python: >=3.9
9
9
  Description-Content-Type: text/markdown
10
10
 
@@ -1,9 +1,9 @@
1
1
  [project]
2
2
  name = "ltc-code"
3
- version = "0.2.13"
3
+ version = "0.2.14"
4
4
  description = "Add your description here"
5
5
  readme = "README.md"
6
- requires-python = ">=3.10"
6
+ requires-python = ">=3.9"
7
7
  dependencies = [
8
8
  "fastexcel>=0.20.2",
9
9
  "polars>=1.42.1",
@@ -1,5 +1,6 @@
1
1
  # --- Import necessary packages ---
2
2
 
3
+ from datetime import date
3
4
  from pathlib import Path
4
5
 
5
6
  import polars as pl
@@ -31,6 +32,17 @@ DIRECTORY = RAW_NSC / "directory.dta"
31
32
  CREDENTIAL_LOOKUP = RAW_NSC / "CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx"
32
33
  # NSC credential-title lookup used before regex-based degree classification.
33
34
 
35
+ CHETTY_MRC_TABLE2_CANDIDATES = [
36
+ RAW_NSC / "chetty" / "mrc_table2.dta",
37
+ RAW_NSC / "mrc_table2.dta",
38
+ PACKAGE_NSC.parents[3] / "College" / "chetty" / "mrc_table2.dta",
39
+ ]
40
+ CHETTY_MRC_TABLE2 = next(
41
+ (path for path in CHETTY_MRC_TABLE2_CANDIDATES if path.exists()),
42
+ CHETTY_MRC_TABLE2_CANDIDATES[0],
43
+ )
44
+ # Chetty/Barron college tier lookup. Numeric tier 1/2 identifies elite colleges.
45
+
34
46
  INPUT_NSC = PACKAGE_NSC / "input"
35
47
  # Folder for secured Census inputs that are not packaged with the code.
36
48
 
@@ -40,6 +52,12 @@ APPS = INPUT_NSC / "apps.csv"
40
52
  NSC_RECORDS = INPUT_NSC / "nsc_records.csv"
41
53
  # Long NSC enrollment and degree records.
42
54
 
55
+ INPUT_NSC_OLD = INPUT_NSC / "nsc_records_old.dta"
56
+ # Older NSC pull, kept so the build can append old and new NSC versions.
57
+
58
+ INPUT_NSC_NEW = INPUT_NSC / "nsc_records_new.dta"
59
+ # Newer NSC pull with studyid-based rows.
60
+
43
61
  OUTPUT_NSC = PACKAGE_NSC / "output"
44
62
  # Package-local output folder for NSC outcome files.
45
63
 
@@ -53,9 +71,10 @@ OUTCOMES = OUTPUT_NSC / "nsc_outcomes.parquet"
53
71
 
54
72
  N_YEARS_OUT = 8
55
73
  NSC_OBSERVATION_YEAR = 2026
74
+ NSC_CUTOFF_DATE = date(2026, 12, 31)
56
75
  RANDOM_SEED = 3852804
57
76
  STEM_CIP_FAMILIES = {11, 14, 15, 26, 27, 40, 41}
58
- COLLEGE_TYPES = ["any", "4yr", "2yr"]
77
+ COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
59
78
 
60
79
 
61
80
  ###########################################################
@@ -191,6 +210,26 @@ directory = (
191
210
  .select("unitid", "college_name_directory", "years_directory", "ownership_directory")
192
211
  )
193
212
 
213
+ # Chetty/Barron tiers are keyed at the super-OPE level. The NSC college
214
+ # crosswalk carries OPE IDs, so joining tier onto college_ref makes the tier
215
+ # available on each enrollment spell.
216
+ chetty_tiers = (
217
+ scan_readstat(
218
+ CHETTY_MRC_TABLE2,
219
+ missing_string_as_null=True,
220
+ value_labels_as_strings=False,
221
+ )
222
+ .select("super_opeid", "tier")
223
+ .collect()
224
+ .with_columns(
225
+ pl.col("super_opeid").cast(pl.Int64, strict=False).alias("ID_OPE"),
226
+ pl.col("tier").cast(pl.Int8, strict=False),
227
+ )
228
+ .select("ID_OPE", "tier")
229
+ .drop_nulls("ID_OPE")
230
+ .unique("ID_OPE", keep="first")
231
+ )
232
+
194
233
 
195
234
  ###########################################################
196
235
  # Build college reference data
@@ -201,6 +240,7 @@ college_ref = (
201
240
  college_crosswalk.join(ipeds_2013, on="unitid", how="left")
202
241
  .join(ipeds_manual, on="unitid", how="left")
203
242
  .join(directory, on="unitid", how="left")
243
+ .join(chetty_tiers, on="ID_OPE", how="left", validate="m:1")
204
244
  .with_columns(
205
245
  pl.coalesce(
206
246
  [
@@ -228,6 +268,7 @@ college_ref = (
228
268
  .with_columns(
229
269
  (pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
230
270
  (pl.col("college_years") == 2).cast(pl.Int8).alias("college_2yr"),
271
+ pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
231
272
  )
232
273
  .with_columns(
233
274
  (pl.col("college_4yr") + pl.col("college_2yr") > 0)
@@ -244,6 +285,8 @@ college_ref = (
244
285
  "college_any",
245
286
  "college_4yr",
246
287
  "college_2yr",
288
+ "tier",
289
+ "college_elite",
247
290
  )
248
291
  )
249
292
 
@@ -354,22 +397,41 @@ fsc_branch_fixes = [
354
397
 
355
398
  # The real NSC file uses raw NSC names; Sarah renames these before building
356
399
  # outcomes. This block makes those same names explicit in Polars.
357
- if str(NSC_RECORDS).endswith(".dta"):
358
- nsc = scan_readstat(
359
- NSC_RECORDS, missing_string_as_null=True, value_labels_as_strings=True
360
- ).collect()
361
- else:
362
- nsc = pl.read_csv(NSC_RECORDS, infer_schema_length=0)
400
+ nsc_old = (
401
+ scan_readstat(INPUT_NSC_OLD, missing_string_as_null=True, value_labels_as_strings=True)
402
+ .with_columns(
403
+ pl.lit(2023).alias("NSCdatayear"),
363
404
 
364
- if "sid_cepr" not in nsc.columns and "studyid" in nsc.columns:
365
- nsc = (
366
- nsc.with_row_index("origsrt")
367
- .with_columns(
368
- pl.col("studyid").cast(pl.String).str.split("_").alias("sid_cepr")
369
- )
370
- .explode("sid_cepr")
371
- .filter(pl.col("sid_cepr") != "")
405
+ pl.col("searchdate")
406
+ .str.slice(0, 4)
407
+ .cast(pl.Int32)
408
+ .alias("searchbeginyear")
372
409
  )
410
+ .collect()
411
+ )
412
+
413
+ nsc_new = (
414
+ scan_readstat(INPUT_NSC_NEW, missing_string_as_null=True, value_labels_as_strings=True)
415
+ .with_row_index("origsrt", offset=1)
416
+ .filter(pl.col("studyid").is_not_null())
417
+ .with_columns(
418
+ pl.col("studyid").cast(pl.String).str.replace_all("_", "").alias("sid_cepr")
419
+ )
420
+ .explode("sid_cepr")
421
+ .filter(pl.col("sid_cepr") != "")
422
+ .with_columns(
423
+ pl.col("sid_cepr").cast(pl.String).str.replace_all("_", "").cast(pl.Int64, strict=False)
424
+ )
425
+ .filter(
426
+ pl.col("sid_cepr").is_not_null()
427
+ )
428
+ .drop("studyid", "studentageatstartofterm")
429
+ .with_columns(
430
+ pl.lit(2026).alias("NSCdatayear"),
431
+ )
432
+ .collect()
433
+ )
434
+
373
435
 
374
436
  nsc_renames = {
375
437
  "collegecodebranch": "ID_FSC",
@@ -380,28 +442,28 @@ nsc_renames = {
380
442
  "enrollmentend": "term_end_date",
381
443
  "graduationdate": "graduated_date",
382
444
  "degreetitle": "degree_title",
383
- "searchsubmit": "NSCdatayear",
384
445
  }
385
446
 
386
- nsc = nsc.rename(
447
+ nsc_old = nsc_old.rename(
387
448
  {
388
449
  source: target
389
450
  for source, target in nsc_renames.items()
390
- if source in nsc.columns and target not in nsc.columns
451
+ if source in nsc_old.columns and target not in nsc_old.columns
391
452
  }
392
453
  )
393
454
 
394
- for column in ["degreecip1", "degreecip2", "degreecip3", "degreecip4"]:
395
- if column not in nsc.columns:
396
- nsc = nsc.with_columns(pl.lit(None).cast(pl.String).alias(column))
397
-
398
- for column in ["degreemajor1", "degreemajor2", "degreemajor3", "degreemajor4"]:
399
- if column not in nsc.columns:
400
- nsc = nsc.with_columns(pl.lit(None).cast(pl.String).alias(column))
455
+ nsc_new = nsc_new.rename(
456
+ {
457
+ source: target
458
+ for source, target in nsc_renames.items()
459
+ if source in nsc_new.columns and target not in nsc_new.columns
460
+ }
461
+ )
401
462
 
402
- for column in ["NSCdatayear", "searchbeginyear", "collegesequence"]:
403
- if column not in nsc.columns:
404
- nsc = nsc.with_columns(pl.lit(None).cast(pl.Int64).alias(column))
463
+ # Append
464
+ nsc = (
465
+ pl.concat([nsc_old, nsc_new], how="diagonal_relaxed")
466
+ )
405
467
 
406
468
  required_nsc_columns = [
407
469
  "sid_cepr",
@@ -493,10 +555,19 @@ enrollment_status = pl.col("enrollment").str.to_uppercase().fill_null("")
493
555
  enroll = (
494
556
  nsc.filter(pl.col("graduated") == "N")
495
557
  .filter(pl.col("term_start_date").is_not_null() & pl.col("term_end_date").is_not_null())
558
+ .filter(pl.col("term_start_date") <= NSC_CUTOFF_DATE)
559
+ .with_columns(
560
+ pl.min_horizontal(
561
+ pl.col("term_end_date"),
562
+ pl.lit(NSC_CUTOFF_DATE),
563
+ ).alias("term_end_date")
564
+ )
496
565
  .join(apps, on="sid_cepr", how="inner")
497
- .join(college_ref, on="ID_FSC", how="left")
566
+ .join(college_ref, on="ID_FSC", how="left", validate="m:1")
498
567
  .with_columns(
499
- pl.when(enrollment_status.str.contains("FULL|\\bF\\b", literal=False))
568
+ pl.when(enrollment_status == "") # unknown
569
+ .then(9)
570
+ .when(enrollment_status.str.contains("FULL|\\bF\\b", literal=False))
500
571
  .then(5)
501
572
  .when(enrollment_status.str.contains("PART", literal=False))
502
573
  .then(4)
@@ -514,15 +585,7 @@ enroll = (
514
585
  ((pl.col("term_start_date").cast(pl.Int64) // 7).cast(pl.Int64)).alias(
515
586
  "week_start"
516
587
  ),
517
- ((pl.col("term_end_date").cast(pl.Int64) // 7).cast(pl.Int64)).alias("week_end"),
518
- pl.coalesce([pl.col("NSCdatayear"), pl.lit(-1)]).alias("_NSCdatayear_sort"),
519
- pl.coalesce([pl.col("searchbeginyear"), pl.lit(-1)]).alias(
520
- "_searchbeginyear_sort"
521
- ),
522
- pl.coalesce([pl.col("collegesequence"), pl.lit(999999)]).alias(
523
- "_collegesequence_sort"
524
- ),
525
- pl.coalesce([pl.col("unitid"), pl.lit(999999999)]).alias("_unitid_sort"),
588
+ ((pl.col("term_end_date").cast(pl.Int64) // 7).cast(pl.Int64)).alias("week_end"),
526
589
  )
527
590
  .filter(pl.col("week_end") >= pl.col("week_start"))
528
591
  .unique(keep="first")
@@ -532,12 +595,13 @@ enroll = (
532
595
  "college",
533
596
  "term_start_date",
534
597
  "term_end_date",
535
- "_NSCdatayear_sort",
536
- "_searchbeginyear_sort",
537
- "_collegesequence_sort",
598
+ "NSCdatayear",
599
+ "searchbeginyear",
600
+ "collegesequence",
538
601
  "cohort_18",
539
602
  "_enrollment_score",
540
- ]
603
+ ],
604
+ nulls_last=True
541
605
  )
542
606
  .unique(
543
607
  [
@@ -545,9 +609,9 @@ enroll = (
545
609
  "college",
546
610
  "term_start_date",
547
611
  "term_end_date",
548
- "_NSCdatayear_sort",
549
- "_searchbeginyear_sort",
550
- "_collegesequence_sort",
612
+ "NSCdatayear",
613
+ "searchbeginyear",
614
+ "collegesequence",
551
615
  "cohort_18",
552
616
  ],
553
617
  keep="first",
@@ -558,9 +622,9 @@ enroll = (
558
622
  "college",
559
623
  "term_start_date",
560
624
  "term_end_date",
561
- "_NSCdatayear_sort",
562
- "_searchbeginyear_sort",
563
- "_collegesequence_sort",
625
+ "NSCdatayear",
626
+ "searchbeginyear",
627
+ "collegesequence",
564
628
  ],
565
629
  keep="first",
566
630
  )
@@ -570,16 +634,16 @@ enroll = (
570
634
  "college",
571
635
  "term_start_date",
572
636
  "term_end_date",
573
- "_NSCdatayear_sort",
574
- "_searchbeginyear_sort",
637
+ "NSCdatayear",
638
+ "searchbeginyear",
575
639
  ],
576
640
  keep="first",
577
641
  )
578
642
  .unique(
579
- ["sid_cepr", "college", "term_start_date", "term_end_date", "_NSCdatayear_sort"],
643
+ ["sid_cepr", "college", "term_start_date", "term_end_date", "NSCdatayear"],
580
644
  keep="first",
581
645
  )
582
- .sort(["sid_cepr", "college", "term_start_date", "term_end_date", "_NSCdatayear_sort"])
646
+ .sort(["sid_cepr", "college", "term_start_date", "term_end_date", "NSCdatayear"])
583
647
  .unique(["sid_cepr", "college", "term_start_date", "term_end_date"], keep="last")
584
648
  )
585
649
 
@@ -597,16 +661,18 @@ enroll_weeks = (
597
661
  "college_any",
598
662
  "college_4yr",
599
663
  "college_2yr",
664
+ "tier",
665
+ "college_elite",
600
666
  "term_start_date",
601
667
  "term_end_date",
602
668
  "week_start",
603
669
  "week_end",
604
670
  "_term_att",
605
671
  "_enrollment_score",
606
- "_NSCdatayear_sort",
607
- "_searchbeginyear_sort",
608
- "_collegesequence_sort",
609
- "_unitid_sort",
672
+ "NSCdatayear",
673
+ "searchbeginyear",
674
+ "collegesequence",
675
+ "unitid",
610
676
  )
611
677
  .with_columns(
612
678
  pl.int_ranges("week_start", pl.col("week_end") + 1).alias("week")
@@ -617,13 +683,14 @@ enroll_weeks = (
617
683
  "sid_cepr",
618
684
  "week",
619
685
  "_enrollment_score",
620
- "_NSCdatayear_sort",
686
+ "NSCdatayear",
621
687
  "college_4yr",
622
- "_collegesequence_sort",
688
+ "collegesequence",
623
689
  "term_start_date",
624
- "_unitid_sort",
690
+ "unitid",
625
691
  ],
626
692
  descending=[False, False, True, True, True, False, False, False],
693
+ nulls_last=True
627
694
  )
628
695
  .unique(["sid_cepr", "week"], keep="first")
629
696
  .sort(["sid_cepr", "ID_FSC", "week"])
@@ -651,9 +718,12 @@ enroll = (
651
718
  pl.col("college_any").max(),
652
719
  pl.col("college_4yr").max(),
653
720
  pl.col("college_2yr").max(),
721
+ pl.col("tier").drop_nulls().min(),
722
+ pl.col("college_elite").max(),
654
723
  pl.col("_term_att").max(),
655
724
  pl.col("week").min().alias("week_start"),
656
725
  pl.col("week").max().alias("week_end"),
726
+ pl.col("_enrollment_score").max(),
657
727
  )
658
728
  .with_columns(
659
729
  (pl.col("week_start") * 7).cast(pl.Date).alias("term_start_date"),
@@ -1141,21 +1211,60 @@ nsc_outcomes = (
1141
1211
  )
1142
1212
 
1143
1213
  for year in range(1, N_YEARS_OUT + 1):
1144
- observable = pl.col("cohort_18") <= NSC_OBSERVATION_YEAR - year
1145
- observed_columns = [
1214
+ in_start = pl.date(pl.col("cohort_18") + year - 1, 7, 1)
1215
+ by_start = pl.date(pl.col("cohort_18"), 7, 1)
1216
+ by_end = pl.date(pl.col("cohort_18") + year, 6, 30)
1217
+
1218
+ att_in_columns = [
1146
1219
  column
1147
1220
  for column in nsc_outcomes.columns
1148
- if f"_byY{year}" in column or f"_inY{year}" in column
1221
+ if column.startswith("att_") and f"_inY{year}" in column
1222
+ ]
1223
+
1224
+ att_by_columns = [
1225
+ column
1226
+ for column in nsc_outcomes.columns
1227
+ if column.startswith("att_") and f"_byY{year}" in column
1228
+ ]
1229
+
1230
+ cmp_by_columns = [
1231
+ column
1232
+ for column in nsc_outcomes.columns
1233
+ if column.startswith("cmp_") and f"_byY{year}" in column
1149
1234
  ]
1150
1235
 
1151
1236
  nsc_outcomes = nsc_outcomes.with_columns(
1152
1237
  [
1153
- pl.when(pl.col(column).is_null() & observable)
1238
+ pl.when(in_start > NSC_CUTOFF_DATE)
1239
+ .then(None)
1240
+ .when(pl.col(column).is_null())
1241
+ .then(0)
1242
+ .otherwise(pl.col(column))
1243
+ .cast(pl.Int8)
1244
+ .alias(column)
1245
+ for column in att_in_columns
1246
+ ]
1247
+ +
1248
+ [
1249
+ pl.when(by_start > NSC_CUTOFF_DATE)
1250
+ .then(None)
1251
+ .when(pl.col(column).is_null())
1252
+ .then(0)
1253
+ .otherwise(pl.col(column))
1254
+ .cast(pl.Int8)
1255
+ .alias(column)
1256
+ for column in att_by_columns
1257
+ ]
1258
+ +
1259
+ [
1260
+ pl.when(by_end > NSC_CUTOFF_DATE)
1261
+ .then(None)
1262
+ .when(pl.col(column).is_null())
1154
1263
  .then(0)
1155
1264
  .otherwise(pl.col(column))
1156
1265
  .cast(pl.Int8)
1157
1266
  .alias(column)
1158
- for column in observed_columns
1267
+ for column in cmp_by_columns
1159
1268
  ]
1160
1269
  )
1161
1270
 
@@ -1165,9 +1274,21 @@ for year in range(1, N_YEARS_OUT + 1):
1165
1274
  ###########################################################
1166
1275
 
1167
1276
  outcome_columns = [
1168
- column
1169
- for column in nsc_outcomes.columns
1170
- if column.startswith("att_") or column.startswith("cmp_")
1277
+ "att_any_inY1", "att_any_inY2", "att_any_inY3", "att_any_inY4",
1278
+ "att_4yr_inY1", "att_4yr_inY2", "att_4yr_inY3", "att_4yr_inY4",
1279
+ "att_2yr_inY1", "att_2yr_inY2", "att_2yr_inY3", "att_2yr_inY4",
1280
+ "att_elite_inY1", "att_elite_inY2", "att_elite_inY3", "att_elite_inY4",
1281
+
1282
+ "att_any_byY2", "att_any_byY3", "att_any_byY4",
1283
+ "att_4yr_byY2", "att_4yr_byY3", "att_4yr_byY4",
1284
+ "att_2yr_byY2", "att_2yr_byY3", "att_2yr_byY4",
1285
+ "att_elite_byY2", "att_elite_byY3", "att_elite_byY4",
1286
+
1287
+ "cmp_any_byY4", "cmp_any_byY5", "cmp_any_byY6", "cmp_any_byY7", "cmp_any_byY8",
1288
+ "cmp_AA_byY4", "cmp_AA_byY5", "cmp_AA_byY6", "cmp_AA_byY7", "cmp_AA_byY8",
1289
+ "cmp_BA_byY4", "cmp_BA_byY5", "cmp_BA_byY6", "cmp_BA_byY7", "cmp_BA_byY8",
1290
+
1291
+
1171
1292
  ]
1172
1293
 
1173
1294
  nsc_outcomes = nsc_outcomes.with_columns(
@@ -1176,6 +1297,12 @@ nsc_outcomes = nsc_outcomes.with_columns(
1176
1297
  .alias("recovered_nsc_outcome")
1177
1298
  )
1178
1299
 
1300
+ keep_columns = [
1301
+ "sid_cepr",
1302
+ ] + outcome_columns
1303
+
1304
+ nsc_outcomes_final = nsc_outcomes.select(keep_columns)
1305
+
1179
1306
  OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
1180
1307
  nsc_outcomes.write_parquet(OUTCOMES)
1181
1308
  nsc_outcomes.with_columns(
File without changes