ltc-code 0.2.29__tar.gz → 0.2.31__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. {ltc_code-0.2.29 → ltc_code-0.2.31}/PKG-INFO +1 -1
  2. {ltc_code-0.2.29 → ltc_code-0.2.31}/pyproject.toml +1 -1
  3. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/NEW_CROSSWALK.md +3 -1
  4. ltc_code-0.2.31/src/ltc_code/nsc/OUTCOMES.md +59 -0
  5. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/build_nsc_outcomes.py +66 -126
  6. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/build_nsc_outcomes_new.py +94 -149
  7. {ltc_code-0.2.29 → ltc_code-0.2.31}/README.md +0 -0
  8. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
  9. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
  10. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
  11. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
  12. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
  13. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
  14. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
  15. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
  16. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
  17. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
  18. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
  19. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
  20. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
  21. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
  22. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
  23. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
  24. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
  25. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
  26. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
  27. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
  28. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
  29. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
  30. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
  31. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
  32. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
  33. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
  34. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
  35. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
  36. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
  37. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
  38. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
  39. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/__init__.py +0 -0
  40. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/aspire.py +0 -0
  41. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/check_cmo_apps.do +0 -0
  42. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/christel_house.py +0 -0
  43. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/green_dot.py +0 -0
  44. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/helpers.py +0 -0
  45. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june13.py +0 -0
  46. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june2.py +0 -0
  47. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june30.py +0 -0
  48. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june5.py +0 -0
  49. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june7.py +0 -0
  50. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/kipp_nj.py +0 -0
  51. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/kipp_tx.py +0 -0
  52. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/main.py +0 -0
  53. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/make_summary_stats_table.py +0 -0
  54. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/mappings.py +0 -0
  55. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/may27.py +0 -0
  56. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/__init__.py +0 -0
  57. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
  58. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
  59. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
  60. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/naics.csv +0 -0
  61. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/naics.py +0 -0
  62. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
  63. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
  64. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
  65. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
  66. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
  67. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
  68. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
  69. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
  70. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/directory.dta +0 -0
  71. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
  72. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
  73. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/plot_bars.py +0 -0
  74. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/polars_dates.py +0 -0
  75. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/rocketship.py +0 -0
  76. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/schema_mapping.py +0 -0
  77. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
  78. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
  79. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
  80. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/signal_var_calcs.py +0 -0
  81. {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/yes_prep.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: ltc-code
3
- Version: 0.2.29
3
+ Version: 0.2.31
4
4
  Summary: Add your description here
5
5
  Requires-Dist: fastexcel>=0.16,<0.20
6
6
  Requires-Dist: polars>=1.36.1,<1.42
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ltc-code"
3
- version = "0.2.29"
3
+ version = "0.2.31"
4
4
  description = "Add your description here"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.9"
@@ -21,4 +21,6 @@ Required files in `nsc/raw/`:
21
21
 
22
22
  The old IPEDS_IC_manual.xlsx is not needed by this variant. Student inputs remain input/apps.csv, input/nsc_records_old.dta, and input/nsc_records_new.csv. Do not supply synthetic records for actual analysis. Institutional support files are bundled in the locally built wheel and source distribution but are Git-ignored. GitHub Actions publishes these verified distributions from release assets; it does not rebuild from a checkout that lacks the data. Student inputs are never bundled.
23
23
 
24
- Institution-only audit CSVs are written alongside the outcomes, including mapping choices, appended candidates, missing UNITIDs, and missing classifications. The 99.48% matched subset is an audit denominator, not a filter that drops student records from the build.
24
+ Intermediate diagnostic CSVs and printed audit tables are not generated. Only the full and selected final outcome files are written, each in Parquet and CSV format. The 99.48% matched subset is an audit denominator, not a filter that drops student records from the build.
25
+
26
+ Version 0.2.30 adds final audit exports and selectivity/BA sensitivity outcomes to both builds; see [OUTCOMES.md](OUTCOMES.md) for names and definitions.
@@ -0,0 +1,59 @@
1
+ # NSC outcomes added in 0.2.30
2
+
3
+ Both `build_nsc_outcomes.py` and `build_nsc_outcomes_new.py` provide these fields.
4
+ The new-crosswalk variant remains an explicit separate build.
5
+
6
+ ## Final institution audit fields
7
+
8
+ `nsc_outcomes_final` retains `ID_FSC_firstinst`, `college_name_firstinst`,
9
+ `unitid_firstinst`, `tier_firstinst`, `college_years_firstinst`, and
10
+ `completion_rate_150pct_firstinst`. Institution years are 4 (four-year),
11
+ 2 (two-year), 1 (less than two-year), or null if unknown.
12
+
13
+ The first institution remains the earliest retained enrollment spell starting
14
+ on or after July 1 of the year the student turns 18, with the existing tie rule.
15
+ It need not meet the attendance-status rule or fall within the by-year-four
16
+ window. Attendance indicators describe qualifying enrollment in their window;
17
+ they do not describe the first institution. A transfer can therefore have a
18
+ two-year first institution and four-year attendance. Prediction formulas and
19
+ first-institution selection have not changed.
20
+
21
+ The build writes the selected table to `nsc_outcomes_final.parquet` and `.csv`
22
+ alongside the existing full `nsc_outcomes.parquet` and `.csv` outputs. These
23
+ are the only four files written; intermediate diagnostic exports and printed
24
+ audit tables have been removed.
25
+
26
+ ## Selectivity outcomes
27
+
28
+ Selective means tiers 3–4. The existing `elite` name means highly selective,
29
+ tiers 1–2. These institution groups do not overlap. A student can attend or
30
+ earn degrees at institutions in both groups, so their student indicators
31
+ are not forced to be mutually exclusive.
32
+
33
+ | Example | Meaning |
34
+ | --- | --- |
35
+ | `att_selective_byY4` | Qualifying attendance at a tier 3–4 institution by year 4 |
36
+ | `att_elite_byY4` | Qualifying attendance at a tier 1–2 institution by year 4 |
37
+ | `cmp_selective_byY8` | AA or BA from a tier 3–4 institution by year 8 |
38
+ | `cmp_elite_byY8` | Existing AA-or-BA measure for tiers 1–2 |
39
+ | `cmp_BA_selective_byY8` | BA from a tier 3–4 institution by year 8 |
40
+ | `cmp_BA_elite_byY8` | BA from a tier 1–2 institution by year 8 |
41
+ | `cmp_BA_noimpute_byY8` | Any BA without institution-type credential imputation |
42
+ | `cmp_BA_selective_noimpute_byY8` | Same sensitivity definition for tiers 3–4 |
43
+ | `cmp_BA_elite_noimpute_byY8` | Same sensitivity definition for tiers 1–2 |
44
+
45
+ Attendance is available for `inY1`–`inY8`, `byY1`–`byY8`, their `fall` and
46
+ `spring` variants, and calendar ages 18–26 (e.g. `att_selective_20`). Completion
47
+ is available for `byY1`–`byY8`. All these selectivity and sensitivity fields are
48
+ retained in the selected final table. They follow the existing observation
49
+ windows and zero/null rules. Unknown institutional tiers do not count as
50
+ selective. Completion uses the degree-granting institution's tier, not the
51
+ first institution's tier.
52
+
53
+ The default BA measures retain the existing institution-type imputation.
54
+ The `noimpute` versions use credential-lookup/title evidence before that step:
55
+ an unknown award at a four-year institution alone does not count as a BA.
56
+ Students remain in the sample, and another identified BA can still qualify.
57
+ This does not change institutional IPEDS rates or coarse predictions.
58
+
59
+ Verification: run `uv run python tests/verify_nsc_outcomes.py` from the repository root with the local institutional support files present. This creates temporary synthetic records and does not require secured student inputs.
@@ -93,8 +93,8 @@ RANDOM_SEED = 3852804
93
93
  SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
94
94
  DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
95
95
  STEM_DEFINITIONS = ["sarah", "dhs"]
96
- COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
97
- AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
96
+ COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
97
+ AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
98
98
  AGE_ATTENDANCE_RANGE = range(18, 27)
99
99
 
100
100
  # These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
@@ -573,7 +573,9 @@ college_ref = (
573
573
  .with_columns(
574
574
  (pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
575
575
  pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
576
+ # Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
576
577
  pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
578
+ pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
577
579
  )
578
580
  .with_columns(
579
581
  (pl.col("college_4yr") + pl.col("college_2yr") > 0)
@@ -592,6 +594,7 @@ college_ref = (
592
594
  "college_2yr",
593
595
  "tier",
594
596
  "college_elite",
597
+ "college_selective",
595
598
  "k_mean",
596
599
  "completion_rate_150pct_ip",
597
600
  )
@@ -1061,6 +1064,7 @@ enroll_weeks = (
1061
1064
  "college_2yr",
1062
1065
  "tier",
1063
1066
  "college_elite",
1067
+ "college_selective",
1064
1068
  "k_mean",
1065
1069
  "completion_rate_150pct_ip",
1066
1070
  "term_start_date",
@@ -1131,6 +1135,7 @@ enroll = (
1131
1135
  pl.col("college_2yr").max(),
1132
1136
  pl.col("tier").drop_nulls().min(),
1133
1137
  pl.col("college_elite").max(),
1138
+ pl.col("college_selective").max(),
1134
1139
  pl.col("k_mean").drop_nulls().first(),
1135
1140
  pl.col("completion_rate_150pct_ip").drop_nulls().first(),
1136
1141
  pl.col("_term_att").max(),
@@ -1453,6 +1458,8 @@ degrees = (
1453
1458
  .otherwise(pl.lit(None))
1454
1459
  .alias("_degree")
1455
1460
  )
1461
+ # Preserve credential/title evidence before filling unknown awards by sector.
1462
+ .with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
1456
1463
  .with_columns(
1457
1464
  pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
1458
1465
  .then(pl.lit("BA"))
@@ -1504,6 +1511,12 @@ for year in range(1, N_YEARS_OUT + 1):
1504
1511
  (pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
1505
1512
  .max()
1506
1513
  .alias(f"cmp_BA_byY{year}"),
1514
+ (
1515
+ pl.col("_by_end")
1516
+ * (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
1517
+ )
1518
+ .max()
1519
+ .alias(f"cmp_BA_noimpute_byY{year}"),
1507
1520
  (pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
1508
1521
  .max()
1509
1522
  .alias(f"cmp_any_byY{year}"),
@@ -1514,6 +1527,28 @@ for year in range(1, N_YEARS_OUT + 1):
1514
1527
  )
1515
1528
  .max()
1516
1529
  .alias(f"cmp_elite_byY{year}"),
1530
+ (
1531
+ pl.col("_by_end")
1532
+ * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
1533
+ * pl.col("college_selective").fill_null(0).cast(pl.Int8)
1534
+ )
1535
+ .max()
1536
+ .alias(f"cmp_selective_byY{year}"),
1537
+ # BA-only outcomes support comparisons with any-BA completion.
1538
+ *[
1539
+ (
1540
+ pl.col("_by_end")
1541
+ * (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
1542
+ * pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
1543
+ )
1544
+ .max()
1545
+ .alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
1546
+ for college_type in ["selective", "elite"]
1547
+ for degree_column, suffix in [
1548
+ ("_degree", ""),
1549
+ ("_degree_no_type_fill", "_noimpute"),
1550
+ ]
1551
+ ],
1517
1552
  *[
1518
1553
  expression
1519
1554
  for definition in STEM_DEFINITIONS
@@ -1932,7 +1967,6 @@ for label, tiers in tier_buckets.items():
1932
1967
  .otherwise(pl.col("adj_cmp_rate_4yr"))
1933
1968
  .alias(f"adj_cmp_rate_4yr_coarse_{label}")
1934
1969
  )
1935
- print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
1936
1970
 
1937
1971
  # Match the available 1098-T calendar years, including the missing 2015 year.
1938
1972
  tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
@@ -2014,6 +2048,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
2014
2048
  .alias("recovered_nsc_outcome")
2015
2049
  )
2016
2050
 
2051
+ # Export every selectivity window, plus the BA sensitivity outcomes. Keep the
2052
+ # existing recovered-outcome definition above independent of this export list.
2053
+ selectivity_outcome_columns = [
2054
+ column for column in nsc_outcomes.columns
2055
+ if column.startswith((
2056
+ "att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
2057
+ "cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
2058
+ ))
2059
+ ]
2060
+
2017
2061
  keep_columns = [
2018
2062
  "sid_cepr",
2019
2063
  "k_mean",
@@ -2032,140 +2076,36 @@ keep_columns = [
2032
2076
  "adj_cmp_rate_coarsen_2yr",
2033
2077
  "ID_FSC_firstinst",
2034
2078
  "college_name_firstinst",
2079
+ "unitid_firstinst",
2080
+ "college_years_firstinst",
2035
2081
  "tier_firstinst",
2036
2082
  "completion_rate_150pct_firstinst",
2037
- ] + outcome_columns + stem_outcome_columns
2083
+ ] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
2084
+ keep_columns = list(dict.fromkeys(keep_columns))
2038
2085
 
2039
2086
  nsc_outcomes_final = nsc_outcomes.select(keep_columns)
2040
2087
 
2088
+ # Retain the existing completeness check without generating audit tables.
2089
+ if nsc_outcomes.filter(
2090
+ pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
2091
+ ).height > 0:
2092
+ raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
2093
+
2041
2094
  OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
2042
2095
  nsc_outcomes.write_parquet(OUTCOMES)
2043
2096
  nsc_outcomes.with_columns(
2044
2097
  pl.all().exclude(pl.Date).cast(pl.String, strict=False)
2045
2098
  ).write_csv(OUTCOMES.with_suffix(".csv"))
2046
2099
 
2100
+ # Save the selected analysis/audit table as well as the existing full output.
2101
+ final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
2102
+ nsc_outcomes_final.write_parquet(final_path)
2103
+ nsc_outcomes_final.with_columns(
2104
+ pl.all().exclude(pl.Date).cast(pl.String, strict=False)
2105
+ ).write_csv(final_path.with_suffix(".csv"))
2106
+
2107
+ print(f"Wrote {final_path}")
2108
+ print(f"Wrote {final_path.with_suffix('.csv')}")
2047
2109
  print(f"Wrote {OUTCOMES}")
2048
2110
  print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
2049
2111
  print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
2050
-
2051
- first_college_coverage = (
2052
- nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
2053
- .group_by("ID_FSC_firstinst")
2054
- .agg(
2055
- pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
2056
- pl.col("completion_rate_150pct_firstinst")
2057
- .is_not_null()
2058
- .any()
2059
- .alias("has_completion_rate"),
2060
- )
2061
- )
2062
- print(
2063
- "First-college coverage: "
2064
- f"{first_college_coverage['has_k_mean'].sum()}/"
2065
- f"{first_college_coverage.height} with college-specific k_mean; "
2066
- f"{first_college_coverage['has_completion_rate'].sum()}/"
2067
- f"{first_college_coverage.height} with IPEDS completion rate"
2068
- )
2069
- print(
2070
- "National student-weighted coarse values: "
2071
- f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
2072
- f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
2073
- f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
2074
- f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
2075
- "observed by-Y4 nonattender completion=0"
2076
- )
2077
- print(
2078
- "Sample-institution IPEDS-cohort-weighted coarse values: "
2079
- f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
2080
- f"({sample_4yr_institutions} institutions), "
2081
- f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
2082
- f"({sample_2yr_institutions} institutions)"
2083
- )
2084
-
2085
- observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
2086
- coarse_audit = observable_y4.select(
2087
- pl.len().alias("students"),
2088
- pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
2089
- pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
2090
- pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
2091
- pl.col("adj_cmp_rate_coarse")
2092
- .null_count()
2093
- .alias("adj_cmp_rate_coarse_missing"),
2094
- pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
2095
- pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
2096
- pl.col("adj_cmp_rate_coarse_sample")
2097
- .null_count()
2098
- .alias("adj_cmp_rate_coarse_sample_missing"),
2099
- pl.col("adj_cmp_rate_coarse_sample")
2100
- .min()
2101
- .alias("adj_cmp_rate_coarse_sample_min"),
2102
- pl.col("adj_cmp_rate_coarse_sample")
2103
- .max()
2104
- .alias("adj_cmp_rate_coarse_sample_max"),
2105
- )
2106
-
2107
- if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
2108
- raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
2109
- # Attendees without a classified first college can have missing coarse rates.
2110
- # Report those counts below rather than treating them as a build failure.
2111
- print("Coarse outcome audit:")
2112
- print(coarse_audit)
2113
-
2114
- attendee_completion_audit = (
2115
- observable_y4.filter(pl.col("att_any_byY4") == 1)
2116
- .select(
2117
- pl.len().alias("by_y4_attendees"),
2118
- pl.col("completion_rate_150pct_firstinst")
2119
- .is_null()
2120
- .sum()
2121
- .alias("missing_first_institution_rate"),
2122
- (
2123
- pl.col("completion_rate_150pct_firstinst").is_null()
2124
- & pl.col("completion_rate_150pct_ip").is_not_null()
2125
- )
2126
- .sum()
2127
- .alias("filled_by_tier_median"),
2128
- pl.col("adj_cmp_rate")
2129
- .is_null()
2130
- .sum()
2131
- .alias("missing_after_tier_median"),
2132
- pl.col("adj_cmp_rate")
2133
- .is_null()
2134
- .mean()
2135
- .alias("missing_after_tier_median_share"),
2136
- (
2137
- pl.col("adj_cmp_rate").is_null()
2138
- & pl.col("tier_firstinst").is_null()
2139
- )
2140
- .sum()
2141
- .alias("missing_after_tier_median_no_tier"),
2142
- (
2143
- pl.col("adj_cmp_rate").is_null()
2144
- & pl.col("tier_firstinst").is_not_null()
2145
- )
2146
- .sum()
2147
- .alias("missing_after_tier_median_with_tier"),
2148
- )
2149
- )
2150
- print("By-Y4 attendee completion-rate audit:")
2151
- print(attendee_completion_audit)
2152
- print(
2153
- nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
2154
- "sid_cepr"
2155
- )
2156
- )
2157
-
2158
- # Nonmissing year-eight completion identifies students with the full Y8 window.
2159
- year8_attendance_audit = (
2160
- nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
2161
- .select(
2162
- pl.len().alias("N_year8_sample"),
2163
- (pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
2164
- (
2165
- (pl.col("att_any_byY4") == 0)
2166
- & (pl.col("cmp_BA_byY8") == 1)
2167
- ).sum().alias("N_no_attendance_byY4_BA_byY8"),
2168
- )
2169
- )
2170
- print("Year-eight sample attendance and late BA completion check:")
2171
- print(year8_attendance_audit)
@@ -60,7 +60,6 @@ INPUT_NSC = PACKAGE_NSC / "input"
60
60
  # Folder for secured Census inputs; no student records are packaged.
61
61
 
62
62
  OUTPUT_NSC = PACKAGE_NSC / "output" / "new_crosswalk"
63
- OUTPUT_NSC.mkdir(parents=True, exist_ok=True)
64
63
  # Folder for final student-level NSC outcomes.
65
64
 
66
65
  APPS = INPUT_NSC / "apps.csv"
@@ -93,8 +92,8 @@ RANDOM_SEED = 3852804
93
92
  SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
94
93
  DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
95
94
  STEM_DEFINITIONS = ["sarah", "dhs"]
96
- COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
97
- AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
95
+ COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
96
+ AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
98
97
  AGE_ATTENDANCE_RANGE = range(18, 27)
99
98
 
100
99
  # These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
@@ -249,14 +248,12 @@ legacy_absent_candidates = legacy_raw.join(
249
248
  official_crosswalk.select("ID_FSC"), on="ID_FSC", how="anti"
250
249
  )
251
250
  # Reproducible legacy fallback selection; approved sole-rate exceptions below
252
- # still take precedence. Save all candidates so ambiguity is reviewable.
253
- legacy_absent_candidates.write_csv(OUTPUT_NSC / "legacy_absent_candidates.csv")
251
+ # still take precedence.
254
252
  legacy_append = legacy_absent_candidates.sort(
255
253
  ["ID_FSC", "legacy_unitid", "ID_OPE", "name"], nulls_last=True
256
254
  ).unique("ID_FSC", keep="first", maintain_order=True).select(
257
255
  "ID_FSC", "legacy_unitid", pl.col("name").alias("college_name_crosswalk")
258
256
  )
259
- legacy_append.write_csv(OUTPUT_NSC / "legacy_appended_codes.csv")
260
257
  combined_crosswalk = pl.concat([
261
258
  official_crosswalk.with_columns(pl.lit("official").alias("mapping_source")),
262
259
  legacy_append.with_columns(pl.lit("legacy_absent_code").alias("mapping_source")),
@@ -300,7 +297,6 @@ college_crosswalk = (
300
297
  pl.col("ID_FSC").str.slice(0, 6).cast(pl.Int64, strict=False).alias("opeid"),
301
298
  )
302
299
  )
303
- college_crosswalk.write_csv(OUTPUT_NSC / "mapping_decisions.csv")
304
300
 
305
301
 
306
302
  ###########################################################
@@ -616,7 +612,9 @@ college_ref = (
616
612
  .with_columns(
617
613
  (pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
618
614
  pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
615
+ # Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
619
616
  pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
617
+ pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
620
618
  )
621
619
  .with_columns(
622
620
  (pl.col("college_4yr") + pl.col("college_2yr") > 0)
@@ -635,26 +633,13 @@ college_ref = (
635
633
  "college_2yr",
636
634
  "tier",
637
635
  "college_elite",
636
+ "college_selective",
638
637
  "k_mean",
639
638
  "completion_rate_150pct_ip",
640
639
  )
641
640
  )
642
641
 
643
642
 
644
- # Institution-universe coverage: code counts, not student counts.
645
- coverage = college_crosswalk.join(
646
- college_ref.select("ID_FSC", "college_years", "completion_rate_150pct_ip"),
647
- on="ID_FSC", how="left", validate="1:1",
648
- )
649
- coverage.write_csv(OUTPUT_NSC / "institution_coverage.csv")
650
- coverage.filter(pl.col("unitid").is_null()).write_csv(OUTPUT_NSC / "codes_without_unitid.csv")
651
- coverage.filter(pl.col("unitid").is_not_null() & pl.col("college_years").is_null()).write_csv(OUTPUT_NSC / "mapped_codes_without_level.csv")
652
- print("CROSSWALK COVERAGE", coverage.select(
653
- pl.len().alias("codes"), pl.col("unitid").is_not_null().sum().alias("mapped"),
654
- pl.col("college_years").is_not_null().sum().alias("classified"),
655
- pl.col("completion_rate_150pct_ip").is_not_null().sum().alias("with_rate"),
656
- ))
657
-
658
643
  ###########################################################
659
644
  # Load student universe and NSC records
660
645
  ###########################################################
@@ -1042,6 +1027,7 @@ enroll_weeks = (
1042
1027
  "college_2yr",
1043
1028
  "tier",
1044
1029
  "college_elite",
1030
+ "college_selective",
1045
1031
  "k_mean",
1046
1032
  "completion_rate_150pct_ip",
1047
1033
  "term_start_date",
@@ -1112,6 +1098,7 @@ enroll = (
1112
1098
  pl.col("college_2yr").max(),
1113
1099
  pl.col("tier").drop_nulls().min(),
1114
1100
  pl.col("college_elite").max(),
1101
+ pl.col("college_selective").max(),
1115
1102
  pl.col("k_mean").drop_nulls().first(),
1116
1103
  pl.col("completion_rate_150pct_ip").drop_nulls().first(),
1117
1104
  pl.col("_term_att").max(),
@@ -1306,11 +1293,34 @@ for column in enrollment_outcomes.columns:
1306
1293
  .alias(column)
1307
1294
  )
1308
1295
 
1309
- # First post-high-school institution is useful for auditing the match. Sarah's
1310
- # final Stata sort keeps the highest ID_FSC when enrollment start dates tie.
1296
+ # Use exactly the status and window-overlap rules for att_any_byY4. A spell
1297
+ # starting before July 1 can count if it continues into the attendance window.
1298
+ qualifying_enroll = enroll.with_columns(
1299
+ pl.date(pl.col("cohort_18"), 7, 1).alias("_window_start"),
1300
+ pl.date(pl.col("cohort_18") + 4, 6, 30).alias("_window_end"),
1301
+ ).filter(
1302
+ (pl.col("_term_att") == 1)
1303
+ & (pl.col("term_end_date") >= pl.col("_window_start"))
1304
+ )
1305
+
1306
+ # Before excluding late starts, count students whose first observed qualifying
1307
+ # enrollment is after Y4. Denominator: observed qualifying enrollees with a
1308
+ # fully observable Y4 window; not all applicants or an eventual-enrollment rate.
1309
+ first_enrollment_timing = qualifying_enroll.group_by("sid_cepr").agg(
1310
+ pl.col("term_start_date").min().alias("first_start"),
1311
+ pl.col("_window_end").first().alias("y4_end"),
1312
+ ).filter(pl.col("y4_end") <= NSC_CUTOFF_DATE)
1313
+
1314
+ print("FIRST QUALIFYING ENROLLMENT AFTER Y4", first_enrollment_timing.select(
1315
+ pl.len().alias("observed_enrollees_with_Y4_followup"),
1316
+ (pl.col("first_start") > pl.col("y4_end")).sum().alias("first_after_Y4"),
1317
+ (100 * (pl.col("first_start") > pl.col("y4_end")).mean()).alias("percent_after_Y4"),
1318
+ ))
1319
+
1320
+ # Select the earliest qualifying spell overlapping the by-Y4 window. Preserve
1321
+ # Sarah's tie-break: highest ID_FSC when enrollment start dates are equal.
1311
1322
  first_institution = (
1312
- enroll.with_columns(pl.date(pl.col("cohort_18"), 7, 1).alias("_hs_grad_start"))
1313
- .filter(pl.col("term_start_date") >= pl.col("_hs_grad_start"))
1323
+ qualifying_enroll.filter(pl.col("term_start_date") <= pl.col("_window_end"))
1314
1324
  .sort(["sid_cepr", "term_start_date", "ID_FSC"], descending=[False, False, True])
1315
1325
  .group_by("sid_cepr", maintain_order=True)
1316
1326
  .agg(
@@ -1434,6 +1444,8 @@ degrees = (
1434
1444
  .otherwise(pl.lit(None))
1435
1445
  .alias("_degree")
1436
1446
  )
1447
+ # Preserve credential/title evidence before filling unknown awards by sector.
1448
+ .with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
1437
1449
  .with_columns(
1438
1450
  pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
1439
1451
  .then(pl.lit("BA"))
@@ -1485,6 +1497,12 @@ for year in range(1, N_YEARS_OUT + 1):
1485
1497
  (pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
1486
1498
  .max()
1487
1499
  .alias(f"cmp_BA_byY{year}"),
1500
+ (
1501
+ pl.col("_by_end")
1502
+ * (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
1503
+ )
1504
+ .max()
1505
+ .alias(f"cmp_BA_noimpute_byY{year}"),
1488
1506
  (pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
1489
1507
  .max()
1490
1508
  .alias(f"cmp_any_byY{year}"),
@@ -1495,6 +1513,28 @@ for year in range(1, N_YEARS_OUT + 1):
1495
1513
  )
1496
1514
  .max()
1497
1515
  .alias(f"cmp_elite_byY{year}"),
1516
+ (
1517
+ pl.col("_by_end")
1518
+ * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
1519
+ * pl.col("college_selective").fill_null(0).cast(pl.Int8)
1520
+ )
1521
+ .max()
1522
+ .alias(f"cmp_selective_byY{year}"),
1523
+ # BA-only outcomes support comparisons with any-BA completion.
1524
+ *[
1525
+ (
1526
+ pl.col("_by_end")
1527
+ * (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
1528
+ * pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
1529
+ )
1530
+ .max()
1531
+ .alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
1532
+ for college_type in ["selective", "elite"]
1533
+ for degree_column, suffix in [
1534
+ ("_degree", ""),
1535
+ ("_degree_no_type_fill", "_noimpute"),
1536
+ ]
1537
+ ],
1498
1538
  *[
1499
1539
  expression
1500
1540
  for definition in STEM_DEFINITIONS
@@ -1923,7 +1963,6 @@ for label, tiers in tier_buckets.items():
1923
1963
  .otherwise(pl.col("adj_cmp_rate_4yr"))
1924
1964
  .alias(f"adj_cmp_rate_4yr_coarse_{label}")
1925
1965
  )
1926
- print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
1927
1966
 
1928
1967
  # Match the available 1098-T calendar years, including the missing 2015 year.
1929
1968
  tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
@@ -2005,6 +2044,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
2005
2044
  .alias("recovered_nsc_outcome")
2006
2045
  )
2007
2046
 
2047
+ # Export every selectivity window, plus the BA sensitivity outcomes. Keep the
2048
+ # existing recovered-outcome definition above independent of this export list.
2049
+ selectivity_outcome_columns = [
2050
+ column for column in nsc_outcomes.columns
2051
+ if column.startswith((
2052
+ "att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
2053
+ "cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
2054
+ ))
2055
+ ]
2056
+
2008
2057
  keep_columns = [
2009
2058
  "sid_cepr",
2010
2059
  "k_mean",
@@ -2023,140 +2072,36 @@ keep_columns = [
2023
2072
  "adj_cmp_rate_coarsen_2yr",
2024
2073
  "ID_FSC_firstinst",
2025
2074
  "college_name_firstinst",
2075
+ "unitid_firstinst",
2076
+ "college_years_firstinst",
2026
2077
  "tier_firstinst",
2027
2078
  "completion_rate_150pct_firstinst",
2028
- ] + outcome_columns + stem_outcome_columns
2079
+ ] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
2080
+ keep_columns = list(dict.fromkeys(keep_columns))
2029
2081
 
2030
2082
  nsc_outcomes_final = nsc_outcomes.select(keep_columns)
2031
2083
 
2084
+ # Retain the existing completeness check without generating audit tables.
2085
+ if nsc_outcomes.filter(
2086
+ pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
2087
+ ).height > 0:
2088
+ raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
2089
+
2032
2090
  OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
2033
2091
  nsc_outcomes.write_parquet(OUTCOMES)
2034
2092
  nsc_outcomes.with_columns(
2035
2093
  pl.all().exclude(pl.Date).cast(pl.String, strict=False)
2036
2094
  ).write_csv(OUTCOMES.with_suffix(".csv"))
2037
2095
 
2096
+ # Save the selected analysis/audit table as well as the existing full output.
2097
+ final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
2098
+ nsc_outcomes_final.write_parquet(final_path)
2099
+ nsc_outcomes_final.with_columns(
2100
+ pl.all().exclude(pl.Date).cast(pl.String, strict=False)
2101
+ ).write_csv(final_path.with_suffix(".csv"))
2102
+
2103
+ print(f"Wrote {final_path}")
2104
+ print(f"Wrote {final_path.with_suffix('.csv')}")
2038
2105
  print(f"Wrote {OUTCOMES}")
2039
2106
  print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
2040
2107
  print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
2041
-
2042
- first_college_coverage = (
2043
- nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
2044
- .group_by("ID_FSC_firstinst")
2045
- .agg(
2046
- pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
2047
- pl.col("completion_rate_150pct_firstinst")
2048
- .is_not_null()
2049
- .any()
2050
- .alias("has_completion_rate"),
2051
- )
2052
- )
2053
- print(
2054
- "First-college coverage: "
2055
- f"{first_college_coverage['has_k_mean'].sum()}/"
2056
- f"{first_college_coverage.height} with college-specific k_mean; "
2057
- f"{first_college_coverage['has_completion_rate'].sum()}/"
2058
- f"{first_college_coverage.height} with IPEDS completion rate"
2059
- )
2060
- print(
2061
- "National student-weighted coarse values: "
2062
- f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
2063
- f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
2064
- f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
2065
- f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
2066
- "observed by-Y4 nonattender completion=0"
2067
- )
2068
- print(
2069
- "Sample-institution IPEDS-cohort-weighted coarse values: "
2070
- f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
2071
- f"({sample_4yr_institutions} institutions), "
2072
- f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
2073
- f"({sample_2yr_institutions} institutions)"
2074
- )
2075
-
2076
- observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
2077
- coarse_audit = observable_y4.select(
2078
- pl.len().alias("students"),
2079
- pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
2080
- pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
2081
- pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
2082
- pl.col("adj_cmp_rate_coarse")
2083
- .null_count()
2084
- .alias("adj_cmp_rate_coarse_missing"),
2085
- pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
2086
- pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
2087
- pl.col("adj_cmp_rate_coarse_sample")
2088
- .null_count()
2089
- .alias("adj_cmp_rate_coarse_sample_missing"),
2090
- pl.col("adj_cmp_rate_coarse_sample")
2091
- .min()
2092
- .alias("adj_cmp_rate_coarse_sample_min"),
2093
- pl.col("adj_cmp_rate_coarse_sample")
2094
- .max()
2095
- .alias("adj_cmp_rate_coarse_sample_max"),
2096
- )
2097
-
2098
- if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
2099
- raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
2100
- # Attendees without a classified first college can have missing coarse rates.
2101
- # Report those counts below rather than treating them as a build failure.
2102
- print("Coarse outcome audit:")
2103
- print(coarse_audit)
2104
-
2105
- attendee_completion_audit = (
2106
- observable_y4.filter(pl.col("att_any_byY4") == 1)
2107
- .select(
2108
- pl.len().alias("by_y4_attendees"),
2109
- pl.col("completion_rate_150pct_firstinst")
2110
- .is_null()
2111
- .sum()
2112
- .alias("missing_first_institution_rate"),
2113
- (
2114
- pl.col("completion_rate_150pct_firstinst").is_null()
2115
- & pl.col("completion_rate_150pct_ip").is_not_null()
2116
- )
2117
- .sum()
2118
- .alias("filled_by_tier_median"),
2119
- pl.col("adj_cmp_rate")
2120
- .is_null()
2121
- .sum()
2122
- .alias("missing_after_tier_median"),
2123
- pl.col("adj_cmp_rate")
2124
- .is_null()
2125
- .mean()
2126
- .alias("missing_after_tier_median_share"),
2127
- (
2128
- pl.col("adj_cmp_rate").is_null()
2129
- & pl.col("tier_firstinst").is_null()
2130
- )
2131
- .sum()
2132
- .alias("missing_after_tier_median_no_tier"),
2133
- (
2134
- pl.col("adj_cmp_rate").is_null()
2135
- & pl.col("tier_firstinst").is_not_null()
2136
- )
2137
- .sum()
2138
- .alias("missing_after_tier_median_with_tier"),
2139
- )
2140
- )
2141
- print("By-Y4 attendee completion-rate audit:")
2142
- print(attendee_completion_audit)
2143
- print(
2144
- nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
2145
- "sid_cepr"
2146
- )
2147
- )
2148
-
2149
- # Nonmissing year-eight completion identifies students with the full Y8 window.
2150
- year8_attendance_audit = (
2151
- nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
2152
- .select(
2153
- pl.len().alias("N_year8_sample"),
2154
- (pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
2155
- (
2156
- (pl.col("att_any_byY4") == 0)
2157
- & (pl.col("cmp_BA_byY8") == 1)
2158
- ).sum().alias("N_no_attendance_byY4_BA_byY8"),
2159
- )
2160
- )
2161
- print("Year-eight sample attendance and late BA completion check:")
2162
- print(year8_attendance_audit)
File without changes