ltc-code 0.2.29__tar.gz → 0.2.30__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ltc_code-0.2.29 → ltc_code-0.2.30}/PKG-INFO +1 -1
- {ltc_code-0.2.29 → ltc_code-0.2.30}/pyproject.toml +1 -1
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/NEW_CROSSWALK.md +3 -1
- ltc_code-0.2.30/src/ltc_code/nsc/OUTCOMES.md +59 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/build_nsc_outcomes.py +66 -126
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/build_nsc_outcomes_new.py +67 -145
- {ltc_code-0.2.29 → ltc_code-0.2.30}/README.md +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/__init__.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/aspire.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/check_cmo_apps.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/christel_house.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/green_dot.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/helpers.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/june13.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/june2.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/june30.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/june5.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/june7.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/kipp_nj.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/kipp_tx.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/make_summary_stats_table.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/mappings.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/may27.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/__init__.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/naics.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/naics.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/directory.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/plot_bars.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/polars_dates.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/rocketship.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/schema_mapping.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/signal_var_calcs.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/yes_prep.py +0 -0
|
@@ -21,4 +21,6 @@ Required files in `nsc/raw/`:
|
|
|
21
21
|
|
|
22
22
|
The old IPEDS_IC_manual.xlsx is not needed by this variant. Student inputs remain input/apps.csv, input/nsc_records_old.dta, and input/nsc_records_new.csv. Do not supply synthetic records for actual analysis. Institutional support files are bundled in the locally built wheel and source distribution but are Git-ignored. GitHub Actions publishes these verified distributions from release assets; it does not rebuild from a checkout that lacks the data. Student inputs are never bundled.
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
Intermediate diagnostic CSVs and printed audit tables are not generated. Only the full and selected final outcome files are written, each in Parquet and CSV format. The 99.48% matched subset is an audit denominator, not a filter that drops student records from the build.
|
|
25
|
+
|
|
26
|
+
Version 0.2.30 adds final audit exports and selectivity/BA sensitivity outcomes to both builds; see [OUTCOMES.md](OUTCOMES.md) for names and definitions.
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# NSC outcomes added in 0.2.30
|
|
2
|
+
|
|
3
|
+
Both `build_nsc_outcomes.py` and `build_nsc_outcomes_new.py` provide these fields.
|
|
4
|
+
The new-crosswalk variant remains an explicit separate build.
|
|
5
|
+
|
|
6
|
+
## Final institution audit fields
|
|
7
|
+
|
|
8
|
+
`nsc_outcomes_final` retains `ID_FSC_firstinst`, `college_name_firstinst`,
|
|
9
|
+
`unitid_firstinst`, `tier_firstinst`, `college_years_firstinst`, and
|
|
10
|
+
`completion_rate_150pct_firstinst`. Institution years are 4 (four-year),
|
|
11
|
+
2 (two-year), 1 (less than two-year), or null if unknown.
|
|
12
|
+
|
|
13
|
+
The first institution remains the earliest retained enrollment spell starting
|
|
14
|
+
on or after July 1 of the year the student turns 18, with the existing tie rule.
|
|
15
|
+
It need not meet the attendance-status rule or fall within the by-year-four
|
|
16
|
+
window. Attendance indicators describe qualifying enrollment in their window;
|
|
17
|
+
they do not describe the first institution. A transfer can therefore have a
|
|
18
|
+
two-year first institution and four-year attendance. Prediction formulas and
|
|
19
|
+
first-institution selection have not changed.
|
|
20
|
+
|
|
21
|
+
The build writes the selected table to `nsc_outcomes_final.parquet` and `.csv`
|
|
22
|
+
alongside the existing full `nsc_outcomes.parquet` and `.csv` outputs. These
|
|
23
|
+
are the only four files written; intermediate diagnostic exports and printed
|
|
24
|
+
audit tables have been removed.
|
|
25
|
+
|
|
26
|
+
## Selectivity outcomes
|
|
27
|
+
|
|
28
|
+
Selective means tiers 3–4. The existing `elite` name means highly selective,
|
|
29
|
+
tiers 1–2. These institution groups do not overlap. A student can attend or
|
|
30
|
+
earn degrees at institutions in both groups, so their student indicators
|
|
31
|
+
are not forced to be mutually exclusive.
|
|
32
|
+
|
|
33
|
+
| Example | Meaning |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| `att_selective_byY4` | Qualifying attendance at a tier 3–4 institution by year 4 |
|
|
36
|
+
| `att_elite_byY4` | Qualifying attendance at a tier 1–2 institution by year 4 |
|
|
37
|
+
| `cmp_selective_byY8` | AA or BA from a tier 3–4 institution by year 8 |
|
|
38
|
+
| `cmp_elite_byY8` | Existing AA-or-BA measure for tiers 1–2 |
|
|
39
|
+
| `cmp_BA_selective_byY8` | BA from a tier 3–4 institution by year 8 |
|
|
40
|
+
| `cmp_BA_elite_byY8` | BA from a tier 1–2 institution by year 8 |
|
|
41
|
+
| `cmp_BA_noimpute_byY8` | Any BA without institution-type credential imputation |
|
|
42
|
+
| `cmp_BA_selective_noimpute_byY8` | Same sensitivity definition for tiers 3–4 |
|
|
43
|
+
| `cmp_BA_elite_noimpute_byY8` | Same sensitivity definition for tiers 1–2 |
|
|
44
|
+
|
|
45
|
+
Attendance is available for `inY1`–`inY8`, `byY1`–`byY8`, their `fall` and
|
|
46
|
+
`spring` variants, and calendar ages 18–26 (e.g. `att_selective_20`). Completion
|
|
47
|
+
is available for `byY1`–`byY8`. All these selectivity and sensitivity fields are
|
|
48
|
+
retained in the selected final table. They follow the existing observation
|
|
49
|
+
windows and zero/null rules. Unknown institutional tiers do not count as
|
|
50
|
+
selective. Completion uses the degree-granting institution's tier, not the
|
|
51
|
+
first institution's tier.
|
|
52
|
+
|
|
53
|
+
The default BA measures retain the existing institution-type imputation.
|
|
54
|
+
The `noimpute` versions use credential-lookup/title evidence before that step:
|
|
55
|
+
an unknown award at a four-year institution alone does not count as a BA.
|
|
56
|
+
Students remain in the sample, and another identified BA can still qualify.
|
|
57
|
+
This does not change institutional IPEDS rates or coarse predictions.
|
|
58
|
+
|
|
59
|
+
Verification: run `uv run python tests/verify_nsc_outcomes.py` from the repository root with the local institutional support files present. This creates temporary synthetic records and does not require secured student inputs.
|
|
@@ -93,8 +93,8 @@ RANDOM_SEED = 3852804
|
|
|
93
93
|
SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
|
|
94
94
|
DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
|
|
95
95
|
STEM_DEFINITIONS = ["sarah", "dhs"]
|
|
96
|
-
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
|
|
97
|
-
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
|
|
96
|
+
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
97
|
+
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
98
98
|
AGE_ATTENDANCE_RANGE = range(18, 27)
|
|
99
99
|
|
|
100
100
|
# These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
|
|
@@ -573,7 +573,9 @@ college_ref = (
|
|
|
573
573
|
.with_columns(
|
|
574
574
|
(pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
|
|
575
575
|
pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
|
|
576
|
+
# Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
|
|
576
577
|
pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
|
|
578
|
+
pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
|
|
577
579
|
)
|
|
578
580
|
.with_columns(
|
|
579
581
|
(pl.col("college_4yr") + pl.col("college_2yr") > 0)
|
|
@@ -592,6 +594,7 @@ college_ref = (
|
|
|
592
594
|
"college_2yr",
|
|
593
595
|
"tier",
|
|
594
596
|
"college_elite",
|
|
597
|
+
"college_selective",
|
|
595
598
|
"k_mean",
|
|
596
599
|
"completion_rate_150pct_ip",
|
|
597
600
|
)
|
|
@@ -1061,6 +1064,7 @@ enroll_weeks = (
|
|
|
1061
1064
|
"college_2yr",
|
|
1062
1065
|
"tier",
|
|
1063
1066
|
"college_elite",
|
|
1067
|
+
"college_selective",
|
|
1064
1068
|
"k_mean",
|
|
1065
1069
|
"completion_rate_150pct_ip",
|
|
1066
1070
|
"term_start_date",
|
|
@@ -1131,6 +1135,7 @@ enroll = (
|
|
|
1131
1135
|
pl.col("college_2yr").max(),
|
|
1132
1136
|
pl.col("tier").drop_nulls().min(),
|
|
1133
1137
|
pl.col("college_elite").max(),
|
|
1138
|
+
pl.col("college_selective").max(),
|
|
1134
1139
|
pl.col("k_mean").drop_nulls().first(),
|
|
1135
1140
|
pl.col("completion_rate_150pct_ip").drop_nulls().first(),
|
|
1136
1141
|
pl.col("_term_att").max(),
|
|
@@ -1453,6 +1458,8 @@ degrees = (
|
|
|
1453
1458
|
.otherwise(pl.lit(None))
|
|
1454
1459
|
.alias("_degree")
|
|
1455
1460
|
)
|
|
1461
|
+
# Preserve credential/title evidence before filling unknown awards by sector.
|
|
1462
|
+
.with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
|
|
1456
1463
|
.with_columns(
|
|
1457
1464
|
pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
|
|
1458
1465
|
.then(pl.lit("BA"))
|
|
@@ -1504,6 +1511,12 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1504
1511
|
(pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
|
|
1505
1512
|
.max()
|
|
1506
1513
|
.alias(f"cmp_BA_byY{year}"),
|
|
1514
|
+
(
|
|
1515
|
+
pl.col("_by_end")
|
|
1516
|
+
* (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
|
|
1517
|
+
)
|
|
1518
|
+
.max()
|
|
1519
|
+
.alias(f"cmp_BA_noimpute_byY{year}"),
|
|
1507
1520
|
(pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
|
|
1508
1521
|
.max()
|
|
1509
1522
|
.alias(f"cmp_any_byY{year}"),
|
|
@@ -1514,6 +1527,28 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1514
1527
|
)
|
|
1515
1528
|
.max()
|
|
1516
1529
|
.alias(f"cmp_elite_byY{year}"),
|
|
1530
|
+
(
|
|
1531
|
+
pl.col("_by_end")
|
|
1532
|
+
* pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
|
|
1533
|
+
* pl.col("college_selective").fill_null(0).cast(pl.Int8)
|
|
1534
|
+
)
|
|
1535
|
+
.max()
|
|
1536
|
+
.alias(f"cmp_selective_byY{year}"),
|
|
1537
|
+
# BA-only outcomes support comparisons with any-BA completion.
|
|
1538
|
+
*[
|
|
1539
|
+
(
|
|
1540
|
+
pl.col("_by_end")
|
|
1541
|
+
* (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
|
|
1542
|
+
* pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
|
|
1543
|
+
)
|
|
1544
|
+
.max()
|
|
1545
|
+
.alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
|
|
1546
|
+
for college_type in ["selective", "elite"]
|
|
1547
|
+
for degree_column, suffix in [
|
|
1548
|
+
("_degree", ""),
|
|
1549
|
+
("_degree_no_type_fill", "_noimpute"),
|
|
1550
|
+
]
|
|
1551
|
+
],
|
|
1517
1552
|
*[
|
|
1518
1553
|
expression
|
|
1519
1554
|
for definition in STEM_DEFINITIONS
|
|
@@ -1932,7 +1967,6 @@ for label, tiers in tier_buckets.items():
|
|
|
1932
1967
|
.otherwise(pl.col("adj_cmp_rate_4yr"))
|
|
1933
1968
|
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
1934
1969
|
)
|
|
1935
|
-
print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
|
|
1936
1970
|
|
|
1937
1971
|
# Match the available 1098-T calendar years, including the missing 2015 year.
|
|
1938
1972
|
tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
|
|
@@ -2014,6 +2048,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
2014
2048
|
.alias("recovered_nsc_outcome")
|
|
2015
2049
|
)
|
|
2016
2050
|
|
|
2051
|
+
# Export every selectivity window, plus the BA sensitivity outcomes. Keep the
|
|
2052
|
+
# existing recovered-outcome definition above independent of this export list.
|
|
2053
|
+
selectivity_outcome_columns = [
|
|
2054
|
+
column for column in nsc_outcomes.columns
|
|
2055
|
+
if column.startswith((
|
|
2056
|
+
"att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
|
|
2057
|
+
"cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
|
|
2058
|
+
))
|
|
2059
|
+
]
|
|
2060
|
+
|
|
2017
2061
|
keep_columns = [
|
|
2018
2062
|
"sid_cepr",
|
|
2019
2063
|
"k_mean",
|
|
@@ -2032,140 +2076,36 @@ keep_columns = [
|
|
|
2032
2076
|
"adj_cmp_rate_coarsen_2yr",
|
|
2033
2077
|
"ID_FSC_firstinst",
|
|
2034
2078
|
"college_name_firstinst",
|
|
2079
|
+
"unitid_firstinst",
|
|
2080
|
+
"college_years_firstinst",
|
|
2035
2081
|
"tier_firstinst",
|
|
2036
2082
|
"completion_rate_150pct_firstinst",
|
|
2037
|
-
] + outcome_columns + stem_outcome_columns
|
|
2083
|
+
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
|
|
2084
|
+
keep_columns = list(dict.fromkeys(keep_columns))
|
|
2038
2085
|
|
|
2039
2086
|
nsc_outcomes_final = nsc_outcomes.select(keep_columns)
|
|
2040
2087
|
|
|
2088
|
+
# Retain the existing completeness check without generating audit tables.
|
|
2089
|
+
if nsc_outcomes.filter(
|
|
2090
|
+
pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
|
|
2091
|
+
).height > 0:
|
|
2092
|
+
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2093
|
+
|
|
2041
2094
|
OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
|
|
2042
2095
|
nsc_outcomes.write_parquet(OUTCOMES)
|
|
2043
2096
|
nsc_outcomes.with_columns(
|
|
2044
2097
|
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2045
2098
|
).write_csv(OUTCOMES.with_suffix(".csv"))
|
|
2046
2099
|
|
|
2100
|
+
# Save the selected analysis/audit table as well as the existing full output.
|
|
2101
|
+
final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
|
|
2102
|
+
nsc_outcomes_final.write_parquet(final_path)
|
|
2103
|
+
nsc_outcomes_final.with_columns(
|
|
2104
|
+
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2105
|
+
).write_csv(final_path.with_suffix(".csv"))
|
|
2106
|
+
|
|
2107
|
+
print(f"Wrote {final_path}")
|
|
2108
|
+
print(f"Wrote {final_path.with_suffix('.csv')}")
|
|
2047
2109
|
print(f"Wrote {OUTCOMES}")
|
|
2048
2110
|
print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
|
|
2049
2111
|
print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
|
|
2050
|
-
|
|
2051
|
-
first_college_coverage = (
|
|
2052
|
-
nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
|
|
2053
|
-
.group_by("ID_FSC_firstinst")
|
|
2054
|
-
.agg(
|
|
2055
|
-
pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
|
|
2056
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2057
|
-
.is_not_null()
|
|
2058
|
-
.any()
|
|
2059
|
-
.alias("has_completion_rate"),
|
|
2060
|
-
)
|
|
2061
|
-
)
|
|
2062
|
-
print(
|
|
2063
|
-
"First-college coverage: "
|
|
2064
|
-
f"{first_college_coverage['has_k_mean'].sum()}/"
|
|
2065
|
-
f"{first_college_coverage.height} with college-specific k_mean; "
|
|
2066
|
-
f"{first_college_coverage['has_completion_rate'].sum()}/"
|
|
2067
|
-
f"{first_college_coverage.height} with IPEDS completion rate"
|
|
2068
|
-
)
|
|
2069
|
-
print(
|
|
2070
|
-
"National student-weighted coarse values: "
|
|
2071
|
-
f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
|
|
2072
|
-
f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
|
|
2073
|
-
f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
|
|
2074
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
|
|
2075
|
-
"observed by-Y4 nonattender completion=0"
|
|
2076
|
-
)
|
|
2077
|
-
print(
|
|
2078
|
-
"Sample-institution IPEDS-cohort-weighted coarse values: "
|
|
2079
|
-
f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
|
|
2080
|
-
f"({sample_4yr_institutions} institutions), "
|
|
2081
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
|
|
2082
|
-
f"({sample_2yr_institutions} institutions)"
|
|
2083
|
-
)
|
|
2084
|
-
|
|
2085
|
-
observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
|
|
2086
|
-
coarse_audit = observable_y4.select(
|
|
2087
|
-
pl.len().alias("students"),
|
|
2088
|
-
pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
|
|
2089
|
-
pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
|
|
2090
|
-
pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
|
|
2091
|
-
pl.col("adj_cmp_rate_coarse")
|
|
2092
|
-
.null_count()
|
|
2093
|
-
.alias("adj_cmp_rate_coarse_missing"),
|
|
2094
|
-
pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
|
|
2095
|
-
pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
|
|
2096
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2097
|
-
.null_count()
|
|
2098
|
-
.alias("adj_cmp_rate_coarse_sample_missing"),
|
|
2099
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2100
|
-
.min()
|
|
2101
|
-
.alias("adj_cmp_rate_coarse_sample_min"),
|
|
2102
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2103
|
-
.max()
|
|
2104
|
-
.alias("adj_cmp_rate_coarse_sample_max"),
|
|
2105
|
-
)
|
|
2106
|
-
|
|
2107
|
-
if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
|
|
2108
|
-
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2109
|
-
# Attendees without a classified first college can have missing coarse rates.
|
|
2110
|
-
# Report those counts below rather than treating them as a build failure.
|
|
2111
|
-
print("Coarse outcome audit:")
|
|
2112
|
-
print(coarse_audit)
|
|
2113
|
-
|
|
2114
|
-
attendee_completion_audit = (
|
|
2115
|
-
observable_y4.filter(pl.col("att_any_byY4") == 1)
|
|
2116
|
-
.select(
|
|
2117
|
-
pl.len().alias("by_y4_attendees"),
|
|
2118
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2119
|
-
.is_null()
|
|
2120
|
-
.sum()
|
|
2121
|
-
.alias("missing_first_institution_rate"),
|
|
2122
|
-
(
|
|
2123
|
-
pl.col("completion_rate_150pct_firstinst").is_null()
|
|
2124
|
-
& pl.col("completion_rate_150pct_ip").is_not_null()
|
|
2125
|
-
)
|
|
2126
|
-
.sum()
|
|
2127
|
-
.alias("filled_by_tier_median"),
|
|
2128
|
-
pl.col("adj_cmp_rate")
|
|
2129
|
-
.is_null()
|
|
2130
|
-
.sum()
|
|
2131
|
-
.alias("missing_after_tier_median"),
|
|
2132
|
-
pl.col("adj_cmp_rate")
|
|
2133
|
-
.is_null()
|
|
2134
|
-
.mean()
|
|
2135
|
-
.alias("missing_after_tier_median_share"),
|
|
2136
|
-
(
|
|
2137
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2138
|
-
& pl.col("tier_firstinst").is_null()
|
|
2139
|
-
)
|
|
2140
|
-
.sum()
|
|
2141
|
-
.alias("missing_after_tier_median_no_tier"),
|
|
2142
|
-
(
|
|
2143
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2144
|
-
& pl.col("tier_firstinst").is_not_null()
|
|
2145
|
-
)
|
|
2146
|
-
.sum()
|
|
2147
|
-
.alias("missing_after_tier_median_with_tier"),
|
|
2148
|
-
)
|
|
2149
|
-
)
|
|
2150
|
-
print("By-Y4 attendee completion-rate audit:")
|
|
2151
|
-
print(attendee_completion_audit)
|
|
2152
|
-
print(
|
|
2153
|
-
nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
|
|
2154
|
-
"sid_cepr"
|
|
2155
|
-
)
|
|
2156
|
-
)
|
|
2157
|
-
|
|
2158
|
-
# Nonmissing year-eight completion identifies students with the full Y8 window.
|
|
2159
|
-
year8_attendance_audit = (
|
|
2160
|
-
nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
|
|
2161
|
-
.select(
|
|
2162
|
-
pl.len().alias("N_year8_sample"),
|
|
2163
|
-
(pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
|
|
2164
|
-
(
|
|
2165
|
-
(pl.col("att_any_byY4") == 0)
|
|
2166
|
-
& (pl.col("cmp_BA_byY8") == 1)
|
|
2167
|
-
).sum().alias("N_no_attendance_byY4_BA_byY8"),
|
|
2168
|
-
)
|
|
2169
|
-
)
|
|
2170
|
-
print("Year-eight sample attendance and late BA completion check:")
|
|
2171
|
-
print(year8_attendance_audit)
|
|
@@ -60,7 +60,6 @@ INPUT_NSC = PACKAGE_NSC / "input"
|
|
|
60
60
|
# Folder for secured Census inputs; no student records are packaged.
|
|
61
61
|
|
|
62
62
|
OUTPUT_NSC = PACKAGE_NSC / "output" / "new_crosswalk"
|
|
63
|
-
OUTPUT_NSC.mkdir(parents=True, exist_ok=True)
|
|
64
63
|
# Folder for final student-level NSC outcomes.
|
|
65
64
|
|
|
66
65
|
APPS = INPUT_NSC / "apps.csv"
|
|
@@ -93,8 +92,8 @@ RANDOM_SEED = 3852804
|
|
|
93
92
|
SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
|
|
94
93
|
DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
|
|
95
94
|
STEM_DEFINITIONS = ["sarah", "dhs"]
|
|
96
|
-
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
|
|
97
|
-
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
|
|
95
|
+
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
96
|
+
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
98
97
|
AGE_ATTENDANCE_RANGE = range(18, 27)
|
|
99
98
|
|
|
100
99
|
# These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
|
|
@@ -249,14 +248,12 @@ legacy_absent_candidates = legacy_raw.join(
|
|
|
249
248
|
official_crosswalk.select("ID_FSC"), on="ID_FSC", how="anti"
|
|
250
249
|
)
|
|
251
250
|
# Reproducible legacy fallback selection; approved sole-rate exceptions below
|
|
252
|
-
# still take precedence.
|
|
253
|
-
legacy_absent_candidates.write_csv(OUTPUT_NSC / "legacy_absent_candidates.csv")
|
|
251
|
+
# still take precedence.
|
|
254
252
|
legacy_append = legacy_absent_candidates.sort(
|
|
255
253
|
["ID_FSC", "legacy_unitid", "ID_OPE", "name"], nulls_last=True
|
|
256
254
|
).unique("ID_FSC", keep="first", maintain_order=True).select(
|
|
257
255
|
"ID_FSC", "legacy_unitid", pl.col("name").alias("college_name_crosswalk")
|
|
258
256
|
)
|
|
259
|
-
legacy_append.write_csv(OUTPUT_NSC / "legacy_appended_codes.csv")
|
|
260
257
|
combined_crosswalk = pl.concat([
|
|
261
258
|
official_crosswalk.with_columns(pl.lit("official").alias("mapping_source")),
|
|
262
259
|
legacy_append.with_columns(pl.lit("legacy_absent_code").alias("mapping_source")),
|
|
@@ -300,7 +297,6 @@ college_crosswalk = (
|
|
|
300
297
|
pl.col("ID_FSC").str.slice(0, 6).cast(pl.Int64, strict=False).alias("opeid"),
|
|
301
298
|
)
|
|
302
299
|
)
|
|
303
|
-
college_crosswalk.write_csv(OUTPUT_NSC / "mapping_decisions.csv")
|
|
304
300
|
|
|
305
301
|
|
|
306
302
|
###########################################################
|
|
@@ -616,7 +612,9 @@ college_ref = (
|
|
|
616
612
|
.with_columns(
|
|
617
613
|
(pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
|
|
618
614
|
pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
|
|
615
|
+
# Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
|
|
619
616
|
pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
|
|
617
|
+
pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
|
|
620
618
|
)
|
|
621
619
|
.with_columns(
|
|
622
620
|
(pl.col("college_4yr") + pl.col("college_2yr") > 0)
|
|
@@ -635,26 +633,13 @@ college_ref = (
|
|
|
635
633
|
"college_2yr",
|
|
636
634
|
"tier",
|
|
637
635
|
"college_elite",
|
|
636
|
+
"college_selective",
|
|
638
637
|
"k_mean",
|
|
639
638
|
"completion_rate_150pct_ip",
|
|
640
639
|
)
|
|
641
640
|
)
|
|
642
641
|
|
|
643
642
|
|
|
644
|
-
# Institution-universe coverage: code counts, not student counts.
|
|
645
|
-
coverage = college_crosswalk.join(
|
|
646
|
-
college_ref.select("ID_FSC", "college_years", "completion_rate_150pct_ip"),
|
|
647
|
-
on="ID_FSC", how="left", validate="1:1",
|
|
648
|
-
)
|
|
649
|
-
coverage.write_csv(OUTPUT_NSC / "institution_coverage.csv")
|
|
650
|
-
coverage.filter(pl.col("unitid").is_null()).write_csv(OUTPUT_NSC / "codes_without_unitid.csv")
|
|
651
|
-
coverage.filter(pl.col("unitid").is_not_null() & pl.col("college_years").is_null()).write_csv(OUTPUT_NSC / "mapped_codes_without_level.csv")
|
|
652
|
-
print("CROSSWALK COVERAGE", coverage.select(
|
|
653
|
-
pl.len().alias("codes"), pl.col("unitid").is_not_null().sum().alias("mapped"),
|
|
654
|
-
pl.col("college_years").is_not_null().sum().alias("classified"),
|
|
655
|
-
pl.col("completion_rate_150pct_ip").is_not_null().sum().alias("with_rate"),
|
|
656
|
-
))
|
|
657
|
-
|
|
658
643
|
###########################################################
|
|
659
644
|
# Load student universe and NSC records
|
|
660
645
|
###########################################################
|
|
@@ -1042,6 +1027,7 @@ enroll_weeks = (
|
|
|
1042
1027
|
"college_2yr",
|
|
1043
1028
|
"tier",
|
|
1044
1029
|
"college_elite",
|
|
1030
|
+
"college_selective",
|
|
1045
1031
|
"k_mean",
|
|
1046
1032
|
"completion_rate_150pct_ip",
|
|
1047
1033
|
"term_start_date",
|
|
@@ -1112,6 +1098,7 @@ enroll = (
|
|
|
1112
1098
|
pl.col("college_2yr").max(),
|
|
1113
1099
|
pl.col("tier").drop_nulls().min(),
|
|
1114
1100
|
pl.col("college_elite").max(),
|
|
1101
|
+
pl.col("college_selective").max(),
|
|
1115
1102
|
pl.col("k_mean").drop_nulls().first(),
|
|
1116
1103
|
pl.col("completion_rate_150pct_ip").drop_nulls().first(),
|
|
1117
1104
|
pl.col("_term_att").max(),
|
|
@@ -1434,6 +1421,8 @@ degrees = (
|
|
|
1434
1421
|
.otherwise(pl.lit(None))
|
|
1435
1422
|
.alias("_degree")
|
|
1436
1423
|
)
|
|
1424
|
+
# Preserve credential/title evidence before filling unknown awards by sector.
|
|
1425
|
+
.with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
|
|
1437
1426
|
.with_columns(
|
|
1438
1427
|
pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
|
|
1439
1428
|
.then(pl.lit("BA"))
|
|
@@ -1485,6 +1474,12 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1485
1474
|
(pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
|
|
1486
1475
|
.max()
|
|
1487
1476
|
.alias(f"cmp_BA_byY{year}"),
|
|
1477
|
+
(
|
|
1478
|
+
pl.col("_by_end")
|
|
1479
|
+
* (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
|
|
1480
|
+
)
|
|
1481
|
+
.max()
|
|
1482
|
+
.alias(f"cmp_BA_noimpute_byY{year}"),
|
|
1488
1483
|
(pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
|
|
1489
1484
|
.max()
|
|
1490
1485
|
.alias(f"cmp_any_byY{year}"),
|
|
@@ -1495,6 +1490,28 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1495
1490
|
)
|
|
1496
1491
|
.max()
|
|
1497
1492
|
.alias(f"cmp_elite_byY{year}"),
|
|
1493
|
+
(
|
|
1494
|
+
pl.col("_by_end")
|
|
1495
|
+
* pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
|
|
1496
|
+
* pl.col("college_selective").fill_null(0).cast(pl.Int8)
|
|
1497
|
+
)
|
|
1498
|
+
.max()
|
|
1499
|
+
.alias(f"cmp_selective_byY{year}"),
|
|
1500
|
+
# BA-only outcomes support comparisons with any-BA completion.
|
|
1501
|
+
*[
|
|
1502
|
+
(
|
|
1503
|
+
pl.col("_by_end")
|
|
1504
|
+
* (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
|
|
1505
|
+
* pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
|
|
1506
|
+
)
|
|
1507
|
+
.max()
|
|
1508
|
+
.alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
|
|
1509
|
+
for college_type in ["selective", "elite"]
|
|
1510
|
+
for degree_column, suffix in [
|
|
1511
|
+
("_degree", ""),
|
|
1512
|
+
("_degree_no_type_fill", "_noimpute"),
|
|
1513
|
+
]
|
|
1514
|
+
],
|
|
1498
1515
|
*[
|
|
1499
1516
|
expression
|
|
1500
1517
|
for definition in STEM_DEFINITIONS
|
|
@@ -1923,7 +1940,6 @@ for label, tiers in tier_buckets.items():
|
|
|
1923
1940
|
.otherwise(pl.col("adj_cmp_rate_4yr"))
|
|
1924
1941
|
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
1925
1942
|
)
|
|
1926
|
-
print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
|
|
1927
1943
|
|
|
1928
1944
|
# Match the available 1098-T calendar years, including the missing 2015 year.
|
|
1929
1945
|
tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
|
|
@@ -2005,6 +2021,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
2005
2021
|
.alias("recovered_nsc_outcome")
|
|
2006
2022
|
)
|
|
2007
2023
|
|
|
2024
|
+
# Export every selectivity window, plus the BA sensitivity outcomes. Keep the
|
|
2025
|
+
# existing recovered-outcome definition above independent of this export list.
|
|
2026
|
+
selectivity_outcome_columns = [
|
|
2027
|
+
column for column in nsc_outcomes.columns
|
|
2028
|
+
if column.startswith((
|
|
2029
|
+
"att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
|
|
2030
|
+
"cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
|
|
2031
|
+
))
|
|
2032
|
+
]
|
|
2033
|
+
|
|
2008
2034
|
keep_columns = [
|
|
2009
2035
|
"sid_cepr",
|
|
2010
2036
|
"k_mean",
|
|
@@ -2023,140 +2049,36 @@ keep_columns = [
|
|
|
2023
2049
|
"adj_cmp_rate_coarsen_2yr",
|
|
2024
2050
|
"ID_FSC_firstinst",
|
|
2025
2051
|
"college_name_firstinst",
|
|
2052
|
+
"unitid_firstinst",
|
|
2053
|
+
"college_years_firstinst",
|
|
2026
2054
|
"tier_firstinst",
|
|
2027
2055
|
"completion_rate_150pct_firstinst",
|
|
2028
|
-
] + outcome_columns + stem_outcome_columns
|
|
2056
|
+
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
|
|
2057
|
+
keep_columns = list(dict.fromkeys(keep_columns))
|
|
2029
2058
|
|
|
2030
2059
|
nsc_outcomes_final = nsc_outcomes.select(keep_columns)
|
|
2031
2060
|
|
|
2061
|
+
# Retain the existing completeness check without generating audit tables.
|
|
2062
|
+
if nsc_outcomes.filter(
|
|
2063
|
+
pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
|
|
2064
|
+
).height > 0:
|
|
2065
|
+
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2066
|
+
|
|
2032
2067
|
OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
|
|
2033
2068
|
nsc_outcomes.write_parquet(OUTCOMES)
|
|
2034
2069
|
nsc_outcomes.with_columns(
|
|
2035
2070
|
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2036
2071
|
).write_csv(OUTCOMES.with_suffix(".csv"))
|
|
2037
2072
|
|
|
2073
|
+
# Save the selected analysis/audit table as well as the existing full output.
|
|
2074
|
+
final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
|
|
2075
|
+
nsc_outcomes_final.write_parquet(final_path)
|
|
2076
|
+
nsc_outcomes_final.with_columns(
|
|
2077
|
+
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2078
|
+
).write_csv(final_path.with_suffix(".csv"))
|
|
2079
|
+
|
|
2080
|
+
print(f"Wrote {final_path}")
|
|
2081
|
+
print(f"Wrote {final_path.with_suffix('.csv')}")
|
|
2038
2082
|
print(f"Wrote {OUTCOMES}")
|
|
2039
2083
|
print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
|
|
2040
2084
|
print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
|
|
2041
|
-
|
|
2042
|
-
first_college_coverage = (
|
|
2043
|
-
nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
|
|
2044
|
-
.group_by("ID_FSC_firstinst")
|
|
2045
|
-
.agg(
|
|
2046
|
-
pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
|
|
2047
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2048
|
-
.is_not_null()
|
|
2049
|
-
.any()
|
|
2050
|
-
.alias("has_completion_rate"),
|
|
2051
|
-
)
|
|
2052
|
-
)
|
|
2053
|
-
print(
|
|
2054
|
-
"First-college coverage: "
|
|
2055
|
-
f"{first_college_coverage['has_k_mean'].sum()}/"
|
|
2056
|
-
f"{first_college_coverage.height} with college-specific k_mean; "
|
|
2057
|
-
f"{first_college_coverage['has_completion_rate'].sum()}/"
|
|
2058
|
-
f"{first_college_coverage.height} with IPEDS completion rate"
|
|
2059
|
-
)
|
|
2060
|
-
print(
|
|
2061
|
-
"National student-weighted coarse values: "
|
|
2062
|
-
f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
|
|
2063
|
-
f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
|
|
2064
|
-
f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
|
|
2065
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
|
|
2066
|
-
"observed by-Y4 nonattender completion=0"
|
|
2067
|
-
)
|
|
2068
|
-
print(
|
|
2069
|
-
"Sample-institution IPEDS-cohort-weighted coarse values: "
|
|
2070
|
-
f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
|
|
2071
|
-
f"({sample_4yr_institutions} institutions), "
|
|
2072
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
|
|
2073
|
-
f"({sample_2yr_institutions} institutions)"
|
|
2074
|
-
)
|
|
2075
|
-
|
|
2076
|
-
observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
|
|
2077
|
-
coarse_audit = observable_y4.select(
|
|
2078
|
-
pl.len().alias("students"),
|
|
2079
|
-
pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
|
|
2080
|
-
pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
|
|
2081
|
-
pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
|
|
2082
|
-
pl.col("adj_cmp_rate_coarse")
|
|
2083
|
-
.null_count()
|
|
2084
|
-
.alias("adj_cmp_rate_coarse_missing"),
|
|
2085
|
-
pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
|
|
2086
|
-
pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
|
|
2087
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2088
|
-
.null_count()
|
|
2089
|
-
.alias("adj_cmp_rate_coarse_sample_missing"),
|
|
2090
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2091
|
-
.min()
|
|
2092
|
-
.alias("adj_cmp_rate_coarse_sample_min"),
|
|
2093
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2094
|
-
.max()
|
|
2095
|
-
.alias("adj_cmp_rate_coarse_sample_max"),
|
|
2096
|
-
)
|
|
2097
|
-
|
|
2098
|
-
if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
|
|
2099
|
-
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2100
|
-
# Attendees without a classified first college can have missing coarse rates.
|
|
2101
|
-
# Report those counts below rather than treating them as a build failure.
|
|
2102
|
-
print("Coarse outcome audit:")
|
|
2103
|
-
print(coarse_audit)
|
|
2104
|
-
|
|
2105
|
-
attendee_completion_audit = (
|
|
2106
|
-
observable_y4.filter(pl.col("att_any_byY4") == 1)
|
|
2107
|
-
.select(
|
|
2108
|
-
pl.len().alias("by_y4_attendees"),
|
|
2109
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2110
|
-
.is_null()
|
|
2111
|
-
.sum()
|
|
2112
|
-
.alias("missing_first_institution_rate"),
|
|
2113
|
-
(
|
|
2114
|
-
pl.col("completion_rate_150pct_firstinst").is_null()
|
|
2115
|
-
& pl.col("completion_rate_150pct_ip").is_not_null()
|
|
2116
|
-
)
|
|
2117
|
-
.sum()
|
|
2118
|
-
.alias("filled_by_tier_median"),
|
|
2119
|
-
pl.col("adj_cmp_rate")
|
|
2120
|
-
.is_null()
|
|
2121
|
-
.sum()
|
|
2122
|
-
.alias("missing_after_tier_median"),
|
|
2123
|
-
pl.col("adj_cmp_rate")
|
|
2124
|
-
.is_null()
|
|
2125
|
-
.mean()
|
|
2126
|
-
.alias("missing_after_tier_median_share"),
|
|
2127
|
-
(
|
|
2128
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2129
|
-
& pl.col("tier_firstinst").is_null()
|
|
2130
|
-
)
|
|
2131
|
-
.sum()
|
|
2132
|
-
.alias("missing_after_tier_median_no_tier"),
|
|
2133
|
-
(
|
|
2134
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2135
|
-
& pl.col("tier_firstinst").is_not_null()
|
|
2136
|
-
)
|
|
2137
|
-
.sum()
|
|
2138
|
-
.alias("missing_after_tier_median_with_tier"),
|
|
2139
|
-
)
|
|
2140
|
-
)
|
|
2141
|
-
print("By-Y4 attendee completion-rate audit:")
|
|
2142
|
-
print(attendee_completion_audit)
|
|
2143
|
-
print(
|
|
2144
|
-
nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
|
|
2145
|
-
"sid_cepr"
|
|
2146
|
-
)
|
|
2147
|
-
)
|
|
2148
|
-
|
|
2149
|
-
# Nonmissing year-eight completion identifies students with the full Y8 window.
|
|
2150
|
-
year8_attendance_audit = (
|
|
2151
|
-
nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
|
|
2152
|
-
.select(
|
|
2153
|
-
pl.len().alias("N_year8_sample"),
|
|
2154
|
-
(pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
|
|
2155
|
-
(
|
|
2156
|
-
(pl.col("att_any_byY4") == 0)
|
|
2157
|
-
& (pl.col("cmp_BA_byY8") == 1)
|
|
2158
|
-
).sum().alias("N_no_attendance_byY4_BA_byY8"),
|
|
2159
|
-
)
|
|
2160
|
-
)
|
|
2161
|
-
print("Year-eight sample attendance and late BA completion check:")
|
|
2162
|
-
print(year8_attendance_audit)
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/aspire.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/helpers.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/mappings.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/christel_house.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/democracy_prep.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|