flight-alloc 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. flight_alloc-0.0.1.dist-info/METADATA +11 -0
  2. flight_alloc-0.0.1.dist-info/RECORD +71 -0
  3. flight_alloc-0.0.1.dist-info/WHEEL +5 -0
  4. flight_alloc-0.0.1.dist-info/entry_points.txt +2 -0
  5. flight_alloc-0.0.1.dist-info/top_level.txt +1 -0
  6. src/__init__.py +0 -0
  7. src/allocator/__init__.py +6 -0
  8. src/allocator/caps.py +513 -0
  9. src/allocator/eligibility.py +302 -0
  10. src/allocator/greedy_fallback.py +106 -0
  11. src/allocator/invariants.py +124 -0
  12. src/allocator/p2f_priority.py +142 -0
  13. src/allocator/pair_validation.py +108 -0
  14. src/allocator/pairings.py +554 -0
  15. src/allocator/postpass_break.py +366 -0
  16. src/allocator/postpass_intl.py +483 -0
  17. src/allocator/postpass_p2f.py +723 -0
  18. src/allocator/postpass_rebalance.py +244 -0
  19. src/allocator/postsolve.py +549 -0
  20. src/allocator/recommender.py +348 -0
  21. src/allocator/windows.py +377 -0
  22. src/cli.py +41 -0
  23. src/config.py +168 -0
  24. src/greedy_fallback.py +102 -0
  25. src/io/__init__.py +0 -0
  26. src/io/export.py +270 -0
  27. src/io/export_xml.py +66 -0
  28. src/io/readers.py +1048 -0
  29. src/io/roster_library.py +89 -0
  30. src/plan.py +192 -0
  31. src/recommender_staffing.py +329 -0
  32. src/roster_store.py +159 -0
  33. src/schemas.py +1244 -0
  34. src/solver/__init__.py +0 -0
  35. src/solver/allocator_cpsat.py +1412 -0
  36. src/staged_overrides.py +468 -0
  37. src/state.py +494 -0
  38. src/step1_clean_flights.py +286 -0
  39. src/step2_extract_roster.py +316 -0
  40. src/step3_allocate_flights.py +1639 -0
  41. src/web/__init__.py +47 -0
  42. src/web/__main__.py +9 -0
  43. src/web/api/__init__.py +56 -0
  44. src/web/api/export.py +37 -0
  45. src/web/api/inputs.py +122 -0
  46. src/web/api/override_rows.py +138 -0
  47. src/web/api/pages.py +30 -0
  48. src/web/api/readbacks.py +72 -0
  49. src/web/api/recommender.py +72 -0
  50. src/web/api/runs.py +102 -0
  51. src/web/api/settings.py +201 -0
  52. src/web/api/zc.py +117 -0
  53. src/web/core/__init__.py +5 -0
  54. src/web/core/responses.py +91 -0
  55. src/web/core/router.py +167 -0
  56. src/web/core/static_files.py +85 -0
  57. src/web/overrides/__init__.py +66 -0
  58. src/web/overrides/airports.py +261 -0
  59. src/web/overrides/break_time.py +83 -0
  60. src/web/overrides/config_yaml.py +21 -0
  61. src/web/overrides/filters.py +187 -0
  62. src/web/overrides/rows.py +110 -0
  63. src/web/readback/__init__.py +67 -0
  64. src/web/readback/common.py +68 -0
  65. src/web/readback/dashboard.py +83 -0
  66. src/web/readback/planning.py +335 -0
  67. src/web/readback/session.py +158 -0
  68. src/web/readback/tables.py +163 -0
  69. src/web/runner.py +168 -0
  70. src/web/server.py +185 -0
  71. src/zc_store.py +221 -0
src/io/readers.py ADDED
@@ -0,0 +1,1048 @@
1
+ """Excel readers for the three uploaded input files.
2
+
3
+ The operator uploads the day's three workbooks from the dashboard; the
4
+ bytes live in ``state.STATE.inputs``. ``workbook_from_bytes`` turns a
5
+ payload into an openpyxl handle and the readers below pull typed rows
6
+ out of it.
7
+
8
+ Sheet lookup is forgiving: each reader asks for its canonical sheet
9
+ name (``IN_SVportal`` / ``IN_Staff`` / ``IN_AM_Roster``) and falls back
10
+ to the workbook's first sheet when that name is absent, because a file
11
+ exported straight out of the SV portal or the rostering tool carries
12
+ whatever sheet name that tool chose.
13
+
14
+ Override rows are no longer read from a worksheet at all — they live in
15
+ memory and are parsed by the ``read_*`` functions in the second half of
16
+ this module.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import io
22
+ import re
23
+ from collections.abc import Mapping, Sequence
24
+ from datetime import date as date_type
25
+ from datetime import datetime, time
26
+ from typing import Any
27
+
28
+ import openpyxl
29
+ from openpyxl import Workbook
30
+ from openpyxl.worksheet.worksheet import Worksheet
31
+
32
+ from ..config import Config, StatusConfig
33
+ from ..schemas import (
34
+ AMRosterRow,
35
+ CrewRosterRow,
36
+ CrewStatus,
37
+ P2FNomination,
38
+ PerStaffOverride,
39
+ RaiseCapForFlight,
40
+ RawFlightRow,
41
+ Role,
42
+ ShiftCode,
43
+ SkipINTLRemoval,
44
+ SkipP2FBuffer,
45
+ WaiveH10Pair,
46
+ )
47
+
48
+ # Canonical sheet names, tried first in every uploaded workbook.
49
+ SHEET_IN_SVPORTAL = "IN_SVportal"
50
+ SHEET_IN_STAFF = "IN_Staff"
51
+ SHEET_IN_AM_ROSTER = "IN_AM_Roster"
52
+
53
+
54
+ def workbook_from_bytes(data: bytes) -> Workbook:
55
+ """Open an uploaded .xlsx payload as a read-only-ish workbook."""
56
+ return openpyxl.load_workbook(io.BytesIO(data), data_only=True)
57
+
58
+
59
+ def pick_sheet(wb: Workbook, preferred: str) -> Worksheet:
60
+ """Return ``wb[preferred]`` when present, else the first sheet.
61
+
62
+ Uploaded files come straight from email, so we cannot insist on a
63
+ sheet name. Raises ValueError only when the workbook has no sheets
64
+ at all.
65
+ """
66
+ if preferred in wb.sheetnames:
67
+ return wb[preferred]
68
+ if not wb.sheetnames:
69
+ raise ValueError("uploaded workbook has no sheets")
70
+ return wb[wb.sheetnames[0]]
71
+
72
+
73
+ # ---------- shared parsing helpers ----------
74
+
75
+ _DATE_FORMAT_FALLBACKS: tuple[str, ...] = (
76
+ # ISO + common slash variants
77
+ "%Y-%m-%d",
78
+ "%Y/%m/%d",
79
+ "%m/%d/%Y", # US: 4/25/2026
80
+ "%d/%m/%Y", # UK / IN: 25/4/2026
81
+ "%m-%d-%Y",
82
+ "%d-%m-%Y",
83
+ # Compact
84
+ "%Y%m%d",
85
+ # With month name
86
+ "%d %b %Y", # 25 Apr 2026
87
+ "%d-%b-%Y", # 25-Apr-2026
88
+ "%d/%b/%Y", # 25/Apr/2026
89
+ "%d-%b-%y", # 25-Apr-26
90
+ "%b %d, %Y", # Apr 25, 2026
91
+ "%d %B %Y", # 25 April 2026
92
+ "%d-%B-%Y",
93
+ # Two-digit year
94
+ "%m/%d/%y",
95
+ "%d/%m/%y",
96
+ "%y-%m-%d",
97
+ )
98
+
99
+
100
+ def _parse_date(v: Any, fmt: str) -> date_type | None:
101
+ """Parse a cell value to a date. The configured ``fmt`` is tried
102
+ first, then a wide set of common variants — per user direction
103
+ 2026-05-10, real exports change date format frequently and the
104
+ engine should accept any common shape without a config edit.
105
+
106
+ Accepted automatically:
107
+ - datetime / date cells (real Excel date types)
108
+ - ISO: 2026-04-25, 2026/04/25, 20260425
109
+ - US: 4/25/2026, 04/25/2026, 4/25/26
110
+ - UK / IN: 25/4/2026, 25-04-2026
111
+ - Month-name: 25 Apr 2026, 25-Apr-2026, Apr 25 2026, 25 April 2026
112
+ """
113
+ if v is None:
114
+ return None
115
+ if isinstance(v, datetime):
116
+ return v.date()
117
+ if isinstance(v, date_type):
118
+ return v
119
+ if not isinstance(v, str):
120
+ raise TypeError(f"unsupported date value: {v!r} ({type(v).__name__})")
121
+ s = v.strip()
122
+ if not s:
123
+ return None
124
+ # Configured format first, then the fallback list.
125
+ for candidate in (fmt, *_DATE_FORMAT_FALLBACKS):
126
+ try:
127
+ return datetime.strptime(s, candidate).date()
128
+ except ValueError:
129
+ continue
130
+ raise ValueError(
131
+ f"date {s!r} did not match {fmt!r} or any common fallback. "
132
+ f"Add the format to _DATE_FORMAT_FALLBACKS in io/readers.py."
133
+ )
134
+
135
+
136
+ def _parse_time(v: Any) -> time | None:
137
+ if v is None:
138
+ return None
139
+ if isinstance(v, time):
140
+ return v
141
+ if isinstance(v, datetime):
142
+ return v.time()
143
+ if isinstance(v, str):
144
+ s = v.strip()
145
+ for fmt in ("%H:%M", "%H:%M:%S"):
146
+ try:
147
+ return datetime.strptime(s, fmt).time()
148
+ except ValueError:
149
+ continue
150
+ raise ValueError(f"unparseable time: {v!r}")
151
+ raise TypeError(f"unsupported time value: {v!r} ({type(v).__name__})")
152
+
153
+
154
+ def normalize_status(raw: Any, status_cfg: StatusConfig) -> tuple[CrewStatus, str]:
155
+ """Normalize a roster cell value to (CrewStatus, original_literal).
156
+
157
+ Per user direction 2026-05-10: accept ZC/M, M/ZC, zc/m, P2F/M,
158
+ M/P2F, p2f/m — any order and any case. Implementation: try direct
159
+ aliases first, then a canonical form (uppercased, slash-parts
160
+ sorted alphabetically) so 'p2f/m' and 'M/P2F' both map to the
161
+ canonical 'M/P2F' which is the enum value for CrewStatus.P2F_M.
162
+ """
163
+ if raw is None:
164
+ return CrewStatus.BLANK, ""
165
+ literal = str(raw).strip()
166
+ if literal == "":
167
+ return CrewStatus.BLANK, ""
168
+ # 1) Direct enum / alias match (covers all literal CrewStatus values).
169
+ aliased = status_cfg.aliases.get(literal, literal)
170
+ try:
171
+ return CrewStatus(aliased), literal
172
+ except ValueError:
173
+ pass
174
+ # 2) Canonical form: uppercase, split on '/', sort alphabetically.
175
+ # 'P2F/m' / 'p2f/M' / 'M/p2f' all become 'M/P2F'.
176
+ if "/" in literal:
177
+ parts = sorted(p.strip().upper() for p in literal.split("/") if p.strip())
178
+ canonical = "/".join(parts)
179
+ # Try alias first then enum.
180
+ aliased2 = status_cfg.aliases.get(canonical, canonical)
181
+ try:
182
+ return CrewStatus(aliased2), literal
183
+ except ValueError:
184
+ pass
185
+ # 3) Last resort: uppercase only.
186
+ upper = literal.upper()
187
+ aliased3 = status_cfg.aliases.get(upper, upper)
188
+ try:
189
+ return CrewStatus(aliased3), literal
190
+ except ValueError:
191
+ pass
192
+ return CrewStatus.OTHER, literal
193
+
194
+
195
+ def _parse_date_header(value: Any, fmt: str, cycle_year: int) -> date_type | None:
196
+ """Parse a wide-roster column header to a date. Accepts:
197
+ - datetime / date cells (real Excel date types) → returned as-is
198
+ - strings matching the configured ``fmt`` (e.g. 'Mon,30Mar') → parsed
199
+ - other types / unmatched strings → None (column ignored)
200
+ """
201
+ if isinstance(value, datetime):
202
+ return value.date()
203
+ if isinstance(value, date_type):
204
+ return value
205
+ if not isinstance(value, str):
206
+ return None
207
+ try:
208
+ parsed = datetime.strptime(value.strip(), fmt)
209
+ except ValueError:
210
+ return None
211
+ return parsed.replace(year=cycle_year).date()
212
+
213
+
214
+ _SHIFT_CODES_TUPLE: tuple[str, ...] = ("M", "A", "N", "M1", "A1")
215
+
216
+
217
+ def _coerce_shift(v: Any) -> ShiftCode | None:
218
+ if v is None:
219
+ return None
220
+ s = str(v).strip().upper()
221
+ if s in _SHIFT_CODES_TUPLE:
222
+ return s # type: ignore[return-value]
223
+ return None
224
+
225
+
226
+ def _coerce_date(v: Any) -> date_type | None:
227
+ if v is None:
228
+ return None
229
+ if isinstance(v, datetime):
230
+ return v.date()
231
+ if isinstance(v, date_type):
232
+ return v
233
+ if isinstance(v, str):
234
+ s = v.strip()
235
+ if not s:
236
+ return None
237
+ for fmt in ("%Y-%m-%d", "%m/%d/%Y", "%d/%m/%Y"):
238
+ try:
239
+ return datetime.strptime(s, fmt).date()
240
+ except ValueError:
241
+ continue
242
+ return None
243
+
244
+
245
+ def _coerce_bool(v: Any) -> bool:
246
+ if isinstance(v, bool):
247
+ return v
248
+ if isinstance(v, str):
249
+ return v.strip().upper() in ("YES", "Y", "TRUE", "1")
250
+ if isinstance(v, (int, float)):
251
+ return bool(v)
252
+ return False
253
+
254
+
255
+ def _coerce_role(v: Any) -> Role:
256
+ s = str(v).strip().upper()
257
+ return Role(s) if s in (r.value for r in Role) else Role.STAFF
258
+
259
+
260
+ # ---------- IN_SVportal ----------
261
+
262
+ def read_sv_portal(wb: Workbook, config: Config) -> list[RawFlightRow]:
263
+ """Read IN_SVportal. Column rename map still comes from config.io.sv_portal.columns.
264
+
265
+ 2026-05-25: also stash every cell into ``extra_columns`` keyed by
266
+ its literal header string, so the extraction-filter system can
267
+ match on arbitrary columns (REG, Crew, MEMO, etc.) that aren't
268
+ part of the canonical RawFlightRow schema.
269
+ """
270
+ cfg = config.io.sv_portal
271
+ ws = pick_sheet(wb, SHEET_IN_SVPORTAL)
272
+ rows_iter = ws.iter_rows(values_only=True)
273
+ header_row = next(rows_iter)
274
+ idx_by_field: dict[str, int] = {}
275
+ header_by_idx: dict[int, str] = {}
276
+ for i, h in enumerate(header_row):
277
+ if isinstance(h, str):
278
+ header_by_idx[i] = h
279
+ if h in cfg.columns:
280
+ idx_by_field[cfg.columns[h]] = i
281
+
282
+ out: list[RawFlightRow] = []
283
+ for row in rows_iter:
284
+ if not any(c is not None for c in row):
285
+ continue
286
+ kwargs: dict[str, Any] = {}
287
+ for field_name, idx in idx_by_field.items():
288
+ value = row[idx]
289
+ if value is None:
290
+ continue
291
+ if field_name == "date":
292
+ kwargs[field_name] = _parse_date(value, cfg.date_format)
293
+ elif field_name in ("dep_time", "arr_time"):
294
+ kwargs[field_name] = _parse_time(value)
295
+ else:
296
+ kwargs[field_name] = value
297
+ # Capture every cell for extraction-filter custom_header matching.
298
+ extra: dict[str, str] = {}
299
+ for idx, header in header_by_idx.items():
300
+ if idx >= len(row):
301
+ continue
302
+ v = row[idx]
303
+ if v is None:
304
+ continue
305
+ extra[header] = str(v).strip()
306
+ kwargs["extra_columns"] = extra
307
+ out.append(RawFlightRow.model_validate(kwargs))
308
+ return out
309
+
310
+
311
+ # ---------- IN_Staff / IN_AM_Roster (wide format) ----------
312
+
313
+ class DuplicateEmployeeIdError(ValueError):
314
+ """Raised when IN_Staff or IN_AM_Roster contains the same Employee_ID
315
+ on more than one data row.
316
+
317
+ Maps to ``W020 ERROR`` in the warnings system. Step orchestrators catch
318
+ this, write a W020 row to ``OUT_Warnings``, and return without producing
319
+ downstream output — silently merging two distinct staff under one ID
320
+ would let the solver double-allocate flights and corrupt workload
321
+ summaries, which is worse than failing loudly.
322
+
323
+ Attributes:
324
+ sheet_label: 'IN_Staff' or 'IN_AM_Roster'.
325
+ duplicates: dict mapping each duplicated Employee_ID to the list of
326
+ crew names that share it.
327
+ """
328
+
329
+ def __init__(self, sheet_label: str, duplicates: dict[str, list[str]]) -> None:
330
+ self.sheet_label = sheet_label
331
+ self.duplicates = duplicates
332
+ details = "; ".join(
333
+ f"{eid!r} → {len(names)} rows ({', '.join(names)})"
334
+ for eid, names in sorted(duplicates.items())
335
+ )
336
+ super().__init__(
337
+ f"W020: duplicate Employee_ID(s) in {sheet_label}: {details}. "
338
+ "Edit the Employee_ID column so each row has a unique value."
339
+ )
340
+
341
+
342
+ def _check_unique_employee_ids(
343
+ rows_with_ids: list[tuple[str, str]], sheet_label: str,
344
+ ) -> None:
345
+ """Raise ``DuplicateEmployeeIdError`` if any employee_id in
346
+ ``rows_with_ids`` (a list of ``(employee_id, name)`` tuples) appears
347
+ on more than one row.
348
+ """
349
+ seen: dict[str, list[str]] = {}
350
+ for eid, name in rows_with_ids:
351
+ seen.setdefault(eid, []).append(name)
352
+ duplicates = {eid: names for eid, names in seen.items() if len(names) > 1}
353
+ if duplicates:
354
+ raise DuplicateEmployeeIdError(sheet_label, duplicates)
355
+
356
+
357
+ class MissingEmployeeIdError(ValueError):
358
+ """Raised when a crew-roster row has a blank Employee_ID cell.
359
+
360
+ Per user direction 2026-05-21: every roster row MUST carry an
361
+ explicit Employee_ID so duplicate-name staff are unambiguously
362
+ distinguishable downstream (override resolution, allocation
363
+ readback, audit trail). The previous auto-gen ``STAFF_NNN`` /
364
+ ``AM_NNN`` fallback was removed — silent fallback let same-named
365
+ staff get conflated by anyone editing the override sheet.
366
+ """
367
+
368
+ def __init__(self, sheet_label: str, blank_names: list[str]) -> None:
369
+ self.sheet_label = sheet_label
370
+ self.blank_names = blank_names
371
+ preview = ", ".join(blank_names[:5])
372
+ if len(blank_names) > 5:
373
+ preview += f", … (+{len(blank_names) - 5} more)"
374
+ super().__init__(
375
+ f"W021: {len(blank_names)} row(s) in {sheet_label} have "
376
+ f"a blank Employee_ID: {preview}. Every staff row MUST "
377
+ "carry an Employee_ID — open the workbook, fill the column, "
378
+ "save, and re-run."
379
+ )
380
+
381
+ # ---- header vocabulary for the wide rosters -------------------------------
382
+ #
383
+ # The monthly staff roster the rostering team publishes looks like:
384
+ #
385
+ # row 1 CLC STAFF ROSTER FROM 31 AUG TILL 27 SEP 2026 (banner)
386
+ # row 2 S.no | Name | IGA | add info | Contact | Mon,31Aug | Tue,1Sep | ...
387
+ # row 3+ 1 | YOGESH BADHAN | 10001 | P2F/Corendon | 78389... | F | A | ...
388
+ #
389
+ # so the reader must (a) find the header row instead of assuming row 1,
390
+ # and (b) accept the names that file uses for each column. Every alias
391
+ # list is matched case-insensitively and the configured name always wins.
392
+
393
+ _NAME_ALIASES: tuple[str, ...] = ("Name", "STAFF NAME", "Staff Name", "Employee Name")
394
+ _ID_ALIASES: tuple[str, ...] = (
395
+ "ID", "Employee_ID", "employee_id", "EmployeeID", "IGA", "IGA No",
396
+ "IGA No.", "IGA ID", "IGA Number", "Emp ID", "Emp. ID",
397
+ )
398
+ _LICENSE_ALIASES: tuple[str, ...] = (
399
+ "License", "Licence", "add info", "Additional Info", "Add. Info",
400
+ )
401
+
402
+ # Rows above the header (a title banner, a blank spacer) are skipped;
403
+ # don't scan the whole sheet for one.
404
+ _HEADER_SCAN_ROWS = 15
405
+ _YEAR_RE = re.compile(r"\b(20\d{2})\b")
406
+
407
+
408
+ def _normalize_license(value: Any) -> str | None:
409
+ """'P2F/Corendon' -> 'P2F+Corendon'; 'P2F/' -> 'P2F'; '/' or blank -> None.
410
+
411
+ The roster's "add info" column separates qualifications with '/', and
412
+ uses a bare '/' for "none". Downstream code only asks whether the
413
+ string contains 'P2F', so the joined form keeps that working while
414
+ the placeholder slash stops being a fake licence.
415
+ """
416
+ if value is None:
417
+ return None
418
+ parts = [p.strip() for p in re.split(r"[/+,]", str(value)) if p.strip()]
419
+ return "+".join(parts) if parts else None
420
+
421
+
422
+ def _find_header_row(
423
+ rows: Sequence[Sequence[Any]], name_candidates: Sequence[str],
424
+ ) -> int:
425
+ """Index of the first row (within the scan window) that has a
426
+ name-column label in it; 0 when none does, so the caller's existing
427
+ 'name column not in header' error still fires with the first row."""
428
+ wanted = {n.strip().lower() for n in name_candidates}
429
+ for i, row in enumerate(rows[:_HEADER_SCAN_ROWS]):
430
+ for cell in row:
431
+ if isinstance(cell, str) and cell.strip().lower() in wanted:
432
+ return i
433
+ return 0
434
+
435
+
436
+ _WEEKDAY_RE = re.compile(r"^(mon|tue|wed|thu|fri|sat|sun)[a-z]*\.?$", re.I)
437
+
438
+
439
+ def _looks_like_subheader(
440
+ row: Sequence[Any], name_idx: int, date_cols: Iterable[int],
441
+ ) -> bool:
442
+ """A weekday strip under the header ('Mon Tue Wed ...') or a row with
443
+ no name in it — as opposed to the first real staff row."""
444
+ name = row[name_idx] if name_idx < len(row) else None
445
+ if not (isinstance(name, str) and name.strip()):
446
+ return True
447
+ cells = [row[c] for c in date_cols if c < len(row) and row[c] not in (None, "")]
448
+ return bool(cells) and all(
449
+ isinstance(c, str) and _WEEKDAY_RE.match(c.strip()) for c in cells
450
+ )
451
+
452
+
453
+ def _banner_year(rows: Sequence[Sequence[Any]], header_idx: int) -> int | None:
454
+ """A 4-digit year mentioned in the rows above the header, if any."""
455
+ for row in rows[:header_idx]:
456
+ for cell in row:
457
+ if isinstance(cell, str):
458
+ m = _YEAR_RE.search(cell)
459
+ if m:
460
+ return int(m.group(1))
461
+ return None
462
+
463
+
464
+ def _resolve_date_columns(
465
+ header: Sequence[Any], fmt: str, ref_year: int,
466
+ anchor: date_type | None = None,
467
+ ) -> dict[int, date_type]:
468
+ """Map column index -> date for every date-looking header cell.
469
+
470
+ Headers like 'Mon,31Aug' carry no year, and a roster can run across
471
+ New Year, so the year is *derived*, not configured:
472
+
473
+ * with a weekday in the header ('Mon,31Aug') the year is the one,
474
+ within a year of ``ref_year``, in which 31 Aug is a Monday;
475
+ * otherwise the year starts at ``ref_year`` and rolls forward
476
+ whenever the dates step backwards (Dec -> Jan).
477
+
478
+ ``anchor`` (the workbook's last-modified date) settles the year for
479
+ weekday-less headers like '26-May': the first date takes whichever
480
+ year lands it nearest the anchor, so a roster keeps resolving
481
+ correctly year after year without a config edit.
482
+
483
+ Real Excel date cells are taken as they are.
484
+ """
485
+ has_weekday = "%a" in fmt or "%A" in fmt
486
+ out: dict[int, date_type] = {}
487
+ prev: date_type | None = None
488
+ for i, h in enumerate(header):
489
+ d: date_type | None = None
490
+ if isinstance(h, datetime):
491
+ d = h.date()
492
+ elif isinstance(h, date_type):
493
+ d = h
494
+ elif isinstance(h, str) and h.strip():
495
+ s = h.strip()
496
+ base = prev.year if prev else ref_year
497
+ years = [base, base + 1, base - 1]
498
+ if has_weekday:
499
+ # The weekday name pins the year down; try nearby years.
500
+ fallback: date_type | None = None
501
+ for y in years:
502
+ try:
503
+ dt = datetime.strptime(f"{s} {y}", f"{fmt} %Y")
504
+ except ValueError:
505
+ continue
506
+ fallback = fallback or dt.date()
507
+ if dt.strftime("%a").lower() == s[:3].lower():
508
+ d = dt.date()
509
+ break
510
+ if d is None:
511
+ d = fallback
512
+ elif prev is None and anchor is not None:
513
+ best: date_type | None = None
514
+ for y in (anchor.year - 1, anchor.year, anchor.year + 1):
515
+ try:
516
+ cand = datetime.strptime(f"{s} {y}", f"{fmt} %Y").date()
517
+ except ValueError:
518
+ continue
519
+ if best is None or abs((cand - anchor).days) < abs((best - anchor).days):
520
+ best = cand
521
+ d = best
522
+ else:
523
+ for y in years[:2]:
524
+ try:
525
+ dt = datetime.strptime(f"{s} {y}", f"{fmt} %Y")
526
+ except ValueError:
527
+ continue
528
+ d = dt.date()
529
+ if prev is not None and d < prev:
530
+ continue # went backwards: try next year
531
+ break
532
+ if d is not None:
533
+ out[i] = d
534
+ prev = d
535
+ return out
536
+
537
+
538
+ def _find_header_idx(header: list[Any], *names: str) -> int | None:
539
+ """Return the column index of the first header in ``names`` that matches
540
+ case-insensitively, else None. Used so the reader can pick up
541
+ Employee_ID / License columns whether the assigner labelled them
542
+ 'Employee_ID', 'employee_id', 'EmployeeID', etc."""
543
+ norm_names = {n.strip().lower() for n in names}
544
+ for i, h in enumerate(header):
545
+ if isinstance(h, str) and h.strip().lower() in norm_names:
546
+ return i
547
+ return None
548
+
549
+
550
+ def _read_wide_roster_sheet(
551
+ ws: Worksheet,
552
+ *,
553
+ name_column: str,
554
+ date_header_format: str,
555
+ cycle_year: int,
556
+ skip_subheader_rows: int,
557
+ status_cfg: StatusConfig,
558
+ id_prefix: str,
559
+ ) -> list[tuple[str, str | None, str, dict[date_type, CrewStatus], dict[date_type, str]]]:
560
+ """Yield (employee_id, license, name, statuses, raw_statuses) per data row.
561
+
562
+ If the sheet has an ``Employee_ID`` header column, its value is used. If
563
+ blank or missing, the reader auto-generates ``{id_prefix}_NNN`` based on
564
+ the sheet row index (1-based, header row counted). This keeps IDs stable
565
+ across runs as long as the assigner doesn't reorder rows.
566
+
567
+ Same pattern for ``License``: blank → None.
568
+ """
569
+ all_rows = [list(r) for r in ws.iter_rows(values_only=True)]
570
+ if not all_rows:
571
+ raise ValueError("uploaded roster sheet is empty")
572
+ name_candidates = [name_column, *_NAME_ALIASES]
573
+ header_idx = _find_header_row(all_rows, name_candidates)
574
+ header = all_rows[header_idx]
575
+ body = all_rows[header_idx + 1:]
576
+ # Case-insensitive name-column match; the configured label first,
577
+ # then the aliases the real files use.
578
+ name_idx: int | None = None
579
+ for cand in name_candidates:
580
+ target = cand.strip().lower()
581
+ for i, h in enumerate(header):
582
+ if isinstance(h, str) and h.strip().lower() == target:
583
+ name_idx = i
584
+ break
585
+ if name_idx is not None:
586
+ break
587
+ if name_idx is None:
588
+ raise ValueError(
589
+ f"name column {name_column!r} (case-insensitive) "
590
+ f"not in header {header}"
591
+ )
592
+ # Canonical ID header is "ID"; "IGA" is what the monthly roster uses.
593
+ id_idx = _find_header_idx(header, *_ID_ALIASES)
594
+ lic_idx = _find_header_idx(header, *_LICENSE_ALIASES)
595
+ banner = _banner_year(all_rows, header_idx)
596
+ anchor: date_type | None = None
597
+ if banner is None:
598
+ props = getattr(getattr(ws, "parent", None), "properties", None)
599
+ stamp = getattr(props, "modified", None) or getattr(props, "created", None)
600
+ anchor = stamp.date() if isinstance(stamp, datetime) else None
601
+ ref_year = banner or cycle_year
602
+ date_by_idx = _resolve_date_columns(
603
+ header, date_header_format, ref_year, anchor,
604
+ )
605
+ # The AM roster may or may not carry a weekday strip under the
606
+ # header, so the configured skip count only applies to rows that
607
+ # actually look like one — never to the first staff row.
608
+ for _ in range(skip_subheader_rows):
609
+ if body and _looks_like_subheader(body[0], name_idx, date_by_idx):
610
+ body = body[1:]
611
+ else:
612
+ break
613
+ rows_iter = iter(body)
614
+ # Per user direction 2026-05-21: Employee_ID is now mandatory.
615
+ # Header must exist AND every row must have a non-blank value.
616
+ # The previous auto-gen fallback (``STAFF_NNN`` / ``AM_NNN``) was
617
+ # removed so two same-named staff can never silently share a
618
+ # solver identity — the roster team has to assign distinct IDs.
619
+ sheet_label = ws.title if hasattr(ws, "title") else "<sheet>"
620
+ if id_idx is None:
621
+ raise MissingEmployeeIdError(sheet_label, ["<ID / IGA column missing>"])
622
+ out: list[tuple[str, str | None, str, dict[date_type, CrewStatus], dict[date_type, str]]] = []
623
+ blank_id_names: list[str] = []
624
+ for row in rows_iter:
625
+ if name_idx >= len(row):
626
+ continue
627
+ name_val = row[name_idx]
628
+ if not isinstance(name_val, str) or not name_val.strip():
629
+ continue
630
+ emp_id = ""
631
+ if id_idx < len(row):
632
+ v = row[id_idx]
633
+ if isinstance(v, str) and v.strip():
634
+ emp_id = v.strip()
635
+ elif isinstance(v, float) and v.is_integer():
636
+ emp_id = str(int(v)) # 10001.0 -> "10001"
637
+ elif v is not None and not isinstance(v, str):
638
+ emp_id = str(v).strip()
639
+ if not emp_id:
640
+ # A footer / note line (text in the name column, nothing in
641
+ # any date column) is not a staff row; a real staff row
642
+ # without an ID still fails loudly below.
643
+ has_status = any(
644
+ c < len(row) and row[c] not in (None, "")
645
+ for c in date_by_idx
646
+ )
647
+ if has_status:
648
+ blank_id_names.append(name_val.strip())
649
+ continue
650
+ # License: from cell if present and non-blank, else None.
651
+ license_val: str | None = None
652
+ if lic_idx is not None and lic_idx < len(row):
653
+ license_val = _normalize_license(row[lic_idx])
654
+ statuses: dict[date_type, CrewStatus] = {}
655
+ raw_statuses: dict[date_type, str] = {}
656
+ for col_idx, d in date_by_idx.items():
657
+ if col_idx >= len(row):
658
+ continue
659
+ normalized, literal = normalize_status(row[col_idx], status_cfg)
660
+ statuses[d] = normalized
661
+ raw_statuses[d] = literal
662
+ out.append((emp_id, license_val, name_val.strip(), statuses, raw_statuses))
663
+ if blank_id_names:
664
+ raise MissingEmployeeIdError(sheet_label, blank_id_names)
665
+ return out
666
+
667
+
668
+ def read_staff_roster(wb: Workbook, config: Config) -> list[CrewRosterRow]:
669
+ cfg = config.io.staff_roster
670
+ rows = _read_wide_roster_sheet(
671
+ pick_sheet(wb, SHEET_IN_STAFF),
672
+ name_column=cfg.name_column,
673
+ date_header_format=cfg.date_header_format,
674
+ cycle_year=config.allocation.cycle_year,
675
+ skip_subheader_rows=0,
676
+ status_cfg=config.status,
677
+ id_prefix="STAFF",
678
+ )
679
+ out = [
680
+ CrewRosterRow(
681
+ employee_id=eid, name=n, role=Role.STAFF, license=lic,
682
+ status_by_date=s, raw_status_by_date=r,
683
+ )
684
+ for eid, lic, n, s, r in rows
685
+ ]
686
+ _check_unique_employee_ids(
687
+ [(r.employee_id, r.name) for r in out], "Staff roster",
688
+ )
689
+ return out
690
+
691
+
692
+ def _infer_am_role(
693
+ raw_status_by_date: dict[date_type, str], status_cfg: StatusConfig,
694
+ ) -> Role:
695
+ am_set = set(status_cfg.am_shifts)
696
+ zc_set = set(status_cfg.zc_shifts)
697
+ has_am = any(lit in am_set for lit in raw_status_by_date.values())
698
+ has_zc = any(lit in zc_set for lit in raw_status_by_date.values())
699
+ if has_am:
700
+ return Role.AM
701
+ if has_zc:
702
+ return Role.ZC
703
+ # Defensive fallback per plan §1.5.1: anyone in the AM roster sheet is AM/ZC
704
+ # by definition even if their visible window contains no /IGT or /ZC entries.
705
+ return Role.AM
706
+
707
+
708
+ def read_am_roster(wb: Workbook, config: Config) -> list[AMRosterRow]:
709
+ cfg = config.io.am_roster
710
+ rows = _read_wide_roster_sheet(
711
+ pick_sheet(wb, SHEET_IN_AM_ROSTER),
712
+ name_column=cfg.name_column,
713
+ date_header_format=cfg.date_header_format,
714
+ cycle_year=config.allocation.cycle_year,
715
+ skip_subheader_rows=cfg.subheader_rows_to_skip,
716
+ status_cfg=config.status,
717
+ id_prefix="AM",
718
+ )
719
+ out: list[AMRosterRow] = []
720
+ for eid, lic, n, s, r in rows:
721
+ role = _infer_am_role(r, config.status)
722
+ out.append(AMRosterRow(
723
+ employee_id=eid, name=n, role=role, license=lic,
724
+ status_by_date=s, raw_status_by_date=r,
725
+ ))
726
+ _check_unique_employee_ids(
727
+ [(r.employee_id, r.name) for r in out], "AM roster",
728
+ )
729
+ return out
730
+
731
+
732
+ # IN_Required reader removed 2026-05-21 alongside the IN_Required
733
+ # sheet itself — no callers left, and the schema RequiredStaffingRow
734
+ # is no longer used. The dashboard "Required / Gap" columns were
735
+ # dropped in the same change.
736
+
737
+
738
+ # ---------- overrides (in-memory) ----------
739
+ #
740
+ # Override rows used to live on an IN_Override worksheet. They now live
741
+ # in ``state.STATE.overrides`` as a list of dicts keyed by the column
742
+ # names in ``state.OVERRIDE_HEADERS``. Each reader below filters that
743
+ # list by the row's ``type`` and returns its own typed list, so the
744
+ # orchestrators never have to discriminate.
745
+ #
746
+ # type=p2f → P2FNomination (date, shift, employee_id)
747
+ # type=max_flights → PerStaffOverride (max_flights)
748
+ # type=cutoff_time → PerStaffOverride (std_start / std_cutoff)
749
+ # type=sick → employee_id list
750
+ # type=change_role → {employee_id: new_role}
751
+ # type=waive_h10_pair → WaiveH10Pair
752
+ # type=raise_cap → RaiseCapForFlight
753
+ # type=skip_p2f_buffer → SkipP2FBuffer
754
+ # type=skip_intl_removal → SkipINTLRemoval
755
+
756
+ OverrideRows = Sequence[Mapping[str, Any]]
757
+
758
+
759
+ def _cell(row: Mapping[str, Any], *names: str) -> str:
760
+ """First non-blank value among ``names``, stripped. '' when none."""
761
+ for n in names:
762
+ v = row.get(n)
763
+ if v is None:
764
+ continue
765
+ s = str(v).strip()
766
+ if s:
767
+ return s
768
+ return ""
769
+
770
+
771
+ def _row_type(row: Mapping[str, Any]) -> str:
772
+ """Lowercased ``type`` value for a row, or '' when blank."""
773
+ return _cell(row, "type").lower()
774
+
775
+
776
+ def _row_employee_name(row: Mapping[str, Any]) -> str | None:
777
+ """Staff name off the row. None when blank."""
778
+ return _cell(row, "employee") or None
779
+
780
+
781
+ def _rows_of_type(overrides: OverrideRows, wanted: str) -> list[Mapping[str, Any]]:
782
+ return [r for r in overrides if _row_type(r) == wanted]
783
+
784
+
785
+ def _resolve_name_to_id(
786
+ name: str,
787
+ name_to_id: dict[str, str] | None,
788
+ ) -> str:
789
+ """Resolve a staff name to an employee_id using the provided map.
790
+ If no map is given, or the name isn't found, return the name as-is —
791
+ Step 4 will surface a W212 if it can't find that staff in the
792
+ roster anyway."""
793
+ if not name_to_id:
794
+ return name
795
+ return name_to_id.get(name.strip().upper()) or name
796
+
797
+
798
+ def read_p2f_nominations(
799
+ overrides: OverrideRows, d_day: date_type,
800
+ name_to_id: dict[str, str] | None = None,
801
+ ) -> list[P2FNomination]:
802
+ """Read all rows where ``type=p2f``. Returns one P2FNomination per
803
+ (date, shift). Date is stamped from ``d_day``; the employee column
804
+ is a name, resolved via ``name_to_id`` when provided."""
805
+ from ..schemas import OverrideType
806
+ out: list[P2FNomination] = []
807
+ for row in _rows_of_type(overrides, OverrideType.P2F.value):
808
+ s = _coerce_shift(_cell(row, "shift"))
809
+ name = _row_employee_name(row)
810
+ if s is None or not name:
811
+ continue
812
+ out.append(P2FNomination(
813
+ date=d_day, shift=s,
814
+ employee_id=_resolve_name_to_id(name, name_to_id),
815
+ ))
816
+ return out
817
+
818
+
819
+ def read_per_staff_overrides(
820
+ overrides: OverrideRows, d_day: date_type,
821
+ name_to_id: dict[str, str] | None = None,
822
+ ) -> list[PerStaffOverride]:
823
+ """Read max_flights / cutoff_time rows. Each row produces one
824
+ PerStaffOverride; the field populated depends on the row's type.
825
+
826
+ cutoff_time rows may carry a ``from_time`` (start of the custom
827
+ window) alongside ``to_time`` / ``limit`` (the upper bound). Either
828
+ may be blank; a row with both blank is skipped."""
829
+ import contextlib
830
+ from ..schemas import OverrideType
831
+ out: list[PerStaffOverride] = []
832
+ _PER_STAFF_TYPES = (
833
+ OverrideType.MAX_FLIGHTS.value,
834
+ OverrideType.CUTOFF_TIME.value,
835
+ )
836
+ for row in overrides:
837
+ rt = _row_type(row)
838
+ if rt not in _PER_STAFF_TYPES:
839
+ continue
840
+ name = _row_employee_name(row)
841
+ if not name:
842
+ continue
843
+ eid = _resolve_name_to_id(name, name_to_id)
844
+ max_flights: int | None = None
845
+ std_cutoff: time | None = None
846
+ std_start: time | None = None
847
+ if rt == OverrideType.MAX_FLIGHTS.value:
848
+ v = _cell(row, "limit")
849
+ if not v:
850
+ continue
851
+ with contextlib.suppress(TypeError, ValueError):
852
+ max_flights = int(float(v))
853
+ if max_flights is None:
854
+ continue
855
+ elif rt == OverrideType.CUTOFF_TIME.value:
856
+ v_to = _cell(row, "to_time", "std_cutoff", "limit")
857
+ if v_to:
858
+ with contextlib.suppress(TypeError, ValueError):
859
+ std_cutoff = _parse_time(v_to)
860
+ v_from = _cell(row, "from_time", "std_start")
861
+ if v_from:
862
+ with contextlib.suppress(TypeError, ValueError):
863
+ std_start = _parse_time(v_from)
864
+ # Nothing to enforce if both bounds are blank.
865
+ if std_cutoff is None and std_start is None:
866
+ continue
867
+ try:
868
+ out.append(PerStaffOverride(
869
+ employee_id=eid, date=d_day,
870
+ max_flights=max_flights, std_cutoff=std_cutoff,
871
+ std_start=std_start, is_newbie=False,
872
+ ))
873
+ except ValueError:
874
+ continue
875
+ return out
876
+
877
+
878
+ def read_role_change_overrides(
879
+ overrides: OverrideRows, d_day: date_type,
880
+ name_to_id: dict[str, str] | None = None,
881
+ ) -> dict[str, str]:
882
+ """Read ``type=change_role`` rows. Returns ``{employee_id: new_role}``.
883
+
884
+ The ``shift`` column carries the new role label (STAFF / ZC / AM) —
885
+ it is otherwise unused for this row type.
886
+
887
+ Engine semantics (step3):
888
+ - STAFF -> ZC: bucket switches to the ZC band (target 14-15 for
889
+ day shifts, 11 for N). Extra flights above that target get
890
+ redistributed via the pin re-solve.
891
+ - STAFF/ZC -> AM: shift_today set to None so the staff becomes
892
+ ineligible for any flight; everything they were carrying
893
+ redistributes.
894
+ - AM -> STAFF/ZC: AM staff who weren't in the assignable pool now
895
+ ARE; engine treats them as a fresh staff member on the resolved
896
+ shift.
897
+
898
+ Invalid new_role values are silently dropped — same as every other
899
+ override reader.
900
+ """
901
+ from ..schemas import OverrideType
902
+ out: dict[str, str] = {}
903
+ for row in _rows_of_type(overrides, OverrideType.CHANGE_ROLE.value):
904
+ name = _row_employee_name(row)
905
+ new_role = _cell(row, "shift", "role").upper()
906
+ if not name or new_role not in ("STAFF", "ZC", "AM"):
907
+ continue
908
+ out[_resolve_name_to_id(name, name_to_id)] = new_role
909
+ return out
910
+
911
+
912
+ def read_sick_overrides(
913
+ overrides: OverrideRows, d_day: date_type,
914
+ name_to_id: dict[str, str] | None = None,
915
+ ) -> list[str]:
916
+ """Read ``type=sick`` rows. Returns employee_ids (or names, when no
917
+ name_to_id map is given) flagged as off-duty for today.
918
+
919
+ Step 3 applies these by zeroing the staff's shift_today before
920
+ eligibility / pair generation, so their flights redistribute to the
921
+ remaining staff on the same shift.
922
+ """
923
+ from ..schemas import OverrideType
924
+ out: list[str] = []
925
+ for row in _rows_of_type(overrides, OverrideType.SICK.value):
926
+ name = _row_employee_name(row)
927
+ if not name:
928
+ continue
929
+ out.append(_resolve_name_to_id(name, name_to_id))
930
+ return out
931
+
932
+
933
+ # ---------- Phase R: per-flight relaxation override readers ----------
934
+ #
935
+ # All four discriminate on the ``type`` key. Common columns: ``flight``
936
+ # (FLT number) and ``std`` (STD of the flight being relaxed).
937
+ # Type-specific: waive_h10_pair → employee + other_std; raise_cap and
938
+ # skip_p2f_buffer → employee; skip_intl_removal → nothing more.
939
+
940
+
941
+ def _row_flight_std(
942
+ row: Mapping[str, Any],
943
+ ) -> tuple[str | None, time | None]:
944
+ """Pull (flight, std) off an override row. (None, None) when either
945
+ is missing or unparseable."""
946
+ flight = _cell(row, "flight", "flt") or None
947
+ std: time | None = None
948
+ raw_std = _cell(row, "std")
949
+ if raw_std:
950
+ try:
951
+ std = _parse_time(raw_std)
952
+ except (TypeError, ValueError):
953
+ std = None
954
+ return flight, std
955
+
956
+
957
+ def read_waive_h10_pairs(
958
+ overrides: OverrideRows, d_day: date_type,
959
+ name_to_id: dict[str, str] | None = None,
960
+ ) -> list[WaiveH10Pair]:
961
+ """Read all ``type=waive_h10_pair`` rows."""
962
+ from ..schemas import OverrideType
963
+ out: list[WaiveH10Pair] = []
964
+ for row in _rows_of_type(overrides, OverrideType.WAIVE_H10_PAIR.value):
965
+ flight, std = _row_flight_std(row)
966
+ name = _row_employee_name(row)
967
+ if flight is None or std is None or not name:
968
+ continue
969
+ other_std: time | None = None
970
+ raw_other = _cell(row, "other_std")
971
+ if raw_other:
972
+ try:
973
+ other_std = _parse_time(raw_other)
974
+ except (TypeError, ValueError):
975
+ other_std = None
976
+ if other_std is None:
977
+ continue
978
+ try:
979
+ out.append(WaiveH10Pair(
980
+ date=d_day, flight=flight, std=std,
981
+ employee_id=_resolve_name_to_id(name, name_to_id),
982
+ other_std=other_std,
983
+ ))
984
+ except ValueError:
985
+ continue
986
+ return out
987
+
988
+
989
+ def read_raise_cap_overrides(
990
+ overrides: OverrideRows, d_day: date_type,
991
+ name_to_id: dict[str, str] | None = None,
992
+ ) -> list[RaiseCapForFlight]:
993
+ """Read all ``type=raise_cap`` rows."""
994
+ from ..schemas import OverrideType
995
+ out: list[RaiseCapForFlight] = []
996
+ for row in _rows_of_type(overrides, OverrideType.RAISE_CAP.value):
997
+ flight, std = _row_flight_std(row)
998
+ name = _row_employee_name(row)
999
+ if flight is None or std is None or not name:
1000
+ continue
1001
+ try:
1002
+ out.append(RaiseCapForFlight(
1003
+ date=d_day, flight=flight, std=std,
1004
+ employee_id=_resolve_name_to_id(name, name_to_id),
1005
+ ))
1006
+ except ValueError:
1007
+ continue
1008
+ return out
1009
+
1010
+
1011
+ def read_skip_p2f_buffer_overrides(
1012
+ overrides: OverrideRows, d_day: date_type,
1013
+ name_to_id: dict[str, str] | None = None,
1014
+ ) -> list[SkipP2FBuffer]:
1015
+ """Read all ``type=skip_p2f_buffer`` rows."""
1016
+ from ..schemas import OverrideType
1017
+ out: list[SkipP2FBuffer] = []
1018
+ for row in _rows_of_type(overrides, OverrideType.SKIP_P2F_BUFFER.value):
1019
+ flight, std = _row_flight_std(row)
1020
+ name = _row_employee_name(row)
1021
+ if flight is None or std is None or not name:
1022
+ continue
1023
+ try:
1024
+ out.append(SkipP2FBuffer(
1025
+ date=d_day, flight=flight, std=std,
1026
+ employee_id=_resolve_name_to_id(name, name_to_id),
1027
+ ))
1028
+ except ValueError:
1029
+ continue
1030
+ return out
1031
+
1032
+
1033
+ def read_skip_intl_removal_overrides(
1034
+ overrides: OverrideRows, d_day: date_type,
1035
+ ) -> list[SkipINTLRemoval]:
1036
+ """Read all ``type=skip_intl_removal`` rows. Per-flight only — the
1037
+ override applies regardless of which handler took the INTL flight."""
1038
+ from ..schemas import OverrideType
1039
+ out: list[SkipINTLRemoval] = []
1040
+ for row in _rows_of_type(overrides, OverrideType.SKIP_INTL_REMOVAL.value):
1041
+ flight, std = _row_flight_std(row)
1042
+ if flight is None or std is None:
1043
+ continue
1044
+ try:
1045
+ out.append(SkipINTLRemoval(date=d_day, flight=flight, std=std))
1046
+ except ValueError:
1047
+ continue
1048
+ return out