openom-cli 0.1.0__tar.gz → 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {openom_cli-0.1.0 → openom_cli-0.1.1}/PKG-INFO +1 -1
- {openom_cli-0.1.0 → openom_cli-0.1.1}/pyproject.toml +1 -1
- openom_cli-0.1.1/src/openom_cli/csv_map.py +171 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/main.py +124 -0
- openom_cli-0.1.1/tests/test_csv_map.py +160 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/.gitignore +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/README.md +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/__init__.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/buildout.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/buildout_pull.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/humanize.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/profile.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/py.typed +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/src/openom_cli/scaffold.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/tests/fixtures/buildout-listing-sample.json +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/tests/test_buildout_map.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/tests/test_buildout_pull.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/tests/test_cli.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/tests/test_cli_dx.py +0 -0
- {openom_cli-0.1.0 → openom_cli-0.1.1}/tests/test_cli_onboarding.py +0 -0
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
"""Deterministic CSV row -> openOM payload mapper (the spreadsheet on-ramp for bulk seeding).
|
|
3
|
+
|
|
4
|
+
A broker with a back catalog usually has a spreadsheet + a folder of PDFs, not Buildout JSON. This
|
|
5
|
+
maps one CSV row (a documented canonical column set) to a schema-valid openOM payload, ready for
|
|
6
|
+
``om embed-batch`` - the same output shape as the Buildout bridge, different input. Pure + zero
|
|
7
|
+
inference: every value comes from the broker's own cell (that IS the assertion); absent cells are
|
|
8
|
+
omitted, never guessed. The assertion identity (assertedBy / assertedDate / noiType / noiAsOfDate)
|
|
9
|
+
is supplied by the caller and stamped verbatim.
|
|
10
|
+
|
|
11
|
+
Numeric/date/state normalization reuses the SAME helpers as the Buildout mapper, so a number, a
|
|
12
|
+
percent, or an M/D/Y date maps identically on both on-ramps. Percentage columns are named ``*Pct``
|
|
13
|
+
(e.g. ``capRatePct`` = 6.25, not 0.0625) so a non-technical broker can't misread the unit.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import csv
|
|
19
|
+
import io
|
|
20
|
+
from collections.abc import Mapping
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from .buildout import (
|
|
24
|
+
NS,
|
|
25
|
+
_compact,
|
|
26
|
+
_int,
|
|
27
|
+
_iso_date,
|
|
28
|
+
_lease_type,
|
|
29
|
+
_months_between,
|
|
30
|
+
_num,
|
|
31
|
+
_pct_to_fraction,
|
|
32
|
+
_state_code,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
# The canonical header vocabulary a broker fills in. Order is the template's column order. Every
|
|
36
|
+
# column is optional except that a row must name its PDF (an ``id`` -> <id>.pdf, or an explicit
|
|
37
|
+
# ``pdf`` filename) and produce a schema-valid payload. ``*Pct`` columns are percentages.
|
|
38
|
+
CANONICAL_COLUMNS: tuple[str, ...] = (
|
|
39
|
+
"id", "pdf",
|
|
40
|
+
"broker", "brokerage", "license", "noiType", "noiAsOfDate",
|
|
41
|
+
"streetAddress", "city", "state", "postalCode", "country",
|
|
42
|
+
"propertyType", "buildingSF", "yearBuilt", "lotAcres", "units", "occupancyPct",
|
|
43
|
+
"latitude", "longitude",
|
|
44
|
+
"askingPrice", "capRatePct", "noi", "status",
|
|
45
|
+
"tenant", "leaseType", "commencement", "expiration", "guarantor",
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
# Per-row overrides of the assertion identity (a catalog can span brokers / NOI types).
|
|
49
|
+
_OVERRIDE_COLUMNS: tuple[str, ...] = ("broker", "brokerage", "license", "noiType", "noiAsOfDate")
|
|
50
|
+
|
|
51
|
+
_EXAMPLE_ROW: dict[str, str] = {
|
|
52
|
+
"id": "123-main", "pdf": "123-main.pdf",
|
|
53
|
+
"streetAddress": "123 Main St", "city": "Austin", "state": "TX", "postalCode": "78701",
|
|
54
|
+
"propertyType": "retail", "buildingSF": "9100", "yearBuilt": "2019",
|
|
55
|
+
"askingPrice": "1850000", "capRatePct": "6.25", "noi": "115625",
|
|
56
|
+
"tenant": "Example Retail, LLC", "leaseType": "NNN",
|
|
57
|
+
"commencement": "5/1/2019", "expiration": "4/30/2034",
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _cell(row: Mapping[str, str], key: str) -> str | None:
|
|
62
|
+
"""A trimmed cell value, or None when the column is absent/blank (so _compact drops it)."""
|
|
63
|
+
v = row.get(key)
|
|
64
|
+
if v is None:
|
|
65
|
+
return None
|
|
66
|
+
s = str(v).strip()
|
|
67
|
+
return s or None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _date(v: str | None) -> str | None:
|
|
71
|
+
"""Accept either an ISO date (pass-through) or an M/D/Y spreadsheet date."""
|
|
72
|
+
if not v:
|
|
73
|
+
return None
|
|
74
|
+
s = str(v).strip()
|
|
75
|
+
parts = s.split("-")
|
|
76
|
+
if len(parts) == 3 and len(parts[0]) == 4: # already ISO-ish -> validate via round-trip
|
|
77
|
+
try:
|
|
78
|
+
y, m, d = (int(p) for p in parts)
|
|
79
|
+
except ValueError:
|
|
80
|
+
return None
|
|
81
|
+
if 1 <= m <= 12 and 1 <= d <= 31 and y > 1900:
|
|
82
|
+
return f"{y:04d}-{m:02d}-{d:02d}"
|
|
83
|
+
return None
|
|
84
|
+
return _iso_date(s)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def override_identity(row: Mapping[str, str]) -> dict[str, str]:
|
|
88
|
+
"""The per-row assertion-identity overrides present in this row (subset of the override set)."""
|
|
89
|
+
return {k: v for k in _OVERRIDE_COLUMNS if (v := _cell(row, k)) is not None}
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def row_to_payload(
|
|
93
|
+
row: Mapping[str, str],
|
|
94
|
+
*,
|
|
95
|
+
asserted_by: dict[str, str],
|
|
96
|
+
asserted_date: str,
|
|
97
|
+
noi_type: str,
|
|
98
|
+
noi_as_of: str | None = None,
|
|
99
|
+
) -> dict[str, Any]:
|
|
100
|
+
"""Map one canonical CSV row to a schema-valid openOM payload (only the fields the row carries).
|
|
101
|
+
``asserted_by``/``asserted_date``/``noi_type``/``noi_as_of`` are the assertion identity and are
|
|
102
|
+
stamped verbatim, never inferred from the row."""
|
|
103
|
+
address = _compact({
|
|
104
|
+
"streetAddress": _cell(row, "streetAddress"),
|
|
105
|
+
"addressLocality": _cell(row, "city"),
|
|
106
|
+
"addressRegion": _state_code(_cell(row, "state")),
|
|
107
|
+
"postalCode": _cell(row, "postalCode"),
|
|
108
|
+
"addressCountry": _cell(row, "country") or ("US" if _cell(row, "state") else None),
|
|
109
|
+
})
|
|
110
|
+
lat, lng = _num(_cell(row, "latitude")), _num(_cell(row, "longitude"))
|
|
111
|
+
geo = {"latitude": lat, "longitude": lng} if lat is not None and lng is not None else None
|
|
112
|
+
building_sf = _int(_cell(row, "buildingSF"))
|
|
113
|
+
units = _int(_cell(row, "units"))
|
|
114
|
+
prop_type = _cell(row, "propertyType")
|
|
115
|
+
property_ = _compact({
|
|
116
|
+
"propertyType": prop_type.lower() if prop_type else None,
|
|
117
|
+
"address": address or None,
|
|
118
|
+
"geo": geo,
|
|
119
|
+
"buildingSF": building_sf,
|
|
120
|
+
"yearBuilt": _int(_cell(row, "yearBuilt")),
|
|
121
|
+
"lotAcres": _num(_cell(row, "lotAcres")),
|
|
122
|
+
"units": units,
|
|
123
|
+
"occupancy": _pct_to_fraction(_cell(row, "occupancyPct")),
|
|
124
|
+
})
|
|
125
|
+
|
|
126
|
+
price = _int(_cell(row, "askingPrice"))
|
|
127
|
+
deal = _compact({
|
|
128
|
+
"askingPrice": price,
|
|
129
|
+
"capRate": _pct_to_fraction(_cell(row, "capRatePct")),
|
|
130
|
+
"noi": _int(_cell(row, "noi")),
|
|
131
|
+
"pricePerUnit": round(price / units) if price and units else None,
|
|
132
|
+
"pricePerSF": round(price / building_sf, 2) if price and building_sf else None,
|
|
133
|
+
"noiType": noi_type,
|
|
134
|
+
"noiAsOfDate": noi_as_of or asserted_date,
|
|
135
|
+
"status": _cell(row, "status") or "active",
|
|
136
|
+
})
|
|
137
|
+
|
|
138
|
+
commencement = _date(_cell(row, "commencement"))
|
|
139
|
+
expiration = _date(_cell(row, "expiration"))
|
|
140
|
+
guarantor_name = _cell(row, "guarantor")
|
|
141
|
+
lease = _compact({
|
|
142
|
+
"tenantEntity": _cell(row, "tenant"),
|
|
143
|
+
"leaseTypeAsserted": _lease_type(_cell(row, "leaseType")),
|
|
144
|
+
"commencement": commencement,
|
|
145
|
+
"expiration": expiration,
|
|
146
|
+
"termMonths": _months_between(commencement, expiration),
|
|
147
|
+
"guarantor": {"name": guarantor_name, "type": "corporate"} if guarantor_name else None,
|
|
148
|
+
})
|
|
149
|
+
|
|
150
|
+
return _compact({
|
|
151
|
+
"@context": ["https://schema.org", NS],
|
|
152
|
+
"@type": "RealEstateListing",
|
|
153
|
+
"specVersion": "0.1",
|
|
154
|
+
"assertedBy": _compact(dict(asserted_by)),
|
|
155
|
+
"assertedDate": asserted_date,
|
|
156
|
+
"property": property_ or None,
|
|
157
|
+
"deal": deal or None,
|
|
158
|
+
"lease": lease or None,
|
|
159
|
+
"meta": {"supersedes": None},
|
|
160
|
+
})
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def template_csv() -> str:
|
|
164
|
+
"""A blank template: the canonical header row + one worked example, so a broker knows exactly
|
|
165
|
+
what to fill in. Assertion identity (broker/brokerage/license/noiType) normally comes from the
|
|
166
|
+
command flags; the columns exist only for a catalog that spans brokers."""
|
|
167
|
+
buf = io.StringIO()
|
|
168
|
+
writer = csv.writer(buf, lineterminator="\n") # csv.writer quotes cells containing commas
|
|
169
|
+
writer.writerow(CANONICAL_COLUMNS)
|
|
170
|
+
writer.writerow([_EXAMPLE_ROW.get(c, "") for c in CANONICAL_COLUMNS])
|
|
171
|
+
return buf.getvalue()
|
|
@@ -11,9 +11,11 @@ affect the exit code.
|
|
|
11
11
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
|
+
import csv
|
|
14
15
|
import dataclasses
|
|
15
16
|
import datetime
|
|
16
17
|
import functools
|
|
18
|
+
import io
|
|
17
19
|
import json
|
|
18
20
|
import sys
|
|
19
21
|
import time
|
|
@@ -40,6 +42,7 @@ from openom_core.validate import validate as _validate
|
|
|
40
42
|
from openom_cli import profile as _profile
|
|
41
43
|
from openom_cli import scaffold as _scaffold
|
|
42
44
|
from openom_cli.buildout import listing_to_payload, payload_coverage
|
|
45
|
+
from openom_cli.csv_map import CANONICAL_COLUMNS, override_identity, row_to_payload, template_csv
|
|
43
46
|
from openom_cli.humanize import footer as _err_footer
|
|
44
47
|
from openom_cli.humanize import humanize_finding as _humanize
|
|
45
48
|
|
|
@@ -635,6 +638,127 @@ def buildout_manifest(
|
|
|
635
638
|
})
|
|
636
639
|
|
|
637
640
|
|
|
641
|
+
@app.command(name="csv-manifest")
|
|
642
|
+
@_guard
|
|
643
|
+
def csv_manifest( # noqa: C901 - a linear map-each-row-then-report, read top-down
|
|
644
|
+
csv_file: Annotated[
|
|
645
|
+
Path | None, typer.Option("--csv", help="Spreadsheet of listings (canonical columns)")
|
|
646
|
+
] = None,
|
|
647
|
+
pdf_dir: Annotated[
|
|
648
|
+
Path | None, typer.Option(help="Dir of the OM PDFs (named per each row's id/pdf column)")
|
|
649
|
+
] = None,
|
|
650
|
+
out_dir: Annotated[
|
|
651
|
+
Path | None, typer.Option(help="Where payload sidecars + manifest.json are written")
|
|
652
|
+
] = None,
|
|
653
|
+
broker: Annotated[str | None, typer.Option(help="assertedBy.broker (who is asserting)")] = None,
|
|
654
|
+
brokerage: Annotated[str | None, typer.Option(help="assertedBy.brokerage")] = None,
|
|
655
|
+
license: Annotated[str | None, typer.Option(help="assertedBy.license")] = None, # noqa: A002
|
|
656
|
+
asserted_date: Annotated[
|
|
657
|
+
str | None, typer.Option(help="ISO 8601 assertion date")
|
|
658
|
+
] = None,
|
|
659
|
+
noi_type: Annotated[
|
|
660
|
+
str | None, typer.Option(help="in-place | pro-forma (a required assertion)")
|
|
661
|
+
] = None,
|
|
662
|
+
noi_as_of: Annotated[
|
|
663
|
+
str | None, typer.Option(help="deal.noiAsOfDate (default: --asserted-date)")
|
|
664
|
+
] = None,
|
|
665
|
+
min_fields: Annotated[
|
|
666
|
+
int, typer.Option(help="Flag any mapped payload with fewer than this many tracked fields")
|
|
667
|
+
] = 3,
|
|
668
|
+
template: Annotated[
|
|
669
|
+
Path | None,
|
|
670
|
+
typer.Option(help="Write a blank template CSV (headers + one example) here, then exit"),
|
|
671
|
+
] = None,
|
|
672
|
+
) -> None:
|
|
673
|
+
"""Bridge: turn a broker's spreadsheet into an ``om embed-batch`` manifest (catalog seed).
|
|
674
|
+
|
|
675
|
+
The low-touch on-ramp: a broker exports their back catalog to a CSV (the canonical columns; run
|
|
676
|
+
``--template`` to get a blank one) and drops the OM PDFs in a folder. Each row maps
|
|
677
|
+
deterministically (zero inference) to a schema-valid openOM payload written as ``<id>.om.json``,
|
|
678
|
+
paired with its ``<id>.pdf`` in a manifest, with a ``coverage.json`` triage report. Assertion
|
|
679
|
+
identity comes from the flags (or per-row ``broker``/``brokerage``/``license``/``noiType``
|
|
680
|
+
columns for a multi-broker catalog), stamped verbatim - never inferred. Review the payloads (the
|
|
681
|
+
assertion gate), then run ``om embed-batch --manifest <out-dir>/manifest.json``.
|
|
682
|
+
"""
|
|
683
|
+
if template is not None:
|
|
684
|
+
template.write_text(template_csv(), encoding="utf-8")
|
|
685
|
+
if not _output.quiet:
|
|
686
|
+
typer.echo(f"csv-manifest: wrote a template with {len(CANONICAL_COLUMNS)} columns "
|
|
687
|
+
f"-> {template}", err=True)
|
|
688
|
+
_emit({"template": str(template), "columns": list(CANONICAL_COLUMNS)})
|
|
689
|
+
return
|
|
690
|
+
|
|
691
|
+
missing = [
|
|
692
|
+
n for n, v in (
|
|
693
|
+
("--csv", csv_file), ("--pdf-dir", pdf_dir), ("--out-dir", out_dir),
|
|
694
|
+
("--broker", broker), ("--brokerage", brokerage), ("--license", license),
|
|
695
|
+
("--asserted-date", asserted_date), ("--noi-type", noi_type),
|
|
696
|
+
) if not v
|
|
697
|
+
]
|
|
698
|
+
if missing:
|
|
699
|
+
typer.echo(f"error: missing required option(s): {', '.join(missing)}", err=True)
|
|
700
|
+
raise typer.Exit(2)
|
|
701
|
+
# Narrow every required option to non-None for the type checker (the guard above proved it).
|
|
702
|
+
assert csv_file and pdf_dir and out_dir and asserted_date
|
|
703
|
+
assert broker and brokerage and license and noi_type
|
|
704
|
+
|
|
705
|
+
default_by = {"broker": broker, "brokerage": brokerage, "license": license}
|
|
706
|
+
rows = list(csv.DictReader(io.StringIO(_read_bytes(csv_file).decode("utf-8-sig"))))
|
|
707
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
708
|
+
manifest: list[dict[str, str]] = []
|
|
709
|
+
skipped: list[dict[str, str]] = []
|
|
710
|
+
coverage: list[dict[str, Any]] = []
|
|
711
|
+
sparse: list[str] = []
|
|
712
|
+
seen: set[str] = set()
|
|
713
|
+
for i, row in enumerate(rows):
|
|
714
|
+
pdf_name = (row.get("pdf") or "").strip() or (
|
|
715
|
+
f"{(row.get('id') or '').strip()}.pdf" if (row.get("id") or "").strip() else ""
|
|
716
|
+
)
|
|
717
|
+
stem = Path(pdf_name).stem if pdf_name else f"row{i + 1}"
|
|
718
|
+
if not pdf_name:
|
|
719
|
+
skipped.append({"id": stem, "reason": "row names no pdf (needs an 'id' or 'pdf')"})
|
|
720
|
+
continue
|
|
721
|
+
if stem in seen:
|
|
722
|
+
skipped.append({"id": stem, "reason": "duplicate id/pdf in the CSV"})
|
|
723
|
+
continue
|
|
724
|
+
seen.add(stem)
|
|
725
|
+
pdf = pdf_dir / pdf_name
|
|
726
|
+
if not pdf.exists():
|
|
727
|
+
skipped.append({"id": stem, "reason": f"no OM PDF at {pdf_name}"})
|
|
728
|
+
continue
|
|
729
|
+
ov = override_identity(row)
|
|
730
|
+
asserted_by = {k: ov.get(k, default_by[k]) for k in ("broker", "brokerage", "license")}
|
|
731
|
+
payload = row_to_payload(
|
|
732
|
+
row, asserted_by=asserted_by, asserted_date=asserted_date,
|
|
733
|
+
noi_type=ov.get("noiType", noi_type), noi_as_of=ov.get("noiAsOfDate", noi_as_of),
|
|
734
|
+
)
|
|
735
|
+
sidecar = out_dir / f"{stem}.om.json"
|
|
736
|
+
sidecar.write_text(json.dumps(payload, indent=2, ensure_ascii=False), encoding="utf-8")
|
|
737
|
+
manifest.append(
|
|
738
|
+
{"pdf": str(pdf.resolve()), "payload": str(sidecar.resolve()),
|
|
739
|
+
"assertedDate": asserted_date}
|
|
740
|
+
)
|
|
741
|
+
cov = payload_coverage(payload)
|
|
742
|
+
coverage.append({"id": stem, **cov})
|
|
743
|
+
if cov["filled"] < min_fields:
|
|
744
|
+
sparse.append(stem)
|
|
745
|
+
(out_dir / "manifest.json").write_text(json.dumps(manifest, indent=2), encoding="utf-8")
|
|
746
|
+
(out_dir / "coverage.json").write_text(
|
|
747
|
+
json.dumps({"listings": coverage, "sparse": sparse, "minFields": min_fields}, indent=2),
|
|
748
|
+
encoding="utf-8",
|
|
749
|
+
)
|
|
750
|
+
if not _output.quiet:
|
|
751
|
+
note = f", {len(sparse)} sparse (<{min_fields} fields)" if sparse else ""
|
|
752
|
+
typer.echo(
|
|
753
|
+
f"csv-manifest: {len(manifest)} mapped, {len(skipped)} skipped{note} -> {out_dir}",
|
|
754
|
+
err=True,
|
|
755
|
+
)
|
|
756
|
+
_emit({
|
|
757
|
+
"mapped": len(manifest), "skipped": skipped, "sparse": sparse,
|
|
758
|
+
"manifest": str(out_dir / "manifest.json"), "coverage": str(out_dir / "coverage.json"),
|
|
759
|
+
})
|
|
760
|
+
|
|
761
|
+
|
|
638
762
|
@app.command(name="buildout-pull")
|
|
639
763
|
@_guard
|
|
640
764
|
def buildout_pull(
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Cover the CSV->openOM mapper (pure, deterministic - the spreadsheet on-ramp for bulk seeding)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
import jsonschema
|
|
12
|
+
import pikepdf
|
|
13
|
+
from typer.testing import CliRunner
|
|
14
|
+
|
|
15
|
+
from openom_cli.csv_map import (
|
|
16
|
+
CANONICAL_COLUMNS,
|
|
17
|
+
override_identity,
|
|
18
|
+
row_to_payload,
|
|
19
|
+
template_csv,
|
|
20
|
+
)
|
|
21
|
+
from openom_cli.main import app
|
|
22
|
+
|
|
23
|
+
_runner = CliRunner()
|
|
24
|
+
|
|
25
|
+
SPEC = Path(__file__).resolve().parents[2] / "spec"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _schema() -> dict[str, Any]:
|
|
29
|
+
return json.loads((SPEC / "om-0.1.schema.json").read_text(encoding="utf-8"))
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
_BY = {"broker": "Jane Example", "brokerage": "Example Advisors", "license": "TX 12345"}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _map(row: dict[str, str]) -> dict[str, Any]:
|
|
36
|
+
return row_to_payload(
|
|
37
|
+
row, asserted_by=_BY, asserted_date="2026-08-15", noi_type="in-place"
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_full_row_maps_to_a_schema_valid_payload() -> None:
|
|
42
|
+
row = {
|
|
43
|
+
"id": "123-main", "pdf": "123-main.pdf",
|
|
44
|
+
"streetAddress": "123 Main St", "city": "Austin", "state": "TX - Texas",
|
|
45
|
+
"postalCode": "78701", "propertyType": "Retail", "buildingSF": "9,100",
|
|
46
|
+
"yearBuilt": "2019", "askingPrice": "1,850,000", "capRatePct": "6.25", "noi": "115625",
|
|
47
|
+
"tenant": "Example Retail, LLC", "leaseType": "NNN Lease",
|
|
48
|
+
"commencement": "5/1/2019", "expiration": "4/30/2034",
|
|
49
|
+
}
|
|
50
|
+
payload = _map(row)
|
|
51
|
+
jsonschema.Draft202012Validator(_schema()).validate(payload) # raises if invalid
|
|
52
|
+
assert payload["deal"]["capRate"] == 0.0625 # percent -> fraction
|
|
53
|
+
assert payload["deal"]["askingPrice"] == 1850000 # commas stripped
|
|
54
|
+
assert payload["property"]["address"]["addressRegion"] == "TX" # 'TX - Texas' -> 'TX'
|
|
55
|
+
assert payload["property"]["propertyType"] == "retail" # lowercased
|
|
56
|
+
assert payload["lease"]["leaseTypeAsserted"] == "NNN"
|
|
57
|
+
assert payload["lease"]["commencement"] == "2019-05-01" # M/D/Y -> ISO
|
|
58
|
+
assert payload["lease"]["termMonths"] == 179 # derived, deterministic
|
|
59
|
+
assert payload["property"]["address"]["addressCountry"] == "US" # defaulted from a US state
|
|
60
|
+
assert payload["deal"]["pricePerSF"] == round(1850000 / 9100, 2) # derived
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def test_iso_dates_pass_through() -> None:
|
|
64
|
+
payload = _map({"commencement": "2019-05-01", "expiration": "2034-04-30"})
|
|
65
|
+
assert payload["lease"]["commencement"] == "2019-05-01"
|
|
66
|
+
assert payload["lease"]["expiration"] == "2034-04-30"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_malformed_dates_are_dropped_not_crashed() -> None:
|
|
70
|
+
# A garbled ISO-looking date or an unparseable value is omitted (never a crash, never a guess).
|
|
71
|
+
payload = _map({"askingPrice": "1000000", "commencement": "2019-13-40", "expiration": "later"})
|
|
72
|
+
assert "commencement" not in payload.get("lease", {})
|
|
73
|
+
assert "expiration" not in payload.get("lease", {})
|
|
74
|
+
# a 4-digit-year, 3-part value with non-integer parts hits the ISO parse-failure branch
|
|
75
|
+
assert "lease" not in _map({"askingPrice": "1", "commencement": "20x9-05-01"})
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_absent_cells_are_omitted_never_guessed() -> None:
|
|
79
|
+
payload = _map({"askingPrice": "1000000"}) # nothing else
|
|
80
|
+
assert "capRate" not in payload["deal"]
|
|
81
|
+
assert "lease" not in payload # no lease cells -> no lease object
|
|
82
|
+
assert "property" not in payload # no property cells -> no property object
|
|
83
|
+
assert payload["assertedBy"] == _BY # identity stamped verbatim
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def test_blank_cells_treated_as_absent() -> None:
|
|
87
|
+
payload = _map({"askingPrice": "1000000", "capRatePct": " ", "tenant": ""})
|
|
88
|
+
assert "capRate" not in payload["deal"]
|
|
89
|
+
assert "lease" not in payload
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_override_identity_reads_only_present_override_columns() -> None:
|
|
93
|
+
row = {"broker": "Sam Other", "askingPrice": "1", "city": "Austin"}
|
|
94
|
+
assert override_identity(row) == {"broker": "Sam Other"}
|
|
95
|
+
assert override_identity({"askingPrice": "1"}) == {}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_template_has_headers_and_one_example_row() -> None:
|
|
99
|
+
text = template_csv()
|
|
100
|
+
rows = list(csv.reader(io.StringIO(text)))
|
|
101
|
+
assert rows[0] == list(CANONICAL_COLUMNS) # header is the canonical vocabulary
|
|
102
|
+
assert len(rows) == 2 # header + one worked example
|
|
103
|
+
# the example row itself maps to a schema-valid payload
|
|
104
|
+
example = dict(zip(CANONICAL_COLUMNS, rows[1], strict=True))
|
|
105
|
+
jsonschema.Draft202012Validator(_schema()).validate(_map(example))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
# --- end-to-end CLI: the full spreadsheet on-ramp (csv-manifest -> embed-batch -> read) ---
|
|
109
|
+
def _blank_pdf(path: Path) -> None:
|
|
110
|
+
pdf = pikepdf.new()
|
|
111
|
+
pdf.add_blank_page(page_size=(612, 792))
|
|
112
|
+
buf = io.BytesIO()
|
|
113
|
+
pdf.save(buf)
|
|
114
|
+
path.write_bytes(buf.getvalue())
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_csv_manifest_template_then_embed_batch_roundtrip(tmp_path: Path) -> None:
|
|
118
|
+
# 1) --template writes a fillable CSV; a broker fills two rows.
|
|
119
|
+
tmpl = tmp_path / "template.csv"
|
|
120
|
+
r0 = _runner.invoke(app, ["csv-manifest", "--template", str(tmpl)])
|
|
121
|
+
assert r0.exit_code == 0, r0.output
|
|
122
|
+
assert tmpl.read_text(encoding="utf-8").splitlines()[0].startswith("id,pdf,")
|
|
123
|
+
|
|
124
|
+
pdf_dir = tmp_path / "pdfs"
|
|
125
|
+
pdf_dir.mkdir()
|
|
126
|
+
_blank_pdf(pdf_dir / "deal-1.pdf")
|
|
127
|
+
_blank_pdf(pdf_dir / "deal-2.pdf")
|
|
128
|
+
csv_file = tmp_path / "catalog.csv"
|
|
129
|
+
csv_file.write_text(
|
|
130
|
+
"id,streetAddress,city,state,askingPrice,capRatePct,noi\n"
|
|
131
|
+
"deal-1,1 A St,Austin,TX,1850000,6.25,115625\n"
|
|
132
|
+
"deal-2,2 B St,Dallas,TX,2400000,7.0,168000\n"
|
|
133
|
+
"deal-3,3 C St,Waco,TX,900000,8,72000\n", # deal-3 has no PDF -> skipped
|
|
134
|
+
encoding="utf-8",
|
|
135
|
+
)
|
|
136
|
+
out = tmp_path / "mapped"
|
|
137
|
+
r1 = _runner.invoke(
|
|
138
|
+
app,
|
|
139
|
+
["csv-manifest", "--csv", str(csv_file), "--pdf-dir", str(pdf_dir), "--out-dir", str(out),
|
|
140
|
+
"--broker", "Jane Example", "--brokerage", "Example Advisors", "--license", "TX 12345",
|
|
141
|
+
"--asserted-date", "2026-08-15", "--noi-type", "in-place"],
|
|
142
|
+
)
|
|
143
|
+
assert r1.exit_code == 0, r1.output
|
|
144
|
+
summary = json.loads(r1.output[r1.output.index("{"):])
|
|
145
|
+
assert summary["mapped"] == 2
|
|
146
|
+
assert any(s["id"] == "deal-3" for s in summary["skipped"]) # no PDF -> skipped with a reason
|
|
147
|
+
|
|
148
|
+
# 2) embed-batch consumes the produced manifest; both OMs round-trip hash-valid.
|
|
149
|
+
r2 = _runner.invoke(
|
|
150
|
+
app,
|
|
151
|
+
["embed-batch", "--manifest", str(out / "manifest.json"),
|
|
152
|
+
"--out-dir", str(tmp_path / "embedded")],
|
|
153
|
+
)
|
|
154
|
+
assert r2.exit_code == 0, r2.output
|
|
155
|
+
for name in ("deal-1.pdf", "deal-2.pdf"):
|
|
156
|
+
rr = _runner.invoke(app, ["read", str(tmp_path / "embedded" / name)])
|
|
157
|
+
assert rr.exit_code == 0, rr.output
|
|
158
|
+
got = json.loads(rr.output)
|
|
159
|
+
assert got["verification"]["hashValid"] is True
|
|
160
|
+
assert got["payload"]["assertedBy"]["broker"] == "Jane Example"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|