from __future__ import annotations from typing import Dict, List, Tuple, Optional import re from schema import ExtractionResult, DealSnapshot, ClassifiedTable, ReadResult from normalize import clean_text, find_first, moneyify from table_classifier import classify_table from utils import rows_to_records PROPERTY_KEYS = { "address": ["address"], "number of units": ["number of units", "units"], "year built": ["year built"], "total size": ["total size", "size", "building sf", "gla"], "unit size range": ["unit size range"], "site acreage": ["site acreage", "site area"], "parcel numbers": ["parcel numbers", "parcel"], "current noi": ["current noi"], "pro forma noi": ["pro forma noi", "proforma noi"], "offering price": ["offering price", "purchase price", "offering"], } def _norm(s: str) -> str: return re.sub(r"\s+", " ", (s or "").strip().lower()) def _table_to_kv(rows: List[List[str]]) -> Dict[str, str]: """ Works for: - two-column tables: key | value - vertical key list then vertical value list (less common) """ kv: Dict[str, str] = {} # common: 2 columns for r in rows: if len(r) >= 2: k = _norm(r[0]) v = (r[1] or "").strip() if k and v: kv[k] = v # sometimes extracted as one row with keys + one row with values if not kv and len(rows) >= 2 and len(rows[0]) == len(rows[1]): keys = [_norm(x) for x in rows[0]] vals = [x.strip() for x in rows[1]] for k, v in zip(keys, vals): if k and v: kv[k] = v return kv def _pick(kv: Dict[str, str], aliases: List[str]) -> Optional[str]: for a in aliases: for kk, vv in kv.items(): if _norm(a) == _norm(kk): return vv return None def extract_snapshot_from_tables(read: ReadResult) -> Tuple[DealSnapshot, bool]: """ Return snapshot + whether we successfully populated core fields from tables. """ snap = DealSnapshot() ok = False for tb in read.tables: ttype = classify_table(tb.rows) if ttype != "property_summary": continue kv = _table_to_kv(tb.rows) if not kv: continue address = _pick(kv, PROPERTY_KEYS["address"]) units = _pick(kv, PROPERTY_KEYS["number of units"]) year_built = _pick(kv, PROPERTY_KEYS["year built"]) total_size = _pick(kv, PROPERTY_KEYS["total size"]) unit_size_range = _pick(kv, PROPERTY_KEYS["unit size range"]) site_acreage = _pick(kv, PROPERTY_KEYS["site acreage"]) parcel_numbers = _pick(kv, PROPERTY_KEYS["parcel numbers"]) noi_current = _pick(kv, PROPERTY_KEYS["current noi"]) noi_pf = _pick(kv, PROPERTY_KEYS["pro forma noi"]) offering_price = _pick(kv, PROPERTY_KEYS["offering price"]) if address: snap.property.address = address # best-effort parse city/state/zip from "City, ST ZIP" m = re.search(r",\s*([A-Z]{2})\s*(\d{5}(?:-\d{4})?)", address) if m: snap.property.state = m.group(1) snap.property.zip = m.group(2) m2 = re.search(r"^(.*?),\s*[A-Z]{2}\s*\d{5}", address) if m2: snap.property.city = m2.group(1).split()[-1] if m2.group(1) else None # lightweight snap.property.units = units snap.property.year_built = year_built snap.property.building_sf = total_size snap.property.unit_size_range = unit_size_range snap.property.site_acreage = site_acreage snap.property.parcel_numbers = parcel_numbers snap.pricing.offering_price = moneyify(offering_price.replace(",", "")) if offering_price else None snap.pricing.noi_current = moneyify(noi_current) if noi_current else None snap.pricing.noi_proforma = moneyify(noi_pf) if noi_pf else None # set ok if we got the “big 3” if address and (offering_price or noi_current or noi_pf): ok = True break return snap, ok def extract_snapshot_from_text(text: str) -> DealSnapshot: t = clean_text(text) snap = DealSnapshot() # property name from first page title line snap.property.property_name = find_first( [ r"^\s*([A-Za-z0-9][A-Za-z0-9\-\s]+Apartments)\s*$", r"^\s*([A-Za-z0-9][A-Za-z0-9\-\s]+)\s*$", ], t, flags=re.IGNORECASE | re.MULTILINE, ) # fallback address snap.property.address = find_first( [r"\n(\d{2,6}\s+.+?\|\s*.+?,\s*[A-Z]{2}\s*\d{5}(?:-\d{4})?)\n"], t, flags=re.IGNORECASE, ) or find_first( [r"Address[:\s]+(.+?)(?:\n|$)"], t, flags=re.IGNORECASE, ) snap.property.units = find_first([r"(\d{1,4})\s+Units"], t) snap.property.year_built = find_first([r"Year Built[:\s]+(\d{4})"], t) snap.pricing.offering_price = find_first([r"Offering Price[:\s]+\$?([\d,]+)"], t) if snap.pricing.offering_price: snap.pricing.offering_price = moneyify(snap.pricing.offering_price) snap.pricing.noi_current = find_first([r"Net Operating Income\s*\$([\d,]+\.\d+|\d[\d,]+)"], t) if snap.pricing.noi_current: snap.pricing.noi_current = moneyify(snap.pricing.noi_current) return snap def extract_tables(read: ReadResult) -> List[ClassifiedTable]: out: List[ClassifiedTable] = [] for tb in read.tables: ttype = classify_table(tb.rows) records = rows_to_records(tb.rows) out.append(ClassifiedTable(page=tb.page, table_type=ttype, rows=records)) return out def extract_deal_snapshot(filename: str, reader: str, read: ReadResult) -> ExtractionResult: # 1) tables first snap, ok = extract_snapshot_from_tables(read) # 2) fallback to text if missing if not ok: snap2 = extract_snapshot_from_text(read.text or "") # fill missing only if not snap.property.property_name: snap.property.property_name = snap2.property.property_name if not snap.property.address: snap.property.address = snap2.property.address if not snap.property.units: snap.property.units = snap2.property.units if not snap.property.year_built: snap.property.year_built = snap2.property.year_built if not snap.pricing.offering_price: snap.pricing.offering_price = snap2.pricing.offering_price if not snap.pricing.noi_current: snap.pricing.noi_current = snap2.pricing.noi_current tables = extract_tables(read) num_pages = max([t.page for t in read.tables], default=0) wc = len((read.text or "").split()) return ExtractionResult( filename=filename, reader=reader, num_pages=num_pages, word_count=wc, snapshot=snap, tables=tables, )