Spaces:
Runtime error
Runtime error
Download src/extractors.py from SarahXia0405/om-extractor-v1: direct link, hf CLI and curl.
- Browser
- Download file 6.94 kB
-
https://huggingface.co/spaces/SarahXia0405/om-extractor-v1/resolve/main/src/extractors.py
- Command line
-
hf download hf://spaces/SarahXia0405/om-extractor-v1/src/extractors.py
-
curl -L -o extractors.py https://huggingface.co/spaces/SarahXia0405/om-extractor-v1/resolve/main/src/extractors.py
6.94 kB
| from __future__ import annotations | |
| from typing import Dict, List, Tuple, Optional | |
| import re | |
| from schema import ExtractionResult, DealSnapshot, ClassifiedTable, ReadResult | |
| from normalize import clean_text, find_first, moneyify | |
| from table_classifier import classify_table | |
| from utils import rows_to_records | |
| PROPERTY_KEYS = { | |
| "address": ["address"], | |
| "number of units": ["number of units", "units"], | |
| "year built": ["year built"], | |
| "total size": ["total size", "size", "building sf", "gla"], | |
| "unit size range": ["unit size range"], | |
| "site acreage": ["site acreage", "site area"], | |
| "parcel numbers": ["parcel numbers", "parcel"], | |
| "current noi": ["current noi"], | |
| "pro forma noi": ["pro forma noi", "proforma noi"], | |
| "offering price": ["offering price", "purchase price", "offering"], | |
| } | |
| def _norm(s: str) -> str: | |
| return re.sub(r"\s+", " ", (s or "").strip().lower()) | |
| def _table_to_kv(rows: List[List[str]]) -> Dict[str, str]: | |
| """ | |
| Works for: | |
| - two-column tables: key | value | |
| - vertical key list then vertical value list (less common) | |
| """ | |
| kv: Dict[str, str] = {} | |
| # common: 2 columns | |
| for r in rows: | |
| if len(r) >= 2: | |
| k = _norm(r[0]) | |
| v = (r[1] or "").strip() | |
| if k and v: | |
| kv[k] = v | |
| # sometimes extracted as one row with keys + one row with values | |
| if not kv and len(rows) >= 2 and len(rows[0]) == len(rows[1]): | |
| keys = [_norm(x) for x in rows[0]] | |
| vals = [x.strip() for x in rows[1]] | |
| for k, v in zip(keys, vals): | |
| if k and v: | |
| kv[k] = v | |
| return kv | |
| def _pick(kv: Dict[str, str], aliases: List[str]) -> Optional[str]: | |
| for a in aliases: | |
| for kk, vv in kv.items(): | |
| if _norm(a) == _norm(kk): | |
| return vv | |
| return None | |
| def extract_snapshot_from_tables(read: ReadResult) -> Tuple[DealSnapshot, bool]: | |
| """ | |
| Return snapshot + whether we successfully populated core fields from tables. | |
| """ | |
| snap = DealSnapshot() | |
| ok = False | |
| for tb in read.tables: | |
| ttype = classify_table(tb.rows) | |
| if ttype != "property_summary": | |
| continue | |
| kv = _table_to_kv(tb.rows) | |
| if not kv: | |
| continue | |
| address = _pick(kv, PROPERTY_KEYS["address"]) | |
| units = _pick(kv, PROPERTY_KEYS["number of units"]) | |
| year_built = _pick(kv, PROPERTY_KEYS["year built"]) | |
| total_size = _pick(kv, PROPERTY_KEYS["total size"]) | |
| unit_size_range = _pick(kv, PROPERTY_KEYS["unit size range"]) | |
| site_acreage = _pick(kv, PROPERTY_KEYS["site acreage"]) | |
| parcel_numbers = _pick(kv, PROPERTY_KEYS["parcel numbers"]) | |
| noi_current = _pick(kv, PROPERTY_KEYS["current noi"]) | |
| noi_pf = _pick(kv, PROPERTY_KEYS["pro forma noi"]) | |
| offering_price = _pick(kv, PROPERTY_KEYS["offering price"]) | |
| if address: | |
| snap.property.address = address | |
| # best-effort parse city/state/zip from "City, ST ZIP" | |
| m = re.search(r",\s*([A-Z]{2})\s*(\d{5}(?:-\d{4})?)", address) | |
| if m: | |
| snap.property.state = m.group(1) | |
| snap.property.zip = m.group(2) | |
| m2 = re.search(r"^(.*?),\s*[A-Z]{2}\s*\d{5}", address) | |
| if m2: | |
| snap.property.city = m2.group(1).split()[-1] if m2.group(1) else None # lightweight | |
| snap.property.units = units | |
| snap.property.year_built = year_built | |
| snap.property.building_sf = total_size | |
| snap.property.unit_size_range = unit_size_range | |
| snap.property.site_acreage = site_acreage | |
| snap.property.parcel_numbers = parcel_numbers | |
| snap.pricing.offering_price = moneyify(offering_price.replace(",", "")) if offering_price else None | |
| snap.pricing.noi_current = moneyify(noi_current) if noi_current else None | |
| snap.pricing.noi_proforma = moneyify(noi_pf) if noi_pf else None | |
| # set ok if we got the “big 3” | |
| if address and (offering_price or noi_current or noi_pf): | |
| ok = True | |
| break | |
| return snap, ok | |
| def extract_snapshot_from_text(text: str) -> DealSnapshot: | |
| t = clean_text(text) | |
| snap = DealSnapshot() | |
| # property name from first page title line | |
| snap.property.property_name = find_first( | |
| [ | |
| r"^\s*([A-Za-z0-9][A-Za-z0-9\-\s]+Apartments)\s*$", | |
| r"^\s*([A-Za-z0-9][A-Za-z0-9\-\s]+)\s*$", | |
| ], | |
| t, | |
| flags=re.IGNORECASE | re.MULTILINE, | |
| ) | |
| # fallback address | |
| snap.property.address = find_first( | |
| [r"\n(\d{2,6}\s+.+?\|\s*.+?,\s*[A-Z]{2}\s*\d{5}(?:-\d{4})?)\n"], | |
| t, | |
| flags=re.IGNORECASE, | |
| ) or find_first( | |
| [r"Address[:\s]+(.+?)(?:\n|$)"], | |
| t, | |
| flags=re.IGNORECASE, | |
| ) | |
| snap.property.units = find_first([r"(\d{1,4})\s+Units"], t) | |
| snap.property.year_built = find_first([r"Year Built[:\s]+(\d{4})"], t) | |
| snap.pricing.offering_price = find_first([r"Offering Price[:\s]+\$?([\d,]+)"], t) | |
| if snap.pricing.offering_price: | |
| snap.pricing.offering_price = moneyify(snap.pricing.offering_price) | |
| snap.pricing.noi_current = find_first([r"Net Operating Income\s*\$([\d,]+\.\d+|\d[\d,]+)"], t) | |
| if snap.pricing.noi_current: | |
| snap.pricing.noi_current = moneyify(snap.pricing.noi_current) | |
| return snap | |
| def extract_tables(read: ReadResult) -> List[ClassifiedTable]: | |
| out: List[ClassifiedTable] = [] | |
| for tb in read.tables: | |
| ttype = classify_table(tb.rows) | |
| records = rows_to_records(tb.rows) | |
| out.append(ClassifiedTable(page=tb.page, table_type=ttype, rows=records)) | |
| return out | |
| def extract_deal_snapshot(filename: str, reader: str, read: ReadResult) -> ExtractionResult: | |
| # 1) tables first | |
| snap, ok = extract_snapshot_from_tables(read) | |
| # 2) fallback to text if missing | |
| if not ok: | |
| snap2 = extract_snapshot_from_text(read.text or "") | |
| # fill missing only | |
| if not snap.property.property_name: | |
| snap.property.property_name = snap2.property.property_name | |
| if not snap.property.address: | |
| snap.property.address = snap2.property.address | |
| if not snap.property.units: | |
| snap.property.units = snap2.property.units | |
| if not snap.property.year_built: | |
| snap.property.year_built = snap2.property.year_built | |
| if not snap.pricing.offering_price: | |
| snap.pricing.offering_price = snap2.pricing.offering_price | |
| if not snap.pricing.noi_current: | |
| snap.pricing.noi_current = snap2.pricing.noi_current | |
| tables = extract_tables(read) | |
| num_pages = max([t.page for t in read.tables], default=0) | |
| wc = len((read.text or "").split()) | |
| return ExtractionResult( | |
| filename=filename, | |
| reader=reader, | |
| num_pages=num_pages, | |
| word_count=wc, | |
| snapshot=snap, | |
| tables=tables, | |
| ) | |