om-extractor-v1 / src /extractors.py
SarahXia0405's picture
Update src/extractors.py
20bc4ff verified
Raw History Blame Contribute Delete
6.94 kB
from __future__ import annotations
from typing import Dict, List, Tuple, Optional
import re
from schema import ExtractionResult, DealSnapshot, ClassifiedTable, ReadResult
from normalize import clean_text, find_first, moneyify
from table_classifier import classify_table
from utils import rows_to_records
PROPERTY_KEYS = {
"address": ["address"],
"number of units": ["number of units", "units"],
"year built": ["year built"],
"total size": ["total size", "size", "building sf", "gla"],
"unit size range": ["unit size range"],
"site acreage": ["site acreage", "site area"],
"parcel numbers": ["parcel numbers", "parcel"],
"current noi": ["current noi"],
"pro forma noi": ["pro forma noi", "proforma noi"],
"offering price": ["offering price", "purchase price", "offering"],
}
def _norm(s: str) -> str:
return re.sub(r"\s+", " ", (s or "").strip().lower())
def _table_to_kv(rows: List[List[str]]) -> Dict[str, str]:
"""
Works for:
- two-column tables: key | value
- vertical key list then vertical value list (less common)
"""
kv: Dict[str, str] = {}
# common: 2 columns
for r in rows:
if len(r) >= 2:
k = _norm(r[0])
v = (r[1] or "").strip()
if k and v:
kv[k] = v
# sometimes extracted as one row with keys + one row with values
if not kv and len(rows) >= 2 and len(rows[0]) == len(rows[1]):
keys = [_norm(x) for x in rows[0]]
vals = [x.strip() for x in rows[1]]
for k, v in zip(keys, vals):
if k and v:
kv[k] = v
return kv
def _pick(kv: Dict[str, str], aliases: List[str]) -> Optional[str]:
for a in aliases:
for kk, vv in kv.items():
if _norm(a) == _norm(kk):
return vv
return None
def extract_snapshot_from_tables(read: ReadResult) -> Tuple[DealSnapshot, bool]:
"""
Return snapshot + whether we successfully populated core fields from tables.
"""
snap = DealSnapshot()
ok = False
for tb in read.tables:
ttype = classify_table(tb.rows)
if ttype != "property_summary":
continue
kv = _table_to_kv(tb.rows)
if not kv:
continue
address = _pick(kv, PROPERTY_KEYS["address"])
units = _pick(kv, PROPERTY_KEYS["number of units"])
year_built = _pick(kv, PROPERTY_KEYS["year built"])
total_size = _pick(kv, PROPERTY_KEYS["total size"])
unit_size_range = _pick(kv, PROPERTY_KEYS["unit size range"])
site_acreage = _pick(kv, PROPERTY_KEYS["site acreage"])
parcel_numbers = _pick(kv, PROPERTY_KEYS["parcel numbers"])
noi_current = _pick(kv, PROPERTY_KEYS["current noi"])
noi_pf = _pick(kv, PROPERTY_KEYS["pro forma noi"])
offering_price = _pick(kv, PROPERTY_KEYS["offering price"])
if address:
snap.property.address = address
# best-effort parse city/state/zip from "City, ST ZIP"
m = re.search(r",\s*([A-Z]{2})\s*(\d{5}(?:-\d{4})?)", address)
if m:
snap.property.state = m.group(1)
snap.property.zip = m.group(2)
m2 = re.search(r"^(.*?),\s*[A-Z]{2}\s*\d{5}", address)
if m2:
snap.property.city = m2.group(1).split()[-1] if m2.group(1) else None # lightweight
snap.property.units = units
snap.property.year_built = year_built
snap.property.building_sf = total_size
snap.property.unit_size_range = unit_size_range
snap.property.site_acreage = site_acreage
snap.property.parcel_numbers = parcel_numbers
snap.pricing.offering_price = moneyify(offering_price.replace(",", "")) if offering_price else None
snap.pricing.noi_current = moneyify(noi_current) if noi_current else None
snap.pricing.noi_proforma = moneyify(noi_pf) if noi_pf else None
# set ok if we got the “big 3”
if address and (offering_price or noi_current or noi_pf):
ok = True
break
return snap, ok
def extract_snapshot_from_text(text: str) -> DealSnapshot:
t = clean_text(text)
snap = DealSnapshot()
# property name from first page title line
snap.property.property_name = find_first(
[
r"^\s*([A-Za-z0-9][A-Za-z0-9\-\s]+Apartments)\s*$",
r"^\s*([A-Za-z0-9][A-Za-z0-9\-\s]+)\s*$",
],
t,
flags=re.IGNORECASE | re.MULTILINE,
)
# fallback address
snap.property.address = find_first(
[r"\n(\d{2,6}\s+.+?\|\s*.+?,\s*[A-Z]{2}\s*\d{5}(?:-\d{4})?)\n"],
t,
flags=re.IGNORECASE,
) or find_first(
[r"Address[:\s]+(.+?)(?:\n|$)"],
t,
flags=re.IGNORECASE,
)
snap.property.units = find_first([r"(\d{1,4})\s+Units"], t)
snap.property.year_built = find_first([r"Year Built[:\s]+(\d{4})"], t)
snap.pricing.offering_price = find_first([r"Offering Price[:\s]+\$?([\d,]+)"], t)
if snap.pricing.offering_price:
snap.pricing.offering_price = moneyify(snap.pricing.offering_price)
snap.pricing.noi_current = find_first([r"Net Operating Income\s*\$([\d,]+\.\d+|\d[\d,]+)"], t)
if snap.pricing.noi_current:
snap.pricing.noi_current = moneyify(snap.pricing.noi_current)
return snap
def extract_tables(read: ReadResult) -> List[ClassifiedTable]:
out: List[ClassifiedTable] = []
for tb in read.tables:
ttype = classify_table(tb.rows)
records = rows_to_records(tb.rows)
out.append(ClassifiedTable(page=tb.page, table_type=ttype, rows=records))
return out
def extract_deal_snapshot(filename: str, reader: str, read: ReadResult) -> ExtractionResult:
# 1) tables first
snap, ok = extract_snapshot_from_tables(read)
# 2) fallback to text if missing
if not ok:
snap2 = extract_snapshot_from_text(read.text or "")
# fill missing only
if not snap.property.property_name:
snap.property.property_name = snap2.property.property_name
if not snap.property.address:
snap.property.address = snap2.property.address
if not snap.property.units:
snap.property.units = snap2.property.units
if not snap.property.year_built:
snap.property.year_built = snap2.property.year_built
if not snap.pricing.offering_price:
snap.pricing.offering_price = snap2.pricing.offering_price
if not snap.pricing.noi_current:
snap.pricing.noi_current = snap2.pricing.noi_current
tables = extract_tables(read)
num_pages = max([t.page for t in read.tables], default=0)
wc = len((read.text or "").split())
return ExtractionResult(
filename=filename,
reader=reader,
num_pages=num_pages,
word_count=wc,
snapshot=snap,
tables=tables,
)