""" document_processor.py - Text Extraction & Paragraph Segmentation Handles reading .docx, .txt, and .pdf files and splitting them into paragraph-level segments for analysis. """ import re from pathlib import Path class DocumentProcessor: """Extract text from various document formats.""" @staticmethod def read_docx(filepath: str) -> str: from docx import Document doc = Document(filepath) paragraphs = [p.text.strip() for p in doc.paragraphs if p.text.strip()] return "\n\n".join(paragraphs) @staticmethod def read_txt(filepath: str) -> str: with open(filepath, "r", encoding="utf-8") as f: return f.read().strip() @staticmethod def read_pdf(filepath: str) -> str: import fitz # PyMuPDF doc = fitz.open(filepath) text_parts = [] for page in doc: text_parts.append(page.get_text()) doc.close() return "\n\n".join(text_parts).strip() def process(self, filepath: str) -> str: path = Path(filepath) ext = path.suffix.lower() if ext == ".docx": return self.read_docx(filepath) elif ext == ".txt": return self.read_txt(filepath) elif ext == ".pdf": return self.read_pdf(filepath) else: raise ValueError(f"Unsupported file type: {ext}") class StreamlitFileProcessor: """Process files uploaded through Streamlit's file_uploader.""" def process_uploaded_file(self, uploaded_file) -> str: import tempfile, os suffix = Path(uploaded_file.name).suffix with tempfile.NamedTemporaryFile(delete=False, suffix=suffix) as tmp: tmp.write(uploaded_file.getvalue()) tmp_path = tmp.name try: processor = DocumentProcessor() return processor.process(tmp_path) finally: os.unlink(tmp_path) def segment_paragraphs(text: str) -> list[str]: """ Split a personal statement into paragraph-level segments. Rules: - Split on double newlines (standard paragraph breaks). - Merge very short fragments (< 40 words) into the previous segment, since they are usually transitions or sentence fragments. - Strip any segment that is only whitespace. """ raw = re.split(r"\n\s*\n", text.strip()) raw = [p.strip() for p in raw if p.strip()] if not raw: return [] # First pass: collect paragraphs, merging very short ones segments: list[str] = [] for para in raw: word_count = len(para.split()) if word_count < 40: if segments: # Merge short fragment into previous segment segments[-1] = segments[-1] + " " + para else: # First segment is short (e.g. a title/name); hold it and # merge forward into the next real paragraph. segments.append(para) else: if segments and len(segments[-1].split()) < 40: # Previous segment was a short stub; merge it forward segments[-1] = segments[-1] + " " + para else: segments.append(para) return segments