"""File parsing for uploaded documents (pdf / markdown / txt).""" from __future__ import annotations from pathlib import Path class ParseError(Exception): pass def parse_pdf(path: Path) -> str: import fitz # PyMuPDF try: doc = fitz.open(path) except Exception as exc: raise ParseError(f"cannot open PDF: {exc}") from exc parts = [] for page in doc: parts.append(page.get_text()) doc.close() return "\n".join(parts) def parse_text(path: Path) -> str: import chardet raw = path.read_bytes() # Try utf-8 first, else detect encoding try: return raw.decode("utf-8") except UnicodeDecodeError: pass guess = chardet.detect(raw) enc = guess.get("encoding") or "utf-8" try: return raw.decode(enc, errors="replace") except Exception: return raw.decode("utf-8", errors="replace") def parse_document(path: Path) -> str: ext = path.suffix.lower().lstrip(".") if ext == "pdf": return parse_pdf(path) return parse_text(path)