init
This commit is contained in:
@@ -0,0 +1,52 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
MAX_DOCUMENT_BYTES = 5 * 1024 * 1024
|
||||
|
||||
|
||||
class DocumentError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def read_document(path: Path) -> str:
|
||||
path = path.expanduser().resolve()
|
||||
if not path.is_file():
|
||||
raise DocumentError(f"Document does not exist: {path}")
|
||||
if path.stat().st_size > MAX_DOCUMENT_BYTES:
|
||||
raise DocumentError("Document is larger than the 5 MB safety limit.")
|
||||
|
||||
return read_document_bytes(path.name, path.read_bytes())
|
||||
|
||||
|
||||
def read_document_bytes(filename: str, data: bytes) -> str:
|
||||
if len(data) > MAX_DOCUMENT_BYTES:
|
||||
raise DocumentError("Document is larger than the 5 MB safety limit.")
|
||||
|
||||
suffix = Path(filename).suffix.lower()
|
||||
if suffix in {".txt", ".md", ".json"}:
|
||||
try:
|
||||
text = data.decode("utf-8")
|
||||
except UnicodeDecodeError as exc:
|
||||
raise DocumentError(f"{filename} is not valid UTF-8 text.") from exc
|
||||
elif suffix == ".pdf":
|
||||
from pypdf import PdfReader
|
||||
|
||||
text = "\n".join(page.extract_text() or "" for page in PdfReader(BytesIO(data)).pages)
|
||||
elif suffix == ".docx":
|
||||
from docx import Document
|
||||
|
||||
document = Document(BytesIO(data))
|
||||
parts = [paragraph.text for paragraph in document.paragraphs]
|
||||
for table in document.tables:
|
||||
for row in table.rows:
|
||||
parts.append(" | ".join(cell.text for cell in row.cells))
|
||||
text = "\n".join(parts)
|
||||
else:
|
||||
raise DocumentError("Supported resume formats: .pdf, .docx, .txt, .md, and .json")
|
||||
|
||||
text = text.strip()
|
||||
if not text:
|
||||
raise DocumentError(f"No readable text was found in {filename}.")
|
||||
return text
|
||||
Reference in New Issue
Block a user