This commit is contained in:
2026-07-27 00:56:18 +03:30
commit 1c75afe0fd
244 changed files with 27710 additions and 0 deletions
+52
View File
@@ -0,0 +1,52 @@
from __future__ import annotations
from io import BytesIO
from pathlib import Path
MAX_DOCUMENT_BYTES = 5 * 1024 * 1024
class DocumentError(ValueError):
pass
def read_document(path: Path) -> str:
path = path.expanduser().resolve()
if not path.is_file():
raise DocumentError(f"Document does not exist: {path}")
if path.stat().st_size > MAX_DOCUMENT_BYTES:
raise DocumentError("Document is larger than the 5 MB safety limit.")
return read_document_bytes(path.name, path.read_bytes())
def read_document_bytes(filename: str, data: bytes) -> str:
if len(data) > MAX_DOCUMENT_BYTES:
raise DocumentError("Document is larger than the 5 MB safety limit.")
suffix = Path(filename).suffix.lower()
if suffix in {".txt", ".md", ".json"}:
try:
text = data.decode("utf-8")
except UnicodeDecodeError as exc:
raise DocumentError(f"{filename} is not valid UTF-8 text.") from exc
elif suffix == ".pdf":
from pypdf import PdfReader
text = "\n".join(page.extract_text() or "" for page in PdfReader(BytesIO(data)).pages)
elif suffix == ".docx":
from docx import Document
document = Document(BytesIO(data))
parts = [paragraph.text for paragraph in document.paragraphs]
for table in document.tables:
for row in table.rows:
parts.append(" | ".join(cell.text for cell in row.cells))
text = "\n".join(parts)
else:
raise DocumentError("Supported resume formats: .pdf, .docx, .txt, .md, and .json")
text = text.strip()
if not text:
raise DocumentError(f"No readable text was found in {filename}.")
return text