25 lines
992 B
Python
25 lines
992 B
Python
import csv
|
|
import io
|
|
from pathlib import Path
|
|
|
|
from docx import Document
|
|
from pypdf import PdfReader
|
|
|
|
SUPPORTED_EXTENSIONS = {".txt", ".md", ".json", ".yaml", ".yml", ".csv", ".pdf", ".docx"}
|
|
|
|
|
|
def extract_text(filename: str, content: bytes, max_chars: int = 60_000) -> str:
|
|
suffix = Path(filename).suffix.lower()
|
|
if suffix not in SUPPORTED_EXTENSIONS:
|
|
raise ValueError("Этот тип файла пока не поддерживается")
|
|
if suffix in {".txt", ".md", ".json", ".yaml", ".yml"}:
|
|
text = content.decode("utf-8", errors="replace")
|
|
elif suffix == ".csv":
|
|
rows = csv.reader(io.StringIO(content.decode("utf-8", errors="replace")))
|
|
text = "\n".join(" | ".join(row) for row in rows)
|
|
elif suffix == ".pdf":
|
|
text = "\n".join(page.extract_text() or "" for page in PdfReader(io.BytesIO(content)).pages)
|
|
else:
|
|
text = "\n".join(p.text for p in Document(io.BytesIO(content)).paragraphs)
|
|
return text[:max_chars]
|