import csv import io from pathlib import Path from docx import Document from pypdf import PdfReader SUPPORTED_EXTENSIONS = {".txt", ".md", ".json", ".yaml", ".yml", ".csv", ".pdf", ".docx"} def extract_text(filename: str, content: bytes, max_chars: int = 60_000) -> str: suffix = Path(filename).suffix.lower() if suffix not in SUPPORTED_EXTENSIONS: raise ValueError("Этот тип файла пока не поддерживается") if suffix in {".txt", ".md", ".json", ".yaml", ".yml"}: text = content.decode("utf-8", errors="replace") elif suffix == ".csv": rows = csv.reader(io.StringIO(content.decode("utf-8", errors="replace"))) text = "\n".join(" | ".join(row) for row in rows) elif suffix == ".pdf": text = "\n".join(page.extract_text() or "" for page in PdfReader(io.BytesIO(content)).pages) else: text = "\n".join(p.text for p in Document(io.BytesIO(content)).paragraphs) return text[:max_chars]