# All comments in English. import pypandoc from bs4 import BeautifulSoup def normalize_newlines(value): if value is None: return "" return ( str(value) .replace("\r\n", "\\n") .replace("\n\r", "\\n") .replace("\r", "\\n") .replace("\n", "\\n") ) def convert_document(path): """ Converts any supported document to plain text using Pandoc. Supported formats: - doc - docx - odt - txt - md - html """ try: # Pandoc converts everything to plain text text = pypandoc.convert_file(path, "plain") except Exception as e: raise RuntimeError(f"Pandoc conversion failed: {str(e)}") # Clean HTML artifacts if any soup = BeautifulSoup(text, "html.parser") cleaned = soup.get_text() return normalize_newlines(cleaned)