INNER CODE UNIT · Python
epub_to_text
cognitivetech/ollama-ebook-summary · book2text.py:94
def epub_to_text(epub_path):
book = epub.read_epub(epub_path)
text_content = []
for item in book.get_items():
if item.get_type() == ebooklib.ITEM_DOCUMENT:
# Parse the HTML content using BeautifulSoup
soup = BeautifulSoup(item.get_content(), 'html.parser')
# Extract just the text, removing HTML tags
chapter_text = soup.get_text(separator=' ', strip=True)
text_content.append(chapter_text)
return '\n'.join(text_content)
def html_to_text(html_path):
with open(html_path, 'r', encoding='utf-8') as file:
soup = BeautifulSoup(file, 'html.parser')
return soup.get_text()
def pdf_to_text(pdf_path):