INNER CODE UNIT · Python
get_title_from_html
cognitivetech/ollama-ebook-summary · book2text.py:74
def get_title_from_html(filepath):
try:
with open(filepath, 'r', encoding='utf-8') as file:
soup = BeautifulSoup(file, 'html.parser')
# Try to get the title from the <title> tag
title_tag = soup.find('title')
if title_tag and title_tag.string:
return title_tag.string.strip()
# If no title tag, try to get the first <h1> tag
h1_tag = soup.find('h1')
if h1_tag and h1_tag.string:
return h1_tag.string.strip()
except Exception as e:
print(f"Error reading HTML file: {e}")
# If no title found in HTML, use the filename as backup