INNER CODE UNIT · Python
clean_html
essamamdani/search-result-scraper-markdown · main.py:146
def clean_html(html):
soup = BeautifulSoup(html, 'html.parser')
# Remove all script, style, and other unnecessary elements
for script_or_style in soup(["script", "style", "header", "footer", "noscript", "form", "input", "textarea", "select", "option", "button", "svg", "iframe", "object", "embed", "applet", "nav", "navbar"]):
script_or_style.decompose()
# remove ids "layers"
ids = ['layers']
for id_ in ids:
tag = soup.find(id=id_)
if tag:
tag.decompose()
# Remove unwanted classes and ids
for tag in soup.find_all(True):
tag.attrs = {key: value for key, value in tag.attrs.items() if key not in ['class', 'id', 'style']}