crawler-intra-mart/extractor/html_cleaner.py
Do Duy eb46e739e7 Add intra-mart documentation crawler
Playwright-based crawler that renders pages from api.intra-mart.jp and
document.intra-mart.jp, extracts main content, and converts to Markdown
via Pandoc. SQLite-backed queue drives a resumable sequential pipeline
across crawler, extractor, converter, and storage modules.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-20 16:29:54 +07:00

227 lines
4.5 KiB
Python

from bs4 import BeautifulSoup
from utils.logger import logger
def _ensure_soup(html_or_soup):
"""
Accept either a BeautifulSoup/Tag or an HTML string and
return a BeautifulSoup-like object. Cheap when input is
already parsed (this is the whole point — avoid reparsing).
"""
if hasattr(html_or_soup, "select_one") and hasattr(
html_or_soup, "find_all"
):
return html_or_soup
return BeautifulSoup(html_or_soup, "lxml")
class HTMLCleaner:
"""
Clean extracted HTML
before markdown conversion.
"""
def __init__(
self,
remove_selectors
):
self.remove_selectors = remove_selectors
def clean(
self,
html
):
"""
Clean HTML in place when given a soup; otherwise parse
the string, clean, and return the string. Returning
the same type the caller passed lets process_page hold
a soup across the whole pipeline without re-parsing.
"""
return_string = isinstance(html, (str, bytes))
soup = _ensure_soup(html)
self.remove_unwanted_nodes(soup)
self.normalize_code_blocks(soup)
self.normalize_tables(soup)
self.fix_relative_links(soup)
self.remove_empty_tags(soup)
self.unwrap_redundant_tags(soup)
return str(soup) if return_string else soup
def remove_unwanted_nodes(
self,
soup
):
"""
Remove navigation/UI noise.
"""
for selector in self.remove_selectors:
try:
for tag in soup.select(selector):
tag.decompose()
except Exception as e:
logger.warning(
f"Remove selector failed "
f"{selector}: {e}"
)
additional_noise = [
".sidebar",
".toc",
".breadcrumb",
".pagination",
".edit-page",
".theme-doc-version-badge",
".header",
".footer",
".ads",
".search",
".search-box",
".navigation"
]
for selector in additional_noise:
for tag in soup.select(selector):
tag.decompose()
def normalize_code_blocks(
self,
soup
):
"""
Normalize code blocks
for better markdown conversion.
"""
for pre in soup.find_all("pre"):
code = pre.find("code")
if not code:
continue
classes = code.get("class", [])
language = None
for cls in classes:
if cls.startswith("language-"):
language = cls.replace(
"language-",
""
)
if language:
pre["data-language"] = language
def normalize_tables(
self,
soup
):
"""
Improve HTML table structure.
"""
for table in soup.find_all("table"):
if not table.find("thead"):
first_row = table.find("tr")
if first_row:
thead = soup.new_tag(
"thead"
)
first_row.extract()
thead.append(first_row)
table.insert(0, thead)
def fix_relative_links(
self,
soup
):
"""
Remove javascript links.
"""
for a in soup.find_all(
"a",
href=True
):
href = a["href"].strip()
if href.startswith(
"javascript:"
):
del a["href"]
def remove_empty_tags(
self,
soup
):
"""
Remove empty useless tags.
"""
removable = [
"div",
"span",
"p"
]
for tag_name in removable:
for tag in soup.find_all(tag_name):
if tag.get_text(strip=True):
continue
if tag.find():
continue
tag.decompose()
def unwrap_redundant_tags(
self,
soup
):
"""
Flatten unnecessary wrappers.
"""
unwrap_tags = [
"span"
]
for tag_name in unwrap_tags:
for tag in soup.find_all(tag_name):
if tag.attrs:
continue
tag.unwrap()