Playwright-based crawler that renders pages from api.intra-mart.jp and document.intra-mart.jp, extracts main content, and converts to Markdown via Pandoc. SQLite-backed queue drives a resumable sequential pipeline across crawler, extractor, converter, and storage modules. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
815 lines
23 KiB
Python
815 lines
23 KiB
Python
import asyncio
|
|
from pathlib import Path
|
|
|
|
from bs4 import BeautifulSoup
|
|
|
|
from crawler.http_client import HTTPClient
|
|
from crawler.playwright_client import PlaywrightClient
|
|
from crawler.url_manager import URLManager
|
|
|
|
from converter.markdown_converter import MarkdownConverter
|
|
|
|
from extractor.html_cleaner import HTMLCleaner
|
|
from extractor.html_extractor import HTMLExtractor
|
|
|
|
from models.crawl_result import CrawlResult
|
|
|
|
from storage.file_writer import FileWriter
|
|
from storage.metadata_db import MetadataDB
|
|
from storage.path_mapper import PathMapper
|
|
|
|
from utils.logger import logger
|
|
|
|
from extractor.asset_downloader import AssetDownloader
|
|
from crawler.sphinx_discovery import SphinxDiscovery
|
|
from crawler.navigation_parser import NavigationParser
|
|
|
|
START_URLS = [
|
|
"https://api.intra-mart.jp/iap/",
|
|
"https://document.intra-mart.jp/library/"
|
|
]
|
|
|
|
|
|
OUTPUT_DIR = "output"
|
|
|
|
|
|
ALLOWED_DOMAINS = [
|
|
"api.intra-mart.jp",
|
|
"document.intra-mart.jp"
|
|
]
|
|
|
|
|
|
REMOVE_SELECTORS = [
|
|
"script",
|
|
"style",
|
|
"noscript",
|
|
|
|
".sidebar-ad",
|
|
|
|
".navigation-footer",
|
|
|
|
".breadcrumb",
|
|
|
|
".toc",
|
|
|
|
".page-footer",
|
|
|
|
".header",
|
|
|
|
".search",
|
|
|
|
".ads"
|
|
]
|
|
|
|
|
|
CONTENT_SELECTORS = [
|
|
"main",
|
|
|
|
"article",
|
|
|
|
".content",
|
|
|
|
".document",
|
|
|
|
".markdown-body",
|
|
|
|
".theme-doc-markdown",
|
|
|
|
"#content",
|
|
|
|
# Javadoc (api.intra-mart.jp/iap/javadoc/...) — static
|
|
# pages, by far the bulk of the queue. Matching this
|
|
# selector lets the HTTP fast-path skip Playwright.
|
|
"div.contentContainer",
|
|
|
|
# apilist-ssjs (api.intra-mart.jp/iap/apilist-ssjs/...)
|
|
# wraps its body content in <div class="main">. Use a
|
|
# body-scoped selector so we don't match `.main`
|
|
# elements nested inside navigation widgets.
|
|
"body > div.main",
|
|
|
|
# document.intra-mart.jp/library/... — most user-guide
|
|
# pages put their content in a single top-level
|
|
# <div class="container">.
|
|
"body > div.container",
|
|
]
|
|
|
|
|
|
MAX_PAGES_PER_RUN = 100000
|
|
MAX_ATTEMPTS = 3
|
|
BROWSER_RECYCLE_EVERY = 500
|
|
|
|
|
|
class _PermanentFetchFailure(Exception):
|
|
"""Raised when re-fetching the URL cannot succeed (4xx)."""
|
|
|
|
# Number of concurrent page workers. Each worker shares the
|
|
# single PlaywrightContext + single httpx client; only DB
|
|
# writes serialize through self.db_lock. The bottleneck is
|
|
# network I/O, so this is the multiplier on throughput.
|
|
CONCURRENCY = 8
|
|
|
|
class DocumentationCrawler:
|
|
"""
|
|
Main crawler orchestrator.
|
|
"""
|
|
|
|
def __init__(self):
|
|
self.client = PlaywrightClient(
|
|
headless=True,
|
|
content_selectors=CONTENT_SELECTORS,
|
|
)
|
|
|
|
self.http_client = HTTPClient()
|
|
|
|
self.url_manager = URLManager(
|
|
allowed_domains=ALLOWED_DOMAINS
|
|
)
|
|
|
|
self.sphinx_discovery = SphinxDiscovery(
|
|
url_manager=self.url_manager
|
|
)
|
|
|
|
self.navigation_parser = NavigationParser(
|
|
url_manager=self.url_manager
|
|
)
|
|
|
|
self.navigation_trees = []
|
|
|
|
self.cleaner = HTMLCleaner(
|
|
REMOVE_SELECTORS
|
|
)
|
|
|
|
self.extractor = HTMLExtractor(
|
|
CONTENT_SELECTORS
|
|
)
|
|
|
|
self.asset_downloader = AssetDownloader(
|
|
output_dir=str(Path(OUTPUT_DIR) / "assets"),
|
|
http_client=self.http_client.client,
|
|
)
|
|
|
|
self.converter = MarkdownConverter()
|
|
|
|
self.writer = FileWriter()
|
|
|
|
self.db = MetadataDB()
|
|
|
|
self.path_mapper = PathMapper(
|
|
Path(OUTPUT_DIR) / "markdown"
|
|
)
|
|
|
|
self.processed_this_run = 0
|
|
|
|
# Single lock serializing DB write blocks across
|
|
# concurrent workers. SQLite + Python's sqlite3 module
|
|
# already mutex on the connection, but a transaction()
|
|
# context spans many statements — without this lock,
|
|
# two workers could interleave statements mid-tx.
|
|
self.db_lock = asyncio.Lock()
|
|
|
|
# In-flight worker count for clean shutdown: a worker
|
|
# that sees an empty queue must wait until all peers
|
|
# finish (their process_page may enqueue new URLs).
|
|
self._in_flight = 0
|
|
self._stop = False
|
|
|
|
# Track domains we've already navigation-parsed.
|
|
# Sidebars are site-wide on Sphinx/Javadoc, so the same
|
|
# tree is rediscovered on every page — running the
|
|
# heavy parser repeatedly wastes CPU. Any per-section
|
|
# URLs we miss here will still be picked up by
|
|
# extract_links (it iterates all <a href>).
|
|
self._nav_parsed_domains = set()
|
|
|
|
async def initialize(self):
|
|
"""
|
|
Initialize crawler resources.
|
|
"""
|
|
|
|
logger.info(
|
|
"Initializing crawler..."
|
|
)
|
|
|
|
self.db.reset_stuck_processing()
|
|
|
|
await self.client.start()
|
|
|
|
for url in START_URLS:
|
|
normalized = self.url_manager.normalize(url)
|
|
|
|
if normalized and self.url_manager.is_allowed(normalized):
|
|
self.db.enqueue_url(
|
|
normalized,
|
|
parent_url=None,
|
|
depth=0,
|
|
priority=100
|
|
)
|
|
|
|
for url in START_URLS:
|
|
discovered = await self.sphinx_discovery.discover_from_base(url)
|
|
|
|
for discovered_url in discovered:
|
|
if not self.url_manager.should_enqueue(discovered_url):
|
|
continue
|
|
|
|
if self.db.is_done(discovered_url):
|
|
continue
|
|
|
|
if self.db.is_known(discovered_url):
|
|
continue
|
|
|
|
self.db.enqueue_url(
|
|
discovered_url,
|
|
parent_url=url,
|
|
depth=1,
|
|
priority=50
|
|
)
|
|
|
|
self.db.save_discovery_source(
|
|
url=discovered_url,
|
|
discovery_type="sphinx_base_discovery",
|
|
source_url=url
|
|
)
|
|
logger.info(f"Initial queue stats: {self.db.queue_stats()}")
|
|
|
|
async def shutdown(self):
|
|
"""
|
|
Cleanup resources.
|
|
"""
|
|
|
|
logger.info(
|
|
"Shutting down crawler..."
|
|
)
|
|
|
|
self.export_summary()
|
|
|
|
await self.client.stop()
|
|
|
|
await self.http_client.aclose()
|
|
|
|
self.db.close()
|
|
|
|
logger.info(
|
|
"Crawler shutdown complete."
|
|
)
|
|
|
|
async def crawl(self):
|
|
"""
|
|
Main crawl entry point. Spawns CONCURRENCY workers that
|
|
share the single browser context + httpx client and
|
|
serialize through self.db_lock for DB writes.
|
|
"""
|
|
|
|
await self.initialize()
|
|
|
|
try:
|
|
workers = [
|
|
asyncio.create_task(self._worker(i))
|
|
for i in range(CONCURRENCY)
|
|
]
|
|
await asyncio.gather(*workers)
|
|
finally:
|
|
await self.shutdown()
|
|
|
|
async def _claim_next_url(self):
|
|
"""
|
|
Atomically pick the next pending URL and mark it
|
|
processing. Returns None when the queue is empty.
|
|
"""
|
|
async with self.db_lock:
|
|
item = self.db.get_next_pending_url()
|
|
if not item:
|
|
return None
|
|
url = item["url"]
|
|
if self.db.is_done(url):
|
|
# Stale row — skip without claiming.
|
|
return "skip"
|
|
self.db.mark_processing(url)
|
|
self._in_flight += 1
|
|
return item
|
|
|
|
async def _worker(self, worker_id: int):
|
|
"""
|
|
One concurrent worker. Loops until the queue stays
|
|
empty *and* no peer worker is in-flight (peers can
|
|
enqueue new URLs as they discover links).
|
|
"""
|
|
while not self._stop:
|
|
if self.processed_this_run >= MAX_PAGES_PER_RUN:
|
|
return
|
|
|
|
item = await self._claim_next_url()
|
|
|
|
if item == "skip":
|
|
continue
|
|
|
|
if item is None:
|
|
# No work available — but a peer might enqueue
|
|
# more. Exit only if everyone else is idle too.
|
|
if self._in_flight == 0:
|
|
return
|
|
await asyncio.sleep(0.3)
|
|
continue
|
|
|
|
url = item["url"]
|
|
depth = item["depth"]
|
|
|
|
try:
|
|
result = await self.process_page(
|
|
url=url,
|
|
depth=depth,
|
|
)
|
|
|
|
async with self.db_lock:
|
|
self.db.mark_done(url)
|
|
if result and result.url != url:
|
|
self.db.mark_done(result.url)
|
|
self.processed_this_run += 1
|
|
n = self.processed_this_run
|
|
|
|
if n % 20 == 0:
|
|
logger.info(
|
|
f"Progress: {n} pages this run. "
|
|
f"Queue: {self.db.queue_stats()}"
|
|
)
|
|
if n % 100 == 0:
|
|
self.export_summary()
|
|
logger.info(
|
|
"Intermediate summary exported."
|
|
)
|
|
|
|
except asyncio.CancelledError:
|
|
raise
|
|
|
|
except _PermanentFetchFailure as e:
|
|
# Skip the retry ramp — re-fetching can't
|
|
# succeed.
|
|
logger.warning(f"Permanent failure on {url}: {e}")
|
|
async with self.db_lock:
|
|
self.db.mark_permanently_failed(url, str(e))
|
|
|
|
except Exception as e:
|
|
logger.exception(
|
|
f"Failed processing {url}: {e}"
|
|
)
|
|
async with self.db_lock:
|
|
self.db.mark_failed(
|
|
url,
|
|
str(e),
|
|
max_attempts=MAX_ATTEMPTS,
|
|
)
|
|
|
|
finally:
|
|
async with self.db_lock:
|
|
self._in_flight -= 1
|
|
async def _fetch_and_extract(self, url: str):
|
|
"""
|
|
Try a cheap HTTP fetch first. If a configured content
|
|
selector matches we're done. Otherwise fall back to a
|
|
full Playwright render (handles JS-only pages).
|
|
|
|
Returns (response_dict, extraction_dict) or (None, None)
|
|
when neither path produces usable content.
|
|
"""
|
|
|
|
# Conditional GET: ask the server whether the page
|
|
# has changed since we last cached it. A 304 lets us
|
|
# skip all downstream work.
|
|
etag, last_modified = self.db.get_cached_headers(url)
|
|
|
|
http_response = await self.http_client.fetch_page(
|
|
url,
|
|
etag=etag,
|
|
last_modified=last_modified,
|
|
)
|
|
|
|
if http_response is not None and http_response.get("is_binary"):
|
|
return http_response, None
|
|
|
|
if http_response is not None and http_response.get("not_modified"):
|
|
# Signal "nothing changed" to process_page.
|
|
return http_response, None
|
|
|
|
# Hard 4xx (404, 410, 451, ...): there is nothing
|
|
# Playwright can do that httpx cannot. Surface this so
|
|
# the worker can mark the URL permanently failed and
|
|
# stop wasting retries.
|
|
if (
|
|
http_response is not None
|
|
and http_response.get("status_code") is not None
|
|
and 400 <= http_response["status_code"] < 500
|
|
):
|
|
return http_response, None
|
|
|
|
if http_response is not None and http_response.get("ok"):
|
|
# Parse the HTML once here. The cleaner mutates
|
|
# the soup in place; the extractor uses it; assets
|
|
# later use the matched node. No re-parses.
|
|
cleaning_soup = BeautifulSoup(
|
|
http_response["html"], "lxml"
|
|
)
|
|
self.cleaner.clean(cleaning_soup)
|
|
extraction = self.extractor.extract_with_status(
|
|
cleaning_soup
|
|
)
|
|
|
|
if extraction["matched"]:
|
|
logger.info(f"HTTP fast-path success: {url}")
|
|
return http_response, extraction
|
|
|
|
logger.info(
|
|
f"HTTP fetch ok but no content selector matched "
|
|
f"— falling back to Playwright: {url}"
|
|
)
|
|
|
|
# Fallback: full browser render
|
|
pw_response = await self.client.fetch_page(url)
|
|
|
|
if pw_response.get("is_binary"):
|
|
return pw_response, None
|
|
|
|
cleaning_soup = BeautifulSoup(pw_response["html"], "lxml")
|
|
self.cleaner.clean(cleaning_soup)
|
|
extraction = self.extractor.extract_with_status(
|
|
cleaning_soup
|
|
)
|
|
|
|
return pw_response, extraction
|
|
|
|
async def process_page(self, url: str, depth: int):
|
|
"""
|
|
Process single page.
|
|
"""
|
|
|
|
logger.info(
|
|
f"Crawling: {url}"
|
|
)
|
|
if self.url_manager.is_binary_url(url):
|
|
logger.info(f"Skipping binary URL as page: {url}")
|
|
return None
|
|
|
|
response, extraction = await self._fetch_and_extract(url)
|
|
|
|
if response is None:
|
|
return None
|
|
|
|
if response.get("is_binary"):
|
|
logger.info(f"Skipping binary response: {url}")
|
|
return None
|
|
|
|
# 304 Not Modified: skip the entire pipeline. Worker
|
|
# will mark_done as usual.
|
|
if response.get("not_modified"):
|
|
logger.info(f"Not modified, skipping: {url}")
|
|
return CrawlResult(
|
|
status_code=304,
|
|
url=url,
|
|
title="",
|
|
markdown_path="",
|
|
links=[],
|
|
success=True,
|
|
)
|
|
|
|
# _fetch_and_extract returns extraction=None with a
|
|
# 4xx response when the URL is a permanent failure
|
|
# (404 etc.). Raise a typed error so the worker maps
|
|
# it to mark_permanently_failed instead of retrying.
|
|
status_code = response.get("status_code")
|
|
if extraction is None and status_code and 400 <= status_code < 500:
|
|
raise _PermanentFetchFailure(
|
|
f"HTTP {status_code} on {url}"
|
|
)
|
|
|
|
final_url = response.get("url", url)
|
|
|
|
html = response["html"]
|
|
|
|
title = response["title"]
|
|
|
|
# The extracted node lives inside the cleaning soup
|
|
# already produced in _fetch_and_extract. Use it
|
|
# directly to avoid one BS4 re-parse per page.
|
|
extracted_node = extraction["node"]
|
|
|
|
# Resolve output path first
|
|
output_path = self.path_mapper.url_to_path(
|
|
final_url
|
|
)
|
|
|
|
# Download assets — pass the already-parsed node
|
|
# rather than its serialized HTML string.
|
|
downloaded_assets = await self.asset_downloader.download_assets_from_html(
|
|
base_url=final_url,
|
|
html=extracted_node,
|
|
)
|
|
|
|
async with self.db_lock:
|
|
with self.db.transaction():
|
|
for asset in downloaded_assets:
|
|
self.db.save_asset(
|
|
page_url=final_url,
|
|
asset_url=asset["url"],
|
|
local_path=asset["local_path"]
|
|
)
|
|
# Rewrite asset links in place on the same parsed
|
|
# node — no re-parse needed.
|
|
self.asset_downloader.rewrite_asset_links(
|
|
html=extracted_node,
|
|
base_url=final_url,
|
|
current_output_path=output_path,
|
|
)
|
|
|
|
# Convert markdown
|
|
markdown = self.converter.html_to_markdown(
|
|
str(extracted_node),
|
|
title=title,
|
|
source_url=final_url,
|
|
media_dir=Path(OUTPUT_DIR) / "assets" / "pandoc",
|
|
)
|
|
|
|
self.writer.write_text(
|
|
output_path,
|
|
markdown
|
|
)
|
|
|
|
# Parse the ORIGINAL html ONCE for all three
|
|
# downstream consumers. Previously each one re-parsed
|
|
# the same string independently.
|
|
original_soup = BeautifulSoup(html, "lxml")
|
|
|
|
discovered_links = self.extract_links(
|
|
final_url,
|
|
original_soup,
|
|
)
|
|
|
|
sphinx_links = await self.sphinx_discovery.discover_from_rendered_html(
|
|
base_url=final_url,
|
|
html=original_soup,
|
|
)
|
|
|
|
# Only parse navigation the first time we visit a
|
|
# given domain. The sidebar tree is site-wide and
|
|
# repeating the parse on every page is the single
|
|
# most expensive bit of per-page CPU we still have.
|
|
nav_domain = self.url_manager.extract_domain(final_url)
|
|
if nav_domain in self._nav_parsed_domains:
|
|
navigation_result = {"urls": [], "tree": []}
|
|
else:
|
|
navigation_result = self.navigation_parser.parse_navigation(
|
|
base_url=final_url,
|
|
html=original_soup,
|
|
)
|
|
self._nav_parsed_domains.add(nav_domain)
|
|
|
|
self.navigation_trees.extend(
|
|
navigation_result["tree"]
|
|
)
|
|
|
|
for nav_url in navigation_result["urls"]:
|
|
if nav_url not in discovered_links:
|
|
discovered_links.append(nav_url)
|
|
|
|
logger.info(
|
|
f"SPHINX_LINKS: {len(sphinx_links)} from {final_url}"
|
|
)
|
|
|
|
for sphinx_link in sphinx_links:
|
|
if sphinx_link not in discovered_links:
|
|
discovered_links.append(sphinx_link)
|
|
|
|
# Batch every per-link write + the page row into one
|
|
# transaction per page — turns hundreds of fsyncs into
|
|
# one, and one acquisition of db_lock per worker turn.
|
|
async with self.db_lock:
|
|
with self.db.transaction():
|
|
for link in discovered_links:
|
|
|
|
logger.debug(f"DISCOVERED: {link}")
|
|
|
|
self.db.save_page_link(
|
|
source_url=final_url,
|
|
target_url=link,
|
|
link_type="sphinx_or_html_internal"
|
|
)
|
|
|
|
if not self.url_manager.should_enqueue(link):
|
|
continue
|
|
|
|
if self.db.is_done(link):
|
|
continue
|
|
|
|
if self.db.is_known(link):
|
|
continue
|
|
|
|
logger.debug(f"ENQUEUED: {link}")
|
|
|
|
self.db.enqueue_url(
|
|
url=link,
|
|
parent_url=final_url,
|
|
depth=depth + 1,
|
|
priority=0
|
|
)
|
|
|
|
self.db.save_discovery_source(
|
|
url=link,
|
|
discovery_type="page_internal_link",
|
|
source_url=final_url
|
|
)
|
|
|
|
self.db.save_page(
|
|
url=final_url,
|
|
title=title,
|
|
markdown_path=str(output_path),
|
|
status="success",
|
|
status_code=response.get(
|
|
"status_code",
|
|
200
|
|
),
|
|
markdown_length=len(markdown),
|
|
word_count=len(
|
|
markdown.split()
|
|
),
|
|
metadata={
|
|
"depth": depth,
|
|
"links_found": len(discovered_links),
|
|
"assets_found": len(downloaded_assets)
|
|
},
|
|
etag=response.get("etag"),
|
|
last_modified=response.get("last_modified"),
|
|
)
|
|
|
|
logger.debug(
|
|
f"Saved markdown: "
|
|
f"{output_path}"
|
|
)
|
|
|
|
return CrawlResult(
|
|
status_code=response.get("status_code"),
|
|
url=final_url,
|
|
title=title,
|
|
markdown_path=str(output_path),
|
|
links=discovered_links,
|
|
success=True
|
|
)
|
|
|
|
def extract_links(
|
|
self,
|
|
base_url,
|
|
html
|
|
):
|
|
"""
|
|
Extract crawlable links. Accepts an HTML string or a
|
|
parsed BS4 soup/Tag.
|
|
"""
|
|
|
|
if hasattr(html, "select") and hasattr(html, "find_all"):
|
|
soup = html
|
|
else:
|
|
soup = BeautifulSoup(html, "lxml")
|
|
|
|
discovered = set()
|
|
|
|
for a in soup.find_all(
|
|
"a",
|
|
href=True
|
|
):
|
|
|
|
href = a.get("href")
|
|
|
|
resolved = self.url_manager.resolve(
|
|
base_url,
|
|
href
|
|
)
|
|
|
|
if not resolved:
|
|
continue
|
|
|
|
if not self.url_manager.is_allowed(
|
|
resolved
|
|
):
|
|
continue
|
|
|
|
discovered.add(
|
|
resolved
|
|
)
|
|
selectors = [
|
|
"a.reference.internal",
|
|
".toctree-wrapper a",
|
|
".wy-menu a",
|
|
".sphinxsidebar a",
|
|
".globaltoc a",
|
|
".localtoc a",
|
|
"link[rel='next']",
|
|
"link[rel='prev']",
|
|
"link[rel='up']"
|
|
]
|
|
|
|
for selector in selectors:
|
|
for node in soup.select(selector):
|
|
href = node.get("href")
|
|
|
|
resolved = self.url_manager.resolve(
|
|
base_url,
|
|
href
|
|
)
|
|
|
|
if not resolved:
|
|
continue
|
|
|
|
if not self.url_manager.is_allowed(resolved):
|
|
continue
|
|
|
|
discovered.add(resolved)
|
|
|
|
return list(discovered)
|
|
|
|
def export_summary(self):
|
|
"""
|
|
Export crawl summary.
|
|
"""
|
|
|
|
summary_path = (
|
|
Path(OUTPUT_DIR)
|
|
/ "crawl_summary.json"
|
|
)
|
|
|
|
data = {
|
|
"processed_this_run": self.processed_this_run,
|
|
"queue_stats": self.db.queue_stats()
|
|
}
|
|
|
|
self.writer.write_json(
|
|
summary_path,
|
|
data
|
|
)
|
|
|
|
graph_path = (
|
|
Path(OUTPUT_DIR)
|
|
/ "site_graph.json"
|
|
)
|
|
|
|
self.db.export_site_graph(
|
|
graph_path
|
|
)
|
|
|
|
navigation_tree = self.navigation_parser.dedupe_tree(
|
|
self.navigation_trees
|
|
)
|
|
|
|
navigation_path = (
|
|
Path(OUTPUT_DIR)
|
|
/ "navigation_tree.json"
|
|
)
|
|
|
|
self.navigation_parser.export_tree_json(
|
|
navigation_tree,
|
|
navigation_path
|
|
)
|
|
|
|
summary_md = self.navigation_parser.generate_summary_markdown(
|
|
navigation_tree
|
|
)
|
|
|
|
summary_md_path = (
|
|
Path(OUTPUT_DIR)
|
|
/ "SUMMARY.md"
|
|
)
|
|
|
|
self.writer.write_text(
|
|
summary_md_path,
|
|
summary_md
|
|
)
|
|
|
|
logger.info(
|
|
f"Summary exported: "
|
|
f"{summary_path}"
|
|
)
|
|
|
|
async def recycle_browser(self):
|
|
"""
|
|
Restart Playwright browser/context periodically
|
|
to avoid long-running memory/socket leaks.
|
|
"""
|
|
|
|
logger.info("Recycling browser context...")
|
|
|
|
await self.client.stop()
|
|
await asyncio.sleep(2)
|
|
await self.client.start()
|
|
|
|
logger.info("Browser context recycled.")
|
|
|
|
async def async_main():
|
|
crawler = DocumentationCrawler()
|
|
await crawler.crawl()
|
|
|
|
|
|
def main():
|
|
asyncio.run(
|
|
async_main()
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |