INNER CODE UNIT · Python
extract_page
tiliondev/fortress · mcp/server.py:304
async def extract_page(url: str, schema: dict | None = None) -> dict:
"""Extract clean, LLM-ready content from a page that may be behind bot
detection — returns markdown + tables + metadata (or a schema-shaped record if
`schema` is given). Use instead of raw HTML scraping when you want structured,
readable content and the site blocks normal scrapers.
"""
await _check_url(url)
t = await _t()
r = await t.extract(url, schema=schema)
r.pop("html", None)
return _cap(r, {"markdown": 40000, "text": 20000})
@mcp.tool(annotations=_READ)
@_safe
async def crawl_site(url: str, depth: int = 2, max_pages: int = 100,
ctx: Context | None = None) -> dict:
"""Crawl a whole website (BFS, auto-handles SPA/JS + lazy-load) through the