INNER CODE UNIT · Python

extract_page

tiliondev/fortress · mcp/server.py:304

async def extract_page(url: str, schema: dict | None = None) -> dict:
    """Extract clean, LLM-ready content from a page that may be behind bot
    detection — returns markdown + tables + metadata (or a schema-shaped record if
    `schema` is given). Use instead of raw HTML scraping when you want structured,
    readable content and the site blocks normal scrapers.
    """
    await _check_url(url)
    t = await _t()
    r = await t.extract(url, schema=schema)
    r.pop("html", None)
    return _cap(r, {"markdown": 40000, "text": 20000})


@mcp.tool(annotations=_READ)
@_safe
async def crawl_site(url: str, depth: int = 2, max_pages: int = 100,
                     ctx: Context | None = None) -> dict:
    """Crawl a whole website (BFS, auto-handles SPA/JS + lazy-load) through the

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…