SDKs
Python
The official scrapeflow package for Python 3.9+. Sync and async clients, typed models, and first-class support for pandas and vector stores.
Install
bash
pip install scrapeflow # or: uv add scrapeflow
Initialize
client.py
import os
from scrapeflow import ScrapeFlow
client = ScrapeFlow(
api_key=os.environ["SCRAPEFLOW_API_KEY"],
# base_url defaults to https://api.scrapeflow.dev
max_retries=3,
timeout=60,
)Usage
python
# Scrape
page = client.web.scrape(
url="https://linear.app",
formats={"markdown": True, "json": True},
json_schema={"title": "string", "pricing_tiers": "string[]"},
)
# Search
hits = client.web.search(query="vector databases", limit=10)
# Answers
research = client.web.answers(query="Who founded Vercel?")
# Brand
brand = client.brand.get(domain="notion.com")
# Async jobs: crawl + batches
crawl = client.web.crawl(url="https://docs.stripe.com", limit=500)
result = client.jobs.wait(crawl.job_id)Async client
python
import asyncio
from scrapeflow import AsyncScrapeFlow
async def main():
client = AsyncScrapeFlow(api_key=os.environ["SCRAPEFLOW_API_KEY"])
page = await client.web.scrape(url="https://linear.app")
print(page.data.markdown)
asyncio.run(main())Error handling
python
from scrapeflow import ScrapeFlowError, RateLimitError
try:
client.web.scrape(url=url)
except RateLimitError as e:
time.sleep(e.retry_after)
except ScrapeFlowError as e:
print(e.status, e.code, e.message)ℹ
Both clients retry
429 and 5xx with exponential backoff. Use client.batches.create(...).to_dataframe() to load results straight into pandas.