Configure TrueProxies with Crawlee
Crawlee takes a ProxyConfiguration. For a new exit IP on each connection,
give it one TrueProxies proxy URL. For a sticky exit per Crawlee session, pass
a function that writes the Crawlee session ID into the username as a
session option. The same approach works in Crawlee for JavaScript and Crawlee
for Python.
Installation
Section titled “Installation”npm install crawlee playwrightnpx playwright install chromiumpip install "crawlee[playwright]"playwright install chromiumBasic usage
Section titled “Basic usage”One proxy URL is enough: TrueProxies picks a new exit for each new connection. URL-encode the username and proxy password when you build the URL.
import { PlaywrightCrawler, ProxyConfiguration } from 'crawlee';
const username = encodeURIComponent('your_username');const password = encodeURIComponent('your_password');
const proxyConfiguration = new ProxyConfiguration({ proxyUrls: [`http://${username}:${password}@YOUR_HOST:YOUR_HTTP_PORT`],});
const crawler = new PlaywrightCrawler({ proxyConfiguration, async requestHandler({ page, request, log }) { log.info(`${request.url}: ${await page.textContent('body')}`); },});
await crawler.run(['https://httpbin.org/ip']);import asynciofrom urllib.parse import quote
from crawlee.crawlers import PlaywrightCrawler, PlaywrightCrawlingContextfrom crawlee.proxy_configuration import ProxyConfiguration
username = quote("your_username", safe="")password = quote("your_password", safe="")
async def main() -> None: proxy_configuration = ProxyConfiguration( proxy_urls=[f"http://{username}:{password}@YOUR_HOST:YOUR_HTTP_PORT"], ) crawler = PlaywrightCrawler(proxy_configuration=proxy_configuration)
@crawler.router.default_handler async def handler(context: PlaywrightCrawlingContext) -> None: context.log.info(f"{context.request.url}: {await context.page.content()}")
await crawler.run(["https://httpbin.org/ip"])
asyncio.run(main())For Residential IPv4 country targeting, add the option to the username, for
example your_username-country-de.
Residential IPv4 sticky exit per Crawlee session
Section titled “Residential IPv4 sticky exit per Crawlee session”Crawlee’s session pool gives each session its own ID and cookies. Pass that ID
through as a TrueProxies session so every request in a Crawlee session asks
for the same exit IP. When Crawlee retires a session, for example after blocked
responses, the next session ID asks for a new exit.
TrueProxies session values are 1 to 32 letters or digits. Crawlee for
JavaScript creates IDs such as session_Ab3dE6gH9k, so remove every other
character before you use the ID. Without this step the gateway refuses the
username.
import { PlaywrightCrawler, ProxyConfiguration } from 'crawlee';
const USERNAME = 'your_username';const PASSWORD = 'your_password';const PROXY = 'YOUR_HOST:YOUR_HTTP_PORT';
const proxyConfiguration = new ProxyConfiguration({ newUrlFunction: (sessionId) => { const session = String(sessionId ?? '').replace(/[^A-Za-z0-9]/g, '').slice(0, 32); const user = session ? `${USERNAME}-session-${session}-lifetime-600` : USERNAME; return `http://${encodeURIComponent(user)}:${encodeURIComponent(PASSWORD)}@${PROXY}`; },});
const crawler = new PlaywrightCrawler({ proxyConfiguration, useSessionPool: true, persistCookiesPerSession: true, async requestHandler({ page, request, session, log }) { log.info(`${request.url} (${session?.id}): ${await page.textContent('body')}`); },});
await crawler.run(['https://httpbin.org/ip']);import asyncioimport refrom urllib.parse import quote
from crawlee import Requestfrom crawlee.crawlers import PlaywrightCrawler, PlaywrightCrawlingContextfrom crawlee.proxy_configuration import ProxyConfiguration
USERNAME = "your_username"PASSWORD = "your_password"PROXY = "YOUR_HOST:YOUR_HTTP_PORT"
def new_url(session_id: str | None = None, request: Request | None = None) -> str: session = re.sub(r"[^A-Za-z0-9]", "", session_id or "")[:32] user = f"{USERNAME}-session-{session}-lifetime-600" if session else USERNAME return f"http://{quote(user, safe='')}:{quote(PASSWORD, safe='')}@{PROXY}"
async def main() -> None: crawler = PlaywrightCrawler( proxy_configuration=ProxyConfiguration(new_url_function=new_url), use_session_pool=True, )
@crawler.router.default_handler async def handler(context: PlaywrightCrawlingContext) -> None: context.log.info(f"{context.request.url}: {await context.page.content()}")
await crawler.run(["https://httpbin.org/ip"])
asyncio.run(main())lifetime is in seconds. The accepted range depends on the product; see the
options table.
When requests fail
Section titled “When requests fail”Crawlee retries failed requests and logs the browser’s or HTTP client’s error,
which does not include the proxy’s reason. Send the same username and proxy
password with curl -v, read the X-Proxy-Reason header, and look it up in
Proxy errors and refusal reasons.