Scrapy
Configure Scrapy with mobile, residential or static ISP proxies — per-request proxy meta, a downloader middleware for residential targeting and sticky sessions, and rotation-aware retry settings.
On this page
Scrapy's built-in HttpProxyMiddleware already handles authenticated HTTP proxies: set request.meta["proxy"] to a URL with credentials and it adds the Proxy-Authorization header for you.
Per request#
import scrapy
PROXY = "http://m_k3v9q2xa:[email protected]:10421"
class PricesSpider(scrapy.Spider):
name = "prices"
start_urls = ["https://example.com/catalogue"]
def start_requests(self):
for url in self.start_urls:
yield scrapy.Request(url, meta={"proxy": PROXY})
def parse(self, response):
for href in response.css("a.product::attr(href)").getall():
yield response.follow(href, self.parse_product, meta={"proxy": PROXY})
def parse_product(self, response):
yield {"url": response.url, "price": response.css(".price::text").get()}
Scrapy does not support SOCKS proxies natively; use the HTTP port.
A middleware for residential targeting#
This downloader middleware builds the residential username per request. Spiders choose the location with meta keys; requests that must share an IP (a login, then pages behind it) share a session_key.
# myproject/middlewares.py
import hashlib
class ProxunoResidentialMiddleware:
"""Route requests through the residential gateway with country/city/session flags."""
def __init__(self, login, password, host, default_country):
self.login, self.password = login, password
self.host, self.default_country = host, default_country
@classmethod
def from_crawler(cls, crawler):
s = crawler.settings
return cls(s["PROXUNO_LOGIN"], s["PROXUNO_PASSWORD"],
s.get("PROXUNO_HOST", "resi.gw.proxuno.com:7000"),
s.get("PROXUNO_COUNTRY", "eu"))
def process_request(self, request, spider):
if "proxy" in request.meta: # explicit proxy wins
return None
user = f"{self.login}-country-{request.meta.get('country', self.default_country)}"
if request.meta.get("city"):
user += f"-city-{request.meta['city']}"
key = request.meta.get("session_key")
if key:
sid = hashlib.sha1(str(key).encode()).hexdigest()[:8]
user += f"-session-{sid}-ttl-{request.meta.get('session_ttl', 10)}"
request.meta["proxy"] = f"http://{user}:{self.password}@{self.host}"
return None
# settings.py
DOWNLOADER_MIDDLEWARES = {
"myproject.middlewares.ProxunoResidentialMiddleware": 350, # before HttpProxyMiddleware (750)
}
PROXUNO_LOGIN = "px4k2m9q7a"
PROXUNO_PASSWORD = "Rb8Tn3Wq6Ys1Kd5P" # better: read from an environment variable
PROXUNO_COUNTRY = "de"
yield scrapy.Request(url, meta={"country": "fr", "city": "paris"})
yield scrapy.Request(login_url, meta={"session_key": "account-17", "session_ttl": 20})
Hashing the session_key gives a stable 8-character session id, so every request for "account-17" leaves through the same residential IP for 20 minutes.
Recommended settings#
CONCURRENT_REQUESTS = 32
CONCURRENT_REQUESTS_PER_DOMAIN = 8
DOWNLOAD_TIMEOUT = 45
RETRY_ENABLED = True
RETRY_TIMES = 3
RETRY_HTTP_CODES = [429, 500, 502, 503, 504, 522, 524, 408]
AUTOTHROTTLE_ENABLED = True
AUTOTHROTTLE_START_DELAY = 1.0
AUTOTHROTTLE_TARGET_CONCURRENCY = 4.0
COOKIES_ENABLED = True
- On rotating residential, every retry automatically goes out through a new IP.
- On a mobile proxy, retries reuse the same IP. Rotate between crawl batches (see below) rather than per request.
- Keep
CONCURRENT_REQUESTS_PER_DOMAINmodest: blocks usually come from request rate per site, not from the IP type.
Rotating a mobile IP from an extension#
# myproject/extensions.py
import requests
from scrapy import signals
class RotateEveryN:
def __init__(self, link, every):
self.link, self.every, self.count = link, every, 0
@classmethod
def from_crawler(cls, crawler):
ext = cls(crawler.settings["PROXUNO_ROTATION_LINK"], crawler.settings.getint("PROXUNO_ROTATE_EVERY", 200))
crawler.signals.connect(ext.response_received, signal=signals.response_received)
return ext
def response_received(self, response, request, spider):
self.count += 1
if self.count % self.every == 0:
r = requests.get(self.link, timeout=45)
spider.logger.info("rotated: %s", r.json().get("ip") if r.ok else r.status_code)
The rotation call blocks the reactor for 5–10 s, which is acceptable at a few rotations per hour. For frequent rotation, run several mobile proxies and spread requests across them instead.
Something unclear, outdated or wrong on this page? Tell us — documentation issues are fixed in the next release at the latest.
Report an issue with this page