Most pycurl tutorials are stuck in 2012. They show you the verbose BytesIO + WRITEFUNCTION pattern when there's a one-liner that does the same thing.
This guide covers what actually matters for high performance HTTP in Python.
Every tutorial shows this:
from io import BytesIO
import pycurl
buffer = BytesIO()
curl = pycurl.Curl()
curl.setopt(curl.URL, 'https://httpbin.org/get')
curl.setopt(curl.WRITEFUNCTION, buffer.write)
curl.perform()
response = buffer.getvalue().decode('utf-8')
curl.close()When you can just do this:
import pycurl
curl = pycurl.Curl()
curl.setopt(curl.URL, 'https://httpbin.org/get')
response = curl.perform_rs() # returns string
curl.close()Or if you need bytes:
response = curl.perform_rb() # returns bytesThese methods have been in pycurl for years. No idea why nobody uses them.
Creating a new Curl() object for every request is slow. The handle has to allocate memory, set up internal state, etc.
Instead, create once and reuse:
import pycurl
curl = pycurl.Curl()
for i in range(1000):
curl.setopt(curl.URL, f'https://httpbin.org/get?i={i}')
response = curl.perform_rs()
print(len(response))
curl.reset() # clears options, keeps the handle and connection alive
curl.close()reset() wipes all your setopt calls but keeps the underlying connection pool. This means subsequent requests to the same host skip the TCP handshake and TLS negotiation.
When you're running multiple threads, create a pool of handles upfront:
import pycurl
import threading
class CurlPool:
def __init__(self, size):
self.handles = [pycurl.Curl() for _ in range(size)]
self.index = 0
self.lock = threading.Lock()
def get(self):
with self.lock:
handle = self.handles[self.index]
self.index = (self.index + 1) % len(self.handles)
return handle
def close(self):
for h in self.handles:
h.close()
# create 50 handles at startup
pool = CurlPool(50)
def worker():
curl = pool.get()
curl.setopt(curl.URL, 'https://httpbin.org/get')
response = curl.perform_rs()
# don't reset() here if you want connection reuse across calls
for _ in range(10):
threading.Thread(target=worker).start()TIMEOUT is in seconds. TIMEOUT_MS gives you millisecond precision:
curl.setopt(curl.TIMEOUT, 10) # 10 seconds
curl.setopt(curl.TIMEOUT_MS, 15000) # 15000 millisecondsSeparate your connection timeout from the overall timeout:
curl.setopt(curl.CONNECTTIMEOUT, 5) # 5 seconds to establish connection
curl.setopt(curl.TIMEOUT, 30) # 30 seconds totalThese actually make a difference for high frequency requests:
curl.setopt(curl.TCP_NODELAY, 1) # disable Nagle algorithm, send small packets immediately
curl.setopt(curl.TCP_KEEPALIVE, 1) # keep connections alive between requestsIf you're making lots of requests to the same host, the keepalive saves you from reconnecting every time.
For production, leave these alone. For testing with proxies or self-signed certs:
curl.setopt(curl.SSL_VERIFYPEER, 0) # don't verify the certificate
curl.setopt(curl.SSL_VERIFYHOST, 0) # don't verify the hostname matchescurl.setopt(curl.PROXY, "http://127.0.0.1:8080")
# with auth
curl.setopt(curl.PROXY, "http://user:pass@127.0.0.1:8080")
# socks5
curl.setopt(curl.PROXYTYPE, pycurl.PROXYTYPE_SOCKS5)
curl.setopt(curl.PROXY, "socks5://127.0.0.1:1080")import json
curl = pycurl.Curl()
curl.setopt(curl.URL, 'https://httpbin.org/post')
curl.setopt(curl.POST, 1)
curl.setopt(curl.POSTFIELDS, json.dumps({"key": "value"}))
curl.setopt(curl.HTTPHEADER, [
'Content-Type: application/json',
'Accept: application/json'
])
response = curl.perform_rs()After perform_rs() or perform_rb(), you can grab metadata:
curl.setopt(curl.URL, 'https://httpbin.org/get')
response = curl.perform_rs()
status_code = curl.getinfo(curl.RESPONSE_CODE) # 200
total_time = curl.getinfo(curl.TOTAL_TIME) # seconds as float
download_speed = curl.getinfo(curl.SPEED_DOWNLOAD) # bytes per secondAccept compressed responses and let curl decompress automatically:
curl.setopt(curl.ENCODING, 'gzip, deflate, br')The response from perform_rs() will already be decompressed.
Putting it all together:
import pycurl
import json
def request(url, method='GET', headers=None, data=None, proxy=None, timeout_ms=10000):
curl = pycurl.Curl()
# url and method
curl.setopt(curl.URL, url)
if method == 'POST':
curl.setopt(curl.POST, 1)
if data:
curl.setopt(curl.POSTFIELDS, data if isinstance(data, str) else json.dumps(data))
# headers
if headers:
curl.setopt(curl.HTTPHEADER, [f'{k}: {v}' for k, v in headers.items()])
# performance
curl.setopt(curl.TIMEOUT_MS, timeout_ms)
curl.setopt(curl.CONNECTTIMEOUT, 5)
curl.setopt(curl.TCP_NODELAY, 1)
curl.setopt(curl.TCP_KEEPALIVE, 1)
curl.setopt(curl.ENCODING, 'gzip, deflate, br')
# proxy
if proxy:
curl.setopt(curl.PROXY, proxy)
curl.setopt(curl.SSL_VERIFYPEER, 0)
curl.setopt(curl.SSL_VERIFYHOST, 0)
try:
response = curl.perform_rs()
status = curl.getinfo(curl.RESPONSE_CODE)
return {'status': status, 'body': response}
except pycurl.error as e:
return {'error': str(e)}
finally:
curl.close()
# usage
resp = request(
'https://httpbin.org/post',
method='POST',
headers={'Content-Type': 'application/json'},
data={'test': 'data'},
timeout_ms=5000
)
print(resp['status'], resp['body'])- pycurl is a thin wrapper around libcurl, the same library that powers curl
- connection pooling and reuse actually works
- way faster for high volume stuff
- more control over low level socket options
The tradeoff is the API is less pythonic. But if you need speed, this is it.