การสแครปแบบ async คือจุดที่การตั้งค่าพร็อกซีพังลงอย่างเงียบ ๆ ด้วย requests คุณหมุน IP หนึ่งตัวต่อหนึ่งการเรียกแล้วเดินหน้าต่อ แต่ด้วย asyncio จู่ ๆ คุณก็มี 200 คำขอกำลังบินอยู่พร้อมกัน ทุกตัวต้องการ exit ของตัวเอง ต้องการความผูกพันกับเซสชันของตัวเอง และต้องการการลองซ้ำของตัวเองเมื่อเป้าหมายเด้ง 403 ออกมา ทำผิดแล้วคุณจะทุบ IP ตัวเดียวจนถูกแบน หรือไม่ก็ฉีกทุกเซสชันล็อกอินทิ้งด้วยการหมุนกลางคัน
คู่มือนี้แสดงแพตเทิร์นที่ยืนหยัดได้จริงภายใต้การทำงานพร้อมกัน: การหมุนต่อคำขอใน httpx และ aiohttp พูล worker ที่มีขอบเขตจำกัด sticky session สำหรับโฟลว์ที่มีสถานะ และการลองซ้ำที่รู้ทันการบล็อก — พร้อมโค้ดที่ใช้งานได้จริงให้คุณก๊อปไปวางได้
สแครปเปอร์แบบ synchronous แตะพร็อกซีทีละตัว ดังนั้น "หมุนในทุกคำขอ" จึงถูกต้องอย่างง่ายดาย แต่ async ทำลายสมมติฐานสามข้อพร้อมกัน:
httpx.AsyncClient ผูกพร็อกซีหนึ่งตัวต่อหนึ่งไคลเอนต์ ดังนั้นการจะหมุนได้คุณต้องเลือกพร็อกซีต่อคำขอและจัดเส้นทางผ่านพูลไคลเอนต์เล็ก ๆ ที่ key ด้วย exit:
import asyncio, itertools, httpx
PROXIES = [
"socks5h://USERNAME:[email protected]:913",
"socks5h://USERNAME:[email protected]:913",
"socks5h://USERNAME:[email protected]:913",
]
pool = itertools.cycle(PROXIES)
# One reusable client per exit (connection pooling stays intact)
clients = {p: httpx.AsyncClient(proxy=p, timeout=30) for p in PROXIES}
async def fetch(url):
proxy = next(pool)
r = await clients[proxy].get(url)
return r.status_code, r.text
async def main(urls):
results = await asyncio.gather(*(fetch(u) for u in urls))
for client in clients.values():
await client.aclose()
return results
การใช้ไคลเอนต์ตัวเดียวซ้ำต่อ exit ทำให้ HTTP/2 และ connection pooling ยังคงอยู่ แทนที่จะต้องจ่ายค่า handshake ใหม่ในทุกคำขอ โปรดสังเกตว่า httpx>=0.28 ได้เปลี่ยนชื่อ proxies= เป็น proxy= บนไคลเอนต์
aiohttp ง่ายกว่าในจุดนี้ — ClientSession ตัวเดียวรับอาร์กิวเมนต์ proxy= ต่อคำขอ ดังนั้นคุณหมุนแบบ inline ได้เลย:
import asyncio, itertools, aiohttp
pool = itertools.cycle(PROXIES) # http:// or socks5h:// (needs aiohttp-socks for SOCKS)
async def fetch(session, url):
proxy = next(pool)
async with session.get(url, proxy=proxy, timeout=aiohttp.ClientTimeout(total=30)) as r:
return r.status, await r.text()
async def main(urls):
async with aiohttp.ClientSession() as session:
return await asyncio.gather(*(fetch(session, u) for u in urls))
aiohttp พูดภาษาพร็อกซี HTTP ได้โดยกำเนิด สำหรับ socks5h:// ให้เพิ่ม aiohttp-socks และส่ง ProxyConnector เข้าไป
gather ที่ไม่มีขอบเขตคือวิธีที่เร็วที่สุดที่จะทำให้ทุก IP ในพูลของคุณถูกตั้งธงพร้อมกัน จงจำกัดมันเพื่อให้แต่ละ exit แบกจำนวนคำขอขนานที่ดูสมเหตุสมผลแบบมนุษย์:
sem = asyncio.Semaphore(10) # at most 10 in flight
async def fetch_capped(session, url):
async with sem:
return await fetch(session, url)
กฎคร่าว ๆ: คงระดับการทำงานพร้อมกันให้เท่ากับหรือต่ำกว่าจำนวน exit ที่แตกต่างกันที่คุณมี เพื่อไม่ให้คุณซ้อนคำขอขนานจำนวนมากบน IP residential ตัวเดียว
การหมุนมีไว้เพื่อเข้าไป sticky session มีไว้เพื่ออยู่ต่อ การล็อกอิน ตะกร้าสินค้า อะไรก็ตามที่ผูกกับคุกกี้ต้องคง exit เดียวไว้ตลอดทั้งลำดับ ปักหมุดพร็อกซีต่องานแทนที่จะต่อคำขอ:
async def run_account(account, proxy):
# Same exit IP for every step of this account's flow
async with httpx.AsyncClient(proxy=proxy, timeout=30) as client:
await client.post("https://target.site/login", data=account.creds)
await client.get("https://target.site/dashboard")
await client.post("https://target.site/cart", json=account.order)
async def main(accounts):
# One sticky exit per account, accounts run concurrently
await asyncio.gather(*(run_account(a, p)
for a, p in zip(accounts, itertools.cycle(PROXIES))))
ใช้ผู้ให้บริการที่รองรับ sticky residential session เพื่อให้ exit เดิมถูกคงไว้ตลอดอายุของงาน ไม่ใช่ถูกหมุนเปลี่ยนใต้มือคุณอย่างเงียบ ๆ
เมื่อทำงานพร้อมกัน คุณไว้ใจ 200 ไม่ได้ — ระบบต่อต้านบอตคืน 200 มาพร้อมเนื้อหา challenge ตรวจจับการบล็อกแล้วลองซ้ำเฉพาะงานเดียวที่ล้มบน exit ที่สดใหม่:
BLOCK_MARKERS = ("just a moment", "/cdn-cgi/challenge-platform",
"datadome", "px-captcha", "access denied")
def looks_blocked(status, text):
return status in (403, 429, 503) or any(m in text[:4000].lower() for m in BLOCK_MARKERS)
async def fetch_retry(url, attempts=4):
for _ in range(attempts):
proxy = next(pool)
async with httpx.AsyncClient(proxy=proxy, timeout=30) as c:
r = await c.get(url)
if not looks_blocked(r.status_code, r.text):
return r
await asyncio.sleep(0.5)
raise RuntimeError(f"blocked after {attempts} exits: {url}")
มีจุดหนึ่งที่ async แก้ไม่ได้: ทั้ง httpx และ aiohttp ต่างก็นั่งอยู่บนสแต็ก TLS ของ Python ดังนั้นมันส่ง JA3 ที่ไม่มีเบราว์เซอร์จริงตัวไหนผลิต พูลการหมุนที่สมบูรณ์แบบบน IP residential สะอาดก็ยังถูกพิมพ์ลายนิ้วมือว่าเป็นการทำงานอัตโนมัติตอน handshake จับคู่การหมุนแบบ async เข้ากับการปลอม TLS — ดู Bypass TLS Fingerprinting with curl_cffi และตรวจสอบ JA3 + ASN ของ exit ของคุณที่ www.jibaoproxy.com/tools/fingerprint.html
ผู้ใช้ใหม่รับ 500MB เมื่อสมัครสมาชิก พร้อมโบนัสในการเติมเงินครั้งแรก ข้อเสนอมีระยะเวลาจำกัด