一、协程 vs 线程
| 维度 | 线程 | 协程 | | --- | --- | --- | | 切换成本 | 内核态,微秒级 | 用户态,纳秒级 | | 并发数 | 数百 | 数万 | | 数据共享 | 需加锁 | 单线程无需锁 | | 适用场景 | 混合任务 | 纯 IO 密集 |二、最简示例
import asyncio
import time
async def say_after(delay, msg):
await asyncio.sleep(delay)
print(msg)
async def main():
print(f"开始 {time.strftime('%X')}")
await say_after(1, 'hello')
await say_after(2, 'world')
print(f"结束 {time.strftime('%X')}")
asyncio.run(main()) # 总耗时 3s(串行)
并发版本:
async def main():
task1 = asyncio.create_task(say_after(1, 'hello'))
task2 = asyncio.create_task(say_after(2, 'world'))
await task1
await task2
# 总耗时 2s
三、高并发爬虫
import aiohttp
import asyncio
async def fetch(session, url, sem):
async with sem: # 限制并发
try:
async with session.get(url, timeout=10) as resp:
return await resp.text()
except Exception as e:
print(f'失败 {url}: {e}')
return None
async def main(urls):
sem = asyncio.Semaphore(50)
async with aiohttp.ClientSession() as session:
tasks = [fetch(session, u, sem) for u in urls]
return await asyncio.gather(*tasks)
urls = [f'https://example.com/page/{i}' for i in range(1000)]
results = asyncio.run(main(urls))
print(f'成功 {sum(1 for r in results if r)} 个')
四、常见错误
# ❌ 在协程中使用同步阻塞调用
async def bad():
time.sleep(1) # 阻塞整个事件循环!
requests.get(url) # 同样阻塞
# ✅ 正确
async def good():
await asyncio.sleep(1)
# 或用 run_in_executor 包装同步代码
loop = asyncio.get_running_loop()
await loop.run_in_executor(None, requests.get, url)
五、性能实测
抓取 1000 个网页:| 方案 | 耗时 | 内存 |
| --- | --- | --- |
| 同步 requests | 268s | 30MB |
| 线程池(50)| 12s | 85MB |
| asyncio(50并发)| 8.4s | 45MB |
六、实用技巧
# 1. 超时控制
await asyncio.wait_for(fetch(url), timeout=5)
# 2. 逐个完成
for coro in asyncio.as_completed(tasks):
result = await coro
# 3. 取消任务
task.cancel()
try:
await task
except asyncio.CancelledError:
print('已取消')
# 4. 同步接口转异步(如数据库)
await loop.run_in_executor(None, cursor.execute, sql)
七、什么时候不要用 asyncio
- CPU 密集任务(用 multiprocessing)
- 依赖的库不支持异步(requests、pymysql)
- 项目规模小,同步代码完全够用
评论(0)
还没有评论,来说两句吧