Python异步编程

时游大约 5 分钟

异步编程(asyncio)

异步适合「IO 密集」的任务:网络请求、读写文件、数据库查询等——大部分时间在等对方响应,CPU 是闲着的。asyncio 让单线程在等待时切换去处理其他任务,用「协作调度」榨干等待时间。爬虫实战 2 中用到的 async_playwright、asyncio.run 正是本篇的内容。

同步的问题:等待时什么也做不了

import time

def fetch(url):
    print(f"开始请求 {url}")
    time.sleep(1)  # 模拟网络请求耗时1秒:同步阻塞,整个程序停在这里干等
    print(f"完成请求 {url}")
    return f"{url} 的数据"


# 串行执行3个请求:3次等待排队,总耗时约3秒
start = time.perf_counter()
for url in ["a.com", "b.com", "c.com"]:
    fetch(url)
print(f"同步总耗时:{time.perf_counter() - start:.2f} 秒")  # 同步总耗时:3.00 秒

协程与 async/await

import asyncio

# async def 定义的函数叫协程函数
async def fetch_data():
    print("执行协程")
    return "数据"


# 调用协程函数并不会执行函数体,而是返回一个协程对象
coro = fetch_data()
print(coro)  # <coroutine object fetch_data at 0x...>

# 协程对象需要交给事件循环执行,await / asyncio.run 都可以驱动它
result = asyncio.run(coro)
print(result)  # 数据
import asyncio
import time

async def fetch(url):
    print(f"开始请求 {url}")
    # await:把控制权交还给事件循环,「等待」期间事件循环可以运行其他协程
    # time.sleep是同步阻塞(谁也别想动),asyncio.sleep是异步等待(等待时让出CPU)
    await asyncio.sleep(1)
    print(f"完成请求 {url}")
    return f"{url} 的数据"


# 协程中可以await另一个协程:async def main是入口协程
async def main():
    # 串行await:和同步一样排队,总耗时约3秒(反面示例)
    start = time.perf_counter()
    await fetch("a.com")
    await fetch("b.com")
    await fetch("c.com")
    print(f"串行总耗时:{time.perf_counter() - start:.2f} 秒")


asyncio.run(main())

asyncio.gather:并发执行

gather 把多个协程交给事件循环同时推进:谁在等待就让出位置给别人,总耗时约等于最慢的那一个,而不是所有任务之和。

import asyncio
import time

async def fetch(url):
    await asyncio.sleep(1)  # 模拟1秒网络请求
    return f"{url} 的数据"


async def main():
    start = time.perf_counter()

    # gather并发执行3个协程,结果按传入顺序返回
    results = await asyncio.gather(
        fetch("a.com"),
        fetch("b.com"),
        fetch("c.com"),
    )

    print(results)  # ['a.com 的数据', 'b.com 的数据', 'c.com 的数据']
    print(f"并发总耗时:{time.perf_counter() - start:.2f} 秒")  # 并发总耗时:1.00 秒


asyncio.run(main())

create_task:先启动,后取结果

gather 适合「一批一起等」;create_task 把协程立刻调度运行,主流程可以继续干别的,最后再 await 取结果——适合「先发请求、同时处理本地逻辑、最后收响应」的场景。

import asyncio

async def fetch(url):
    await asyncio.sleep(1)
    return f"{url} 的数据"


async def main():
    # create_task:立即把协程丢进事件循环开始执行,返回Task对象
    task = asyncio.create_task(fetch("a.com"))

    print("请求进行中,先干点别的...")  # 请求在后台跑,这里不会被阻塞
    await asyncio.sleep(0.5)
    print("本地逻辑处理完毕")

    result = await task  # 等待并获取Task的结果
    print(result)  # a.com 的数据


asyncio.run(main())

超时控制

import asyncio

async def slow_api():
    await asyncio.sleep(10)
    return "结果"


async def main():
    # wait_for:限制等待时长,超时抛出TimeoutError(配合[错误与异常](../基础/6.错误与异常.md)的try/except处理)
    try:
        result = await asyncio.wait_for(slow_api(), timeout=2)
    except TimeoutError:
        print("请求超时,执行降级逻辑")  # 2秒后走到这里

    # Python 3.11+ 也可以用 async with asyncio.timeout(2): 的写法


asyncio.run(main())

实战:异步并发爬虫

把知乎热搜爬虫改造成异步并发版:同步 requests 一次只能等一个请求,用 httpx 异步客户端可以同时抓取多个接口(pip install httpx)。

# pip install httpx
import asyncio
import httpx

# 爬虫目标:并发抓取多个API的数据
TARGET_APIS = [
    ("知乎热榜", "https://api.zhihu.com/topstory/hot-list"),
    ("GitHub趋势", "https://api.github.com/search/repositories?q=stars:>10000"),
]


async def fetch_api(client: httpx.AsyncClient, name: str, url: str) -> list:
    try:
        # 异步发请求,等待响应期间事件循环去跑其他任务
        response = await client.get(url, timeout=5)
        response.raise_for_status()
        return [name, response.json()]
    except httpx.HTTPError as err:
        return [name, f"请求失败:{err}"]


async def main():
    results = []
    # AsyncClient复用连接池,性能优于每次新建连接
    async with httpx.AsyncClient() as client:
        tasks = [fetch_api(client, name, url) for name, url in TARGET_APIS]
        results = await asyncio.gather(*tasks)  # 所有请求并发执行

    for name, data in results:
        print(f"===== {name} =====")
        print(str(data)[:200])


asyncio.run(main())

循环里的两种写法(易错点)

import asyncio
import time

async def fetch(i):
    await asyncio.sleep(1)
    return i


async def main():
    start = time.perf_counter()

    # ❌ 错误写法:循环内逐个await = 串行,3个任务耗时3秒
    # for i in range(3):
    #     await fetch(i)

    # ✅ 正确写法1:先收集协程对象,gather一次并发,耗时约1秒
    results = await asyncio.gather(fetch(0), fetch(1), fetch(2))
    print(results)  # [0, 1, 2]

    # ✅ 正确写法2:列表推导收集协程 + *解包,适合任务数量不固定的场景
    results2 = await asyncio.gather(*[fetch(i) for i in range(3)])
    print(results2)  # [0, 1, 2]
    print(f"总耗时:{time.perf_counter() - start:.2f} 秒")  # 总耗时:2.00 秒(两批各1秒)


asyncio.run(main())

适用场景与边界

场景是否适合异步说明
爬虫并发请求✅ 非常适合等待网络响应占99%时间,Playwright实战同理
Web服务(FastAPI)✅ 非常适合一个请求等待数据库时,可以处理其他请求
读写大文件、数据库IO✅ 适合asyncio.to_thread 包装同步IO即可
大量数学计算、图像处理❌ 不适合CPU一直是满的,没有「等待」可利用;应使用 multiprocessing 多进程
需要精确并行的任务❌ 不适合asyncio是单线程协作调度,不是同时执行

常见坑

  1. 忘记 await:调用协程函数只拿到协程对象,函数体根本没执行(IDE 会提示 "coroutine is never awaited")。
  2. 在协程里写同步阻塞:time.sleep(5)、requests.get(...) 会卡住整个事件循环,所有并发全部停摆;应改用 asyncio.sleep、httpx 等异步版本。
  3. 循环内逐个 await:见上文易错点,并发变串行,性能和同步没区别。
  4. 事件循环重复启动:asyncio.run() 只能在程序入口调用一次,协程内部再调用会报 RuntimeError。

记忆点:async/await 只是「标记哪里可以让出等待」,真正调度的是事件循环;想并发,必须把多个协程同时交给它(gather / create_task)。

上次编辑于:
贡献者: 15327360835
Loading...