Python C 扩展与 FFI:ctypes、cffi、Cython 与 PyO3

Python 调用 C/C++ 的四种方案全对比:ctypes 动态库调用、cffi 安全绑定、Cython 编译加速、PyO3/Rust 扩展,含性能基准、GIL 释放、构建配置与实战案例。

当纯 Python 性能不够时,C 扩展是最直接的出路。本文覆盖四种主流方案(ctypes / cffi / Cython / PyO3),从快速调用到深度编译逐层进阶。


目录

  1. 为什么要用 C 扩展
  2. ctypes 调用动态库
  3. ctypes 高级:结构体与回调
  4. cffi 安全绑定
  5. Cython 编译加速
  6. CPython C API 入门
  7. PyO3:用 Rust 写扩展
  8. GIL 释放与并行
  9. 构建与分发
  10. 速查表与选型决策

1. 为什么要用 C 扩展

场景方案加速比
调用现成 C 库ctypes / cffi函数级
计算热点(循环)Cython10-100x
数字计算NumPy 向量化100x+(无需手写)
系统级 / 高性能PyO3 (Rust)编译语言级别

判断顺序:NumPy/Pandas 能否解决?→ 用已有的 C 扩展库?→ 才考虑自己写。


2. ctypes 调用动态库

ctypes 是标准库,直接加载 .so/.dll/.dylib:

import ctypes

# 加载 C 标准库
libc = ctypes.CDLL("libc.so.6")
libc.printf(b"hello %d\n", 42)

# 指定参数与返回类型(重要!否则默认 int 会错)
libc.strlen.argtypes = [ctypes.c_char_p]
libc.strlen.restype = ctypes.c_size_t
print(libc.strlen(b"hello"))   # 5
// mylib.c —— 要调用的库
int add(int a, int b) { return a + b; }
double distance(double x1, double y1, double x2, double y2) {
    return sqrt((x2-x1)*(x2-x1) + (y2-y1)*(y2-y1));
}
void fill_array(int *arr, int n, int value) {
    for (int i = 0; i < n; i++) arr[i] = value;
}
# 编译
gcc -shared -fPIC -o mylib.so mylib.c
lib = ctypes.CDLL("./mylib.so")
lib.add.argtypes = [ctypes.c_int, ctypes.c_int]
lib.add.restype = ctypes.c_int
print(lib.add(2, 3))   # 5

lib.distance.argtypes = [ctypes.c_double] * 4
lib.distance.restype = ctypes.c_double
print(lib.distance(0, 0, 3, 4))   # 5.0

3. ctypes 高级:结构体与回调

3.1 结构体与数组

import ctypes

class Point(ctypes.Structure):
    _fields_ = [("x", ctypes.c_double), ("y", ctypes.c_double)]

class Rect(ctypes.Structure):
    _fields_ = [("tl", Point), ("br", Point)]

lib.fill_array.argtypes = [ctypes.POINTER(ctypes.c_int), ctypes.c_int, ctypes.c_int]

# 从 Python 列表传到 C
arr = (ctypes.c_int * 5)()          # C 数组
lib.fill_array(arr, 5, 99)
print(list(arr))                    # [99, 99, 99, 99, 99]

# numpy 数组直接传(零拷贝)
import numpy as np
data = np.arange(5, dtype=np.int32)
lib.fill_array(data.ctypes.data_as(ctypes.POINTER(ctypes.c_int)), 5, 7)
print(data)                         # [7 7 7 7 7]

3.2 C 回调 Python 函数

// qsort 用回调
void c_qsort(int *arr, int n, int (*cmp)(const void*, const void*));
CMP = ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_void_p, ctypes.c_void_p)

def py_cmp(a, b):
    a = ctypes.cast(a, ctypes.POINTER(ctypes.c_int)).contents.value
    b = ctypes.cast(b, ctypes.POINTER(ctypes.c_int)).contents.value
    return a - b

lib.c_qsort.argtypes = [ctypes.POINTER(ctypes.c_int), ctypes.c_int, CMP]
arr = (ctypes.c_int * 6)(5, 3, 8, 1, 9, 2)
lib.c_qsort(arr, 6, CMP(py_cmp))
print(list(arr))   # [1, 2, 3, 5, 8, 9]

4. cffi 安全绑定

cffi 从 C 声明自动生成绑定,比 ctypes 更安全、性能更好:

pip install cffi
from cffi import FFI
ffi = FFI()

# 声明 C 接口(把 .h 内容贴进来)
ffi.cdef("""
    int add(int a, int b);
    double distance(double x1, double y1, double x2, double y2);
    void fill_array(int *arr, int n, int value);
""")

# 加载库
lib = ffi.dlopen("./mylib.so")

print(lib.add(2, 3))             # 5
lib.fill_array(ffi.new("int[5]"), 5, 42)   # 自动管理内存

# ABI vs API 模式
# ABI:ffi.dlopen(直接调符号)
# API:ffi.verify(编译期校验签名,更安全)

ctypes vs cffi 对比:

维度ctypescffi
标准库✅ 自带需安装
安全性手动指定签名声明式,编译器校验
性能一般更好
结构体手写 _fields_声明式
适用快速调用正式项目

5. Cython 编译加速

Cython 把 Python 代码编译成 C,可选择性添加静态类型:

# compute.pyx —— 编译成 C 模块
def sum_squares_loop(int n):
    cdef long long total = 0
    cdef int i
    for i in range(n):
        total += i * i
    return total
# 纯 Python 版本
def sum_squares_py(n):
    return sum(i*i for i in range(n))
# pyproject.toml / setup.py
from setuptools import setup
from Cython.Build import cythonize

setup(
    name="compute",
    ext_modules=cythonize("compute.pyx", compiler_directives={"language_level": "3"}),
)
pip install cython
pip install -e .
python -c "from compute import sum_squares_loop; print(sum_squares_loop(10**6))"

基准(1e7 循环):

实现耗时相对
纯 Python1.8s1x
Cython 无类型0.9s2x
Cython cdef int0.02s90x
NumPy 向量化0.01s180x

6. CPython C API 入门

直接写 C 扩展(Python.h),适合深度集成:

// fib.c
#define PY_SSIZE_T_CLEAN
#include <Python.h>

static PyObject* fib(PyObject *self, PyObject *args) {
    int n;
    if (!PyArg_ParseTuple(args, "i", &n)) return NULL;
    long a = 0, b = 1;
    for (int i = 0; i < n; i++) { long t = a; a = b; b = t + b; }
    return PyLong_FromLong(a);
}

static PyMethodDef methods[] = {
    {"fib", fib, METH_VARARGS, "Compute fibonacci"},
    {NULL, NULL, 0, NULL}
};

static struct PyModuleDef module = {
    PyModuleDef_HEAD_INIT, "cfib", NULL, -1, methods
};

PyMODINIT_FUNC PyInit_cfib(void) { return PyModule_Create(&module); }
# setup.py
from setuptools import setup, Extension
setup(ext_modules=[Extension("cfib", ["fib.c"])])

注意:手写 C API 要手动管理引用计数,容易泄漏/崩溃。日常首选 Cython 或 PyO3,只有极端情况才手写 C API。


7. PyO3:用 Rust 写扩展

PyO3 用 Rust 写 Python 扩展,内存安全 + 高性能,已成为主流选择:

# Cargo.toml
[lib]
name = "my_math"
crate-type = ["cdylib", "rlib"]

[dependencies]
pyo3 = { version = "0.22", features = ["extension-module", "abi3-py38"] }
// src/lib.rs
use pyo3::prelude::*;

#[pyfunction]
fn sum_squares(n: usize) -> usize {
    (0..n).map(|i| i * i).sum()
}

#[pyfunction]
fn distance(x1: f64, y1: f64, x2: f64, y2: f64) -> f64 {
    ((x2 - x1).powi(2) + (y2 - y1).powi(2)).sqrt()
}

#[pymodule]
fn my_math(m: &Bound<'_, PyModule>) -> PyResult<()> {
    m.add_function(wrap_pyfunction!(sum_squares, m)?)?;
    m.add_function(wrap_pyfunction!(distance, m)?)?;
    Ok(())
}
# 用 maturin 打包
pip install maturin
maturin develop    # 开发模式:编译并安装到当前环境
import my_math
print(my_math.sum_squares(10**7))   # 近 C 速度
print(my_math.distance(0, 0, 3, 4)) # 5.0

PyO3 进阶:接收 numpy

use numpy::{PyArray1, PyReadonlyArray1};

#[pyfunction]
fn sum_array(arr: PyReadonlyArray1<'_, f64>) -> f64 {
    arr.as_array().iter().sum()
}

8. GIL 释放与并行

CPU 密集扩展应释放 GIL 让多线程并行:

// Cython:with nogil 释放 GIL
cpdef long long par_sum(int n) nogil:
    cdef long long total = 0
    cdef int i
    for i in range(n):
        total += i
    return total
// PyO3:默认释放 GIL 支持并行
use pyo3::prelude::*;
use pyo3::sync::GILOnceCell;

#[pyfunction]
fn par_compute(py: Python<'_>, n: usize) -> usize {
    // Rust 侧多线程,Python GIL 不在 Rust 线程上
    py.allow_threads(|| {
        std::thread::scope(|s| {
            // ... 并行计算 ...
        })
    })
}
# 使用:ThreadPoolExecutor 并行调用释放 GIL 的扩展
from concurrent.futures import ThreadPoolExecutor
with ThreadPoolExecutor(4) as pool:
    results = list(pool.map(my_math.sum_squares, chunks))

关键:若扩展不释放 GIL,多线程调用仍被串行化。性能扩展必须配合 nogil/allow_threads 才能真正并行。


9. 构建与分发

9.1 平台无关分发(abi3 / 多轮子)

# maturin(Rust 扩展)
maturin build --release --out dist
pip install dist/my_math-*.whl

# 多平台构建
pip install maturin
maturin build --target aarch64-apple-darwin
maturin build --target x86_64-unknown-linux-gnu

9.2 cibuildwheel(Cython/C 扩展)

# .github/workflows/wheels.yml
name: Build wheels
on: [push, pull_request]
jobs:
  build:
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@v4
      - uses: pypa/cibuildwheel@v2
      - uses: actions/upload-artifact@v4
        with:
          path: ./wheelhouse/*.whl
# 本地构建所有平台轮子
pip install cibuildwheel
cibuildwheel --platform linux macos windows

10. 速查表与选型决策

需求推荐
快速调现成库ctypes
正式项目绑定 C 库cffi
加速 Python 热点Cython
新写高性能扩展PyO3 (Rust)
极速数组计算NumPy 向量化
解析/生成二进制ctypes + struct

选型决策树:

要调用现成的 C 库?
├── 是 → 需要编译期安全? → cffi : ctypes
└── 否 → 是要加速 Python 热点?
         ├── 是 → 愿写 Rust? → PyO3
         │         ├── 否 → Cython
         └── 否 → 试试 NumPy/已有库先

最佳实践:

  1. 先向量化:90% 场景 NumPy 就够,别急着写 C。
  2. 边界小:C 扩展只放热点函数,胶水留 Python。
  3. 类型声明:ctypes 必须设 argtypes/restype。
  4. 释放 GIL:CPU 密集 + 需要并行时。
  5. 多平台轮子:发布用 cibuildwheel / maturin。
  6. 测试:C 扩展同样要单元测试 + 内存检查。

一句话记忆:快速绑定选 ctypes、安全绑定选 cffi、Python 提速选 Cython、重写扩展选 PyO3;写之前先问 NumPy 能不能解决。

延伸阅读

理解 FFI 不是「必须手写 C」,而是「知道性能边界在哪、用什么工具去够」。从 NumPy 到 PyO3,你的工具箱越丰富,性能瓶颈就越少。

继续阅读

探索更多技术文章

浏览归档,发现更多关于系统设计、工具链和工程实践的内容。

全部文章 返回首页

「python」更多文章

  1. Python 微服务架构:从单体拆分到服务治理
  2. Python 网络爬虫与自动化:从 requests 到 Playwright
  3. Python 库与 API 设计:从包结构到向后兼容