Skip to content

第13章 内存管理与生命周期

正确管理内存是编写高质量 pybind11 绑定的关键。本章深入讲解 GIL 机制、对象所有权和内存泄漏检测。

全局解释器锁(Global Interpreter Lock,GIL)是 Python 的核心机制,用于保护引用计数和线程安全。

#include <pybind11/pybind11.h>
namespace py = pybind11;
// 在持有 GIL 的情况下访问 Python 对象
void access_python_object(const py::object& obj) {
// 安全:GIL 保证引用计数操作是原子的
py::print("Accessing:", obj);
}
// 不安全:GIL 已释放
void unsafe_access() {
// 此时其他线程可能修改 Python 对象
// 不要在这里访问任何 py::object
}

GIL 的存在原因:

  1. 引用计数保护:Python 使用引用计数进行垃圾回收,GIL 确保引用计数的操作是原子的
  2. 线程安全:防止多线程同时修改 Python 内部状态
  3. 简化实现:无需为每个 Python 对象实现复杂的同步机制

关键洞察:GIL 限制了同一时刻只有一个线程执行 Python 字节码。对于纯 Python 代码,多线程无法实现真正的并行。但 C++ 扩展可以在释放 GIL 后执行 CPU 密集型计算。

在执行 CPU 密集型计算且不需要访问 Python 对象时,应释放 GIL 以允许其他线程执行。

#include <pybind11/pybind11.h>
#include <thread>
#include <vector>
namespace py = pybind11;
double compute_sum(const std::vector<double>& data) {
double sum = 0.0;
for (size_t i = 0; i < data.size(); ++i) {
sum += data[i];
}
return sum;
}
// 在 GIL 释放状态下执行计算密集型任务
py::object process_data(const std::vector<double>& data) {
// 创建临时的 Python 对象之前获取 GIL
py::gil_scoped_acquire gil;
double result = 0.0;
{
// 释放 GIL,允许其他 Python 线程执行
py::gil_scoped_release release;
result = compute_sum(data);
// ... 执行耗时计算 ...
}
// 重新获取 GIL 后可以创建 Python 对象
return py::float_(result);
}
// 多线程场景下释放 GIL
void parallel_processing(const std::vector<double>& data, size_t num_threads) {
std::vector<std::thread> threads;
std::vector<double> results(num_threads, 0.0);
for (size_t t = 0; t < num_threads; ++t) {
threads.emplace_back([&, t]() {
// 每个线程在开始时获取 GIL,块结束时释放
py::gil_scoped_acquire gil;
// 计算分配给该线程的数据范围
size_t start = t * (data.size() / num_threads);
size_t end = (t == num_threads - 1) ? data.size()
: (t + 1) * (data.size() / num_threads);
double local_sum = 0.0;
for (size_t i = start; i < end; ++i) {
local_sum += data[i];
}
results[t] = local_sum;
// GIL 在 gil_scoped_acquire 析构时自动释放
});
}
for (auto& t : threads) {
t.join();
}
}

关键洞察:gil_scoped_release 用于释放 GIL,让其他 Python 线程有机会执行。释放 GIL 后,不能访问任何 Python 对象(py::object、py::dict 等),否则会导致未定义行为。

在需要访问 Python 对象时,必须获取 GIL。

#include <pybind11/pybind11.h>
#include <thread>
namespace py = pybind11;
// 在 C++ 线程中获取 GIL 后访问 Python 对象
py::object fetch_data_from_python(const py::dict& data_store, const std::string& key) {
// 获取 GIL
py::gil_scoped_acquire gil;
// 现在可以安全访问 Python 对象
if (data_store.contains(key)) {
return data_store[key];
}
return py::none();
}
void background_worker(py::object callback) {
std::thread([callback]() {
// 在新线程中获取 GIL
py::gil_scoped_acquire gil;
// 调用 Python 回调
callback("background task completed");
}).detach();
}

对于需要精细控制内存的场景,可以实现自定义分配器。

#include <pybind11/pybind11.h>
#include <memory>
#include <vector>
namespace py = pybind11;
// 自定义内存池
class MemoryPool {
public:
static MemoryPool& instance() {
static MemoryPool inst;
return inst;
}
void* allocate(size_t size) {
// 使用自定义内存分配策略
return ::operator new(size);
}
void deallocate(void* ptr) {
::operator delete(ptr);
}
private:
MemoryPool() = default;
};
// 包装器类,管理 C++ 对象的生命周期
template <typename T>
class ManagedBuffer {
public:
ManagedBuffer(size_t size) : size_(size) {
data_ = static_cast<T*>(MemoryPool::instance().allocate(size * sizeof(T)));
}
~ManagedBuffer() {
MemoryPool::instance().deallocate(data_);
}
T* data() { return data_; }
size_t size() const { return size_; }
private:
T* data_;
size_t size_;
};
PYBIND11_MODULE(memory_mgmt_module, m) {
py::class_<ManagedBuffer<double>>(m, "ManagedBuffer")
.def(py::init<size_t>())
.def("data", &ManagedBuffer<double>::data, py::return_value_policy::reference_internal)
.def("size", &ManagedBuffer<double>::size);
}

13.5 对象所有权转移(return_value_policy)

Section titled “13.5 对象所有权转移(return_value_policy)”

return_value_policy 控制返回对象的所有权,是内存管理的核心概念。

#include <pybind11/pybind11.h>
#include <memory>
#include <vector>
namespace py = pybind11;
// 创建新对象(调用者获得所有权)
std::vector<int> create_vector() {
auto vec = std::make_shared<std::vector<int>>(10, 1);
return *vec; // 返回值,pybind11 默认复制
}
// 返回引用(不转移所有权)
std::vector<int>& get_global_vector() {
static std::vector<int> global_vec = {1, 2, 3};
return global_vec; // 返回引用
}
// 返回智能指针(显式转移所有权)
std::unique_ptr<std::vector<int>> create_unique_vector() {
return std::make_unique<std::vector<int>>(10, 5);
}
PYBIND11_MODULE(ownership_module, m) {
// 默认行为:复制
m.def("create_vector", &create_vector);
// 返回引用_internal:保持对象存活
m.def("get_global_vector", &get_global_vector,
py::return_value_policy::reference_internal);
// 转移所有权:unique_ptr 被提取后,原对象不再归 C++ 管理
m.def("create_unique_vector", []() {
return std::make_unique<std::vector<int>>(10, 5);
});
// 使用 move
m.def("create_unique_vector_move", []() {
auto ptr = std::make_unique<std::vector<int>>(10, 5);
std::vector<int> result = *ptr; // 复制
return result;
});
}
返回值策略说明适用场景
automatic(默认)智能指针转移所有权,其他复制一般情况
reference返回引用,不复制全局对象、成员引用
copy强制复制需要独立副本时
reference_internal返回引用,内部对象保持 alive容器元素、成员

关键洞察:return_value_policy::automatic 是默认值,但对 unique_ptr 会转移所有权。如果函数返回的是 shared_ptr,pybind11 会自动管理引用计数。需要特别小心返回内部引用的情况——如果对象被销毁,引用就会失效。

使用 AddressSanitizer、Valgrind 和 pybind11 提供的工具检测内存泄漏。

#include <pybind11/pybind11.h>
#include <pybind11/numpy.h>
#include <vector>
namespace py = pybind11;
// 模拟内存泄漏:持有引用但不释放
class LeakyCache {
public:
void store(const std::string& key, py::object value) {
py::gil_scoped_acquire gil;
cache_[key] = value; // 持有 Python 对象引用
}
py::object get(const std::string& key) {
py::gil_scoped_acquire gil;
auto it = cache_.find(key);
if (it != cache_.end()) {
return it->second;
}
return py::none();
}
private:
std::unordered_map<std::string, py::object> cache_;
};
// 正确清理:使用后主动清空
class ProperCache {
public:
void store(const std::string& key, py::object value) {
py::gil_scoped_acquire gil;
cache_[key] = value;
}
void clear() {
py::gil_scoped_acquire gil;
cache_.clear();
}
~ProperCache() {
// 析构函数中确保 GIL 可用
py::gil_scoped_acquire gil;
cache_.clear();
}
private:
std::unordered_map<std::string, py::object> cache_;
};
PYBIND11_MODULE(leak_detect_module, m) {
py::class_<LeakyCache>(m, "LeakyCache")
.def(py::init<>())
.def("store", &LeakyCache::store)
.def("get", &LeakyCache::get);
py::class_<ProperCache>(m, "ProperCache")
.def(py::init<>())
.def("store", &ProperCache::store)
.def("clear", &ProperCache::clear);
}

编译时添加 AddressSanitizer:

Terminal window
g++ -fsanitize=address -fno-omit-frame-pointer -shared -fPIC \
-std=c++17 -I$(python3 -c "import pybind11; print(pybind11.get_include())") \
-I$(python3 -c "import sysconfig; print(sysconfig.get_path('include'))") \
leak_detect_module.cpp -o leak_detect_module.so

Python 的引用计数无法处理循环引用,C++ 对象与 Python 对象之间也可能形成循环引用。

#include <pybind11/pybind11.h>
#include <memory>
namespace py = pybind11;
// 模拟循环引用问题
class Node {
public:
void set_next(std::shared_ptr<Node> next) {
next_ = next;
}
void set_prev(std::shared_ptr<Node> prev) {
prev_ = prev;
}
private:
std::shared_ptr<Node> next_;
std::shared_ptr<Node> prev_;
};
// 使用 weak_ptr 打破循环
class NodeWeak {
public:
void set_next(std::shared_ptr<NodeWeak> next) {
next_ = next;
}
// 返回 weak_ptr,不增加引用计数
std::weak_ptr<NodeWeak> get_next() const {
return next_;
}
private:
std::shared_ptr<NodeWeak> next_;
};
PYBIND11_MODULE(weakref_module, m) {
py::class_<Node, std::shared_ptr<Node>>(m, "Node")
.def(py::init<>())
.def("set_next", &Node::set_next)
.def("set_prev", &Node::set_prev);
py::class_<NodeWeak, std::shared_ptr<NodeWeak>>(m, "NodeWeak")
.def(py::init<>())
.def("set_next", &NodeWeak::set_next)
.def("get_next", &NodeWeak::get_next,
py::return_value_policy::reference_internal);
// 提供弱引用接口
m.def("create_node", []() {
return std::make_shared<NodeWeak>();
});
}
>>> import weakref_module as m
>>> # 循环引用会导致内存泄漏
>>> n1 = m.Node()
>>> n2 = m.Node()
>>> n1.set_next(n2) # n2 holds n1 via prev_ in full implementation
>>> n2.set_prev(n1)
>>> # n1 and n2 reference each other, will not be freed
>>> # 使用 weak_ptr 打破循环
>>> n1_weak = m.NodeWeak()
>>> n2_weak = m.NodeWeak()
>>> n1_weak.set_next(n2_weak)
>>> # get_next returns weak reference, no cycle formed

关键洞察:循环引用是内存泄漏的常见原因。在 C++ 绑定中使用 std::weak_ptr 打破 Python 对象与 C++ 对象之间的循环。使用 py::keep_alive() 确保生命周期依赖关系正确。

实战建议:

  • 优先使用 std::unique_ptr,明确所有权
  • 返回内部引用时使用 reference_internal
  • 大对象使用 std::shared_ptr + py::return_value_policy::automatic
  • 注意析构函数中的 GIL 获取