Skip to content

Ch 9: 序列化与数据格式

  • 处理 JSON 数据
  • 处理 CSV 文件
  • 理解二进制序列化差异
  • 处理 XML/HTML

Python 的 json 模块简洁易用:

import json
# 序列化(Python 对象 → JSON 字符串)
data = {
"name": "Alice",
"age": 30,
"scores": [95, 87, 92],
"active": True,
"address": None
}
json_str = json.dumps(data, indent=2)
# {
# "name": "Alice",
# "age": 30,
# "scores": [95, 87, 92],
# "active": true,
# "address": null
# }
# 反序列化(JSON 字符串 → Python 对象)
parsed = json.loads(json_str)
print(parsed["name"]) # "Alice"
# 文件操作
with open("data.json", "w") as f:
json.dump(data, f, indent=2)
with open("data.json", "r") as f:
loaded = json.load(f)
# 自定义序列化
class Custom:
def __init__(self, value):
self.value = value
def custom_encoder(obj):
if isinstance(obj, Custom):
return {"__type__": "Custom", "value": obj.value}
raise TypeError(f"Object of type {type(obj)} is not JSON serializable")
json.dumps({"obj": Custom(42)}, default=custom_encoder)

nlohmann/json 是 C++ 最流行的 JSON 库:

// 需要安装: #include <nlohmann/json.hpp>
// 安装方式: cmake FetchContent 或 vcpkg
#include <nlohmann/json.hpp>
#include <iostream>
#include <fstream>
using json = nlohmann::json;
int main() {
// 序列化
json data = {
{"name", "Alice"},
{"age", 30},
{"scores", {95, 87, 92}},
{"active", true},
{"address", nullptr}
};
std::string str = data.dump(2); // 格式化输出,缩进2
// 反序列化
auto parsed = json::parse(str);
std::cout << parsed["name"] << std::endl; // "Alice"
// 文件操作
std::ofstream out("data.json");
out << std::setw(2) << data << std::endl;
std::ifstream in("data.json");
json loaded;
in >> loaded;
return 0;
}
#include <nlohmann/json.hpp>
using json = nlohmann::json;
int main() {
// 构造
json j;
j["name"] = "Alice";
j["age"] = 30;
j["scores"] = {95, 87, 92};
j["active"] = true;
// 嵌套
j["address"] = {
{"city", "Beijing"},
{"zip", "100000"}
};
// 访问
std::string name = j["name"]; // "Alice"
int age = j.at("age"); // 30(安全访问)
int first_score = j["scores"][0]; // 95
std::string city = j["address"]["city"]; // "Beijing"
// 检查
if (j.contains("name")) { /* ... */ }
if (j.find("age") != j.end()) { /* ... */ }
// 类型检查
if (j["age"].is_number()) { /* ... */ }
if (j["name"].is_string()) { /* ... */ }
// 迭代
for (auto& [key, value] : j.items()) {
std::cout << key << ": " << value << std::endl;
}
return 0;
}
import csv
# 读 CSV
with open("data.csv", "r", newline="") as f:
reader = csv.DictReader(f) # 字典读取
for row in reader:
print(row["name"], row["age"])
# 或列表读取
f.seek(0)
reader = csv.reader(f)
for row in reader:
name, age = row[0], row[1]
# 写 CSV
with open("output.csv", "w", newline="") as f:
writer = csv.DictWriter(f, fieldnames=["name", "age"])
writer.writeheader()
writer.writerow({"name": "Alice", "age": 30})
writer.writerows([
{"name": "Bob", "age": 25},
{"name": "Charlie", "age": 35}
])
#include <fstream>
#include <sstream>
#include <vector>
#include <string>
// 简单 CSV 解析
std::vector<std::vector<std::string>> read_csv(const std::string& filename) {
std::vector<std::vector<std::string>> rows;
std::ifstream file(filename);
std::string line;
while (std::getline(file, line)) {
std::vector<std::string> fields;
std::stringstream ss(line);
std::string field;
while (std::getline(ss, field, ',')) {
fields.push_back(field);
}
rows.push_back(fields);
}
return rows;
}
// 带引号处理的完整实现
std::vector<std::vector<std::string>> read_csv_robust(const std::string& filename) {
std::vector<std::vector<std::string>> rows;
std::ifstream file(filename);
std::string line;
while (std::getline(file, line)) {
std::vector<std::string> fields;
std::stringstream ss(line);
std::string field;
bool in_quotes = false;
for (size_t i = 0; i < line.size(); ++i) {
char c = line[i];
if (c == '"') {
in_quotes = !in_quotes;
} else if (c == ',' && !in_quotes) {
fields.push_back(field);
field.clear();
} else {
field += c;
}
}
fields.push_back(field);
rows.push_back(fields);
}
return rows;
}

使用第三方库(生产环境推荐)

Section titled “使用第三方库(生产环境推荐)”
// CSV parser 库:csv-parser, fast-cpp-csv-parser
// 或使用 pandas-style 库
import pickle
# 序列化
data = {
"name": "Alice",
"values": [1, 2, 3],
"nested": {"key": "value"}
}
bytes_data = pickle.dumps(data)
# 反序列化
loaded = pickle.loads(bytes_data)
# 文件
with open("data.pkl", "wb") as f:
pickle.dump(data, f)
with open("data.pkl", "rb") as f:
loaded = pickle.load(f)
# 选择协议
pickle.dumps(data, protocol=pickle.HIGHEST_PROTOCOL)
# 不安全!永远不要反序列化不可信的数据
# loaded = pickle.loads(untrusted_data)

C++ 没有标准序列化库,需要自定义:

#include <fstream>
#include <vector>
#include <string>
// 结构体定义
struct Data {
int id;
std::vector<int> values;
std::string name;
};
// 序列化
void serialize(const Data& d, std::ostream& os) {
// 写 id
os.write(reinterpret_cast<const char*>(&d.id), sizeof(d.id));
// 写 values
size_t values_size = d.values.size();
os.write(reinterpret_cast<const char*>(&values_size), sizeof(values_size));
os.write(reinterpret_cast<const char*>(d.values.data()),
values_size * sizeof(int));
// 写 name
size_t name_size = d.name.size();
os.write(reinterpret_cast<const char*>(&name_size), sizeof(name_size));
os.write(d.name.data(), name_size);
}
// 反序列化
void deserialize(Data& d, std::istream& is) {
// 读 id
is.read(reinterpret_cast<char*>(&d.id), sizeof(d.id));
// 读 values
size_t values_size;
is.read(reinterpret_cast<char*>(&values_size), sizeof(values_size));
d.values.resize(values_size);
is.read(reinterpret_cast<char*>(d.values.data()),
values_size * sizeof(int));
// 读 name
size_t name_size;
is.read(reinterpret_cast<char*>(&name_size), sizeof(name_size));
d.name.resize(name_size);
is.read(d.name.data(), name_size);
}
// 使用
int main() {
Data d{42, {1, 2, 3}, "Alice"};
std::ofstream out("data.bin", std::ios::binary);
serialize(d, out);
Data loaded;
std::ifstream in("data.bin", std::ios::binary);
deserialize(loaded, in);
return 0;
}

9.5 Protocol Buffers(推荐用于跨语言)

Section titled “9.5 Protocol Buffers(推荐用于跨语言)”
addressbook.proto
syntax = "proto3";
message Person {
string name = 1;
int32 id = 2;
string email = 3;
}
message AddressBook {
repeated Person people = 1;
}
Terminal window
# 安装 protoc
protoc --python_out=. addressbook.proto
protoc --cpp_out=. addressbook.proto
// C++ 使用
#include <addressbook.pb.h>
// Python 使用
import addressbook_pb2
xml.etree.ElementTree
import xml.etree.ElementTree as ET
tree = ET.parse("data.xml")
root = tree.getroot()
for child in root:
print(child.tag, child.attrib)
# Element.find, Element.findall
root.find("child")
root.findall(".//child")
# lxml(更强大)
from lxml import etree
doc = etree.parse("data.xml")
print(doc.xpath("//tag/text()"))
# HTML
from bs4 import BeautifulSoup
soup = BeautifulSoup(html_doc, 'html.parser')
print(soup.title.string)
// C++ XML 库:pugiXML, tinyxml2, RapidXML
// HTML 库:libxml2, gumbo-parser
#include <pugixml.hpp>
int main() {
pugi::xml_document doc;
doc.load_file("data.xml");
pugi::xml_node root = doc.root();
for (pugi::xml_node child : root.children("element")) {
std::cout << child.attribute("name").value() << std::endl;
}
// XPath
auto nodes = doc.select_nodes("//tag");
for (auto node : nodes) {
std::cout << node.node().text().get() << std::endl;
}
return 0;
}
  • JSON 是跨语言数据交换标准
  • nlohmann/json 是 C++ 常用 JSON 库
  • CSV 处理可以用标准库简化实现
  • 二进制序列化需要自定义格式或用 protobuf

下章预告:ch10 系统与进程。