Ch 9: 序列化与数据格式
- 处理 JSON 数据
- 处理 CSV 文件
- 理解二进制序列化差异
- 处理 XML/HTML
9.1 Python JSON 模块
Section titled “9.1 Python JSON 模块”Python 的 json 模块简洁易用:
import json
# 序列化(Python 对象 → JSON 字符串)data = { "name": "Alice", "age": 30, "scores": [95, 87, 92], "active": True, "address": None}
json_str = json.dumps(data, indent=2)# {# "name": "Alice",# "age": 30,# "scores": [95, 87, 92],# "active": true,# "address": null# }
# 反序列化(JSON 字符串 → Python 对象)parsed = json.loads(json_str)print(parsed["name"]) # "Alice"
# 文件操作with open("data.json", "w") as f: json.dump(data, f, indent=2)
with open("data.json", "r") as f: loaded = json.load(f)
# 自定义序列化class Custom: def __init__(self, value): self.value = value
def custom_encoder(obj): if isinstance(obj, Custom): return {"__type__": "Custom", "value": obj.value} raise TypeError(f"Object of type {type(obj)} is not JSON serializable")
json.dumps({"obj": Custom(42)}, default=custom_encoder)9.2 C++ nlohmann/json
Section titled “9.2 C++ nlohmann/json”nlohmann/json 是 C++ 最流行的 JSON 库:
// 需要安装: #include <nlohmann/json.hpp>// 安装方式: cmake FetchContent 或 vcpkg
#include <nlohmann/json.hpp>#include <iostream>#include <fstream>
using json = nlohmann::json;
int main() { // 序列化 json data = { {"name", "Alice"}, {"age", 30}, {"scores", {95, 87, 92}}, {"active", true}, {"address", nullptr} };
std::string str = data.dump(2); // 格式化输出,缩进2
// 反序列化 auto parsed = json::parse(str); std::cout << parsed["name"] << std::endl; // "Alice"
// 文件操作 std::ofstream out("data.json"); out << std::setw(2) << data << std::endl;
std::ifstream in("data.json"); json loaded; in >> loaded;
return 0;}JSON 构造和访问
Section titled “JSON 构造和访问”#include <nlohmann/json.hpp>
using json = nlohmann::json;
int main() { // 构造 json j; j["name"] = "Alice"; j["age"] = 30; j["scores"] = {95, 87, 92}; j["active"] = true;
// 嵌套 j["address"] = { {"city", "Beijing"}, {"zip", "100000"} };
// 访问 std::string name = j["name"]; // "Alice" int age = j.at("age"); // 30(安全访问) int first_score = j["scores"][0]; // 95 std::string city = j["address"]["city"]; // "Beijing"
// 检查 if (j.contains("name")) { /* ... */ } if (j.find("age") != j.end()) { /* ... */ }
// 类型检查 if (j["age"].is_number()) { /* ... */ } if (j["name"].is_string()) { /* ... */ }
// 迭代 for (auto& [key, value] : j.items()) { std::cout << key << ": " << value << std::endl; }
return 0;}9.3 CSV 处理
Section titled “9.3 CSV 处理”Python csv 模块
Section titled “Python csv 模块”import csv
# 读 CSVwith open("data.csv", "r", newline="") as f: reader = csv.DictReader(f) # 字典读取 for row in reader: print(row["name"], row["age"])
# 或列表读取 f.seek(0) reader = csv.reader(f) for row in reader: name, age = row[0], row[1]
# 写 CSVwith open("output.csv", "w", newline="") as f: writer = csv.DictWriter(f, fieldnames=["name", "age"]) writer.writeheader() writer.writerow({"name": "Alice", "age": 30}) writer.writerows([ {"name": "Bob", "age": 25}, {"name": "Charlie", "age": 35} ])C++ CSV 处理
Section titled “C++ CSV 处理”#include <fstream>#include <sstream>#include <vector>#include <string>
// 简单 CSV 解析std::vector<std::vector<std::string>> read_csv(const std::string& filename) { std::vector<std::vector<std::string>> rows; std::ifstream file(filename);
std::string line; while (std::getline(file, line)) { std::vector<std::string> fields; std::stringstream ss(line); std::string field;
while (std::getline(ss, field, ',')) { fields.push_back(field); } rows.push_back(fields); }
return rows;}
// 带引号处理的完整实现std::vector<std::vector<std::string>> read_csv_robust(const std::string& filename) { std::vector<std::vector<std::string>> rows; std::ifstream file(filename);
std::string line; while (std::getline(file, line)) { std::vector<std::string> fields; std::stringstream ss(line); std::string field; bool in_quotes = false;
for (size_t i = 0; i < line.size(); ++i) { char c = line[i]; if (c == '"') { in_quotes = !in_quotes; } else if (c == ',' && !in_quotes) { fields.push_back(field); field.clear(); } else { field += c; } } fields.push_back(field); rows.push_back(fields); }
return rows;}使用第三方库(生产环境推荐)
Section titled “使用第三方库(生产环境推荐)”// CSV parser 库:csv-parser, fast-cpp-csv-parser// 或使用 pandas-style 库9.4 Pickle vs 二进制序列化
Section titled “9.4 Pickle vs 二进制序列化”Python pickle
Section titled “Python pickle”import pickle
# 序列化data = { "name": "Alice", "values": [1, 2, 3], "nested": {"key": "value"}}
bytes_data = pickle.dumps(data)
# 反序列化loaded = pickle.loads(bytes_data)
# 文件with open("data.pkl", "wb") as f: pickle.dump(data, f)
with open("data.pkl", "rb") as f: loaded = pickle.load(f)
# 选择协议pickle.dumps(data, protocol=pickle.HIGHEST_PROTOCOL)
# 不安全!永远不要反序列化不可信的数据# loaded = pickle.loads(untrusted_data)C++ 二进制序列化
Section titled “C++ 二进制序列化”C++ 没有标准序列化库,需要自定义:
#include <fstream>#include <vector>#include <string>
// 结构体定义struct Data { int id; std::vector<int> values; std::string name;};
// 序列化void serialize(const Data& d, std::ostream& os) { // 写 id os.write(reinterpret_cast<const char*>(&d.id), sizeof(d.id));
// 写 values size_t values_size = d.values.size(); os.write(reinterpret_cast<const char*>(&values_size), sizeof(values_size)); os.write(reinterpret_cast<const char*>(d.values.data()), values_size * sizeof(int));
// 写 name size_t name_size = d.name.size(); os.write(reinterpret_cast<const char*>(&name_size), sizeof(name_size)); os.write(d.name.data(), name_size);}
// 反序列化void deserialize(Data& d, std::istream& is) { // 读 id is.read(reinterpret_cast<char*>(&d.id), sizeof(d.id));
// 读 values size_t values_size; is.read(reinterpret_cast<char*>(&values_size), sizeof(values_size)); d.values.resize(values_size); is.read(reinterpret_cast<char*>(d.values.data()), values_size * sizeof(int));
// 读 name size_t name_size; is.read(reinterpret_cast<char*>(&name_size), sizeof(name_size)); d.name.resize(name_size); is.read(d.name.data(), name_size);}
// 使用int main() { Data d{42, {1, 2, 3}, "Alice"};
std::ofstream out("data.bin", std::ios::binary); serialize(d, out);
Data loaded; std::ifstream in("data.bin", std::ios::binary); deserialize(loaded, in);
return 0;}9.5 Protocol Buffers(推荐用于跨语言)
Section titled “9.5 Protocol Buffers(推荐用于跨语言)”syntax = "proto3";
message Person { string name = 1; int32 id = 2; string email = 3;}
message AddressBook { repeated Person people = 1;}# 安装 protocprotoc --python_out=. addressbook.protoprotoc --cpp_out=. addressbook.proto// C++ 使用#include <addressbook.pb.h>
// Python 使用import addressbook_pb29.6 XML/HTML 处理
Section titled “9.6 XML/HTML 处理”Python
Section titled “Python”import xml.etree.ElementTree as ET
tree = ET.parse("data.xml")root = tree.getroot()
for child in root: print(child.tag, child.attrib)
# Element.find, Element.findallroot.find("child")root.findall(".//child")
# lxml(更强大)from lxml import etreedoc = etree.parse("data.xml")print(doc.xpath("//tag/text()"))
# HTMLfrom bs4 import BeautifulSoupsoup = BeautifulSoup(html_doc, 'html.parser')print(soup.title.string)// C++ XML 库:pugiXML, tinyxml2, RapidXML// HTML 库:libxml2, gumbo-parser
#include <pugixml.hpp>
int main() { pugi::xml_document doc; doc.load_file("data.xml");
pugi::xml_node root = doc.root();
for (pugi::xml_node child : root.children("element")) { std::cout << child.attribute("name").value() << std::endl; }
// XPath auto nodes = doc.select_nodes("//tag"); for (auto node : nodes) { std::cout << node.node().text().get() << std::endl; }
return 0;}- JSON 是跨语言数据交换标准
- nlohmann/json 是 C++ 常用 JSON 库
- CSV 处理可以用标准库简化实现
- 二进制序列化需要自定义格式或用 protobuf
下章预告:ch10 系统与进程。