核心内容
将 DataFrame 转换为内存中的 Python 对象:字典、记录数组、NumPy 数组、列表。
一、to_dict() — 转换为字典
import pandas as pd
import numpy as np
df = pd.DataFrame({
'id': [1, 2, 3],
'name': ['Alice', 'Bob', 'Charlie'],
'score': [95.5, 88.0, 76.25]
})
# 默认 orient='dict':{列名: {索引: 值}}
d1 = df.to_dict()
# {'id': {0: 1, 1: 2, 2: 3}, 'name': {0: 'Alice', ...}}
# 记录列表:[{列名: 值}, ...]
d2 = df.to_dict(orient='records')
# [{'id': 1, 'name': 'Alice', 'score': 95.5}, ...]
# 按行:{索引: {列名: 值}}
d3 = df.to_dict(orient='index')
# {0: {'id': 1, 'name': 'Alice', ...}, ...}
# 列表:{列名: [值, ...]}
d4 = df.to_dict(orient='list')
# {'id': [1, 2, 3], 'name': ['Alice', 'Bob', 'Charlie'], ...}
# 分割:{'index': [...], 'columns': [...], 'data': [[...]]}
d5 = df.to_dict(orient='split')
# 元组:{(行, 列): 值}
d6 = df.to_dict(orient='tight')
# 系列:{列名: Series}
d7 = df.to_dict(orient='series')| orient | 结构 |
|---|---|
dict(默认) | {列: {行: 值}} |
list | {列: [值]} |
series | {列: Series} |
split | {'index', 'columns', 'data'} |
records | [{列: 值}] |
index | {行: {列: 值}} |
tight | {'index','columns','data','index_names','column_names'} |
# 复杂数据类型处理
df2 = pd.DataFrame({'a': [1, 2], 'b': [pd.Timestamp('2025-01-01'), pd.Timestamp('2025-01-02')]})
# into 参数指定输出类型
from collections import OrderedDict, defaultdict
df2.to_dict(into=OrderedDict)
df2.to_dict(into=defaultdict(list), orient='list')二、to_records() — 转换为记录数组
# 转换为 NumPy 结构化数组(record array)
rec = df.to_records()
# rec[0] → (0, 1, 'Alice', 95.5)
# 参数
df.to_records(
index=True, # 是否包含索引(默认包含,作为第一列)
column_dtypes=None, # 列 dtype 映射
index_dtypes=None # 索引 dtype
)
# 不带索引
rec_no_index = df.to_records(index=False)
# 指定 dtype
rec = df.to_records(
index=False,
column_dtypes={'id': np.int32, 'score': np.float32}
)
# 可以像 ndarray 一样操作
rec[0] # 第一行
rec['name'] # 字段访问
type(rec) # numpy.recarray# 从记录数组创建 DataFrame
df_back = pd.DataFrame.from_records(rec)三、to_numpy() — 转换为 NumPy 数组
# 基础转换
arr = df.to_numpy()
# array([[1, 'Alice', 95.5], [2, 'Bob', 88.0]], dtype=object)
# 参数
df.to_numpy(
dtype=None, # 指定 dtype(混合类型时可用)
copy=False, # 是否复制(False 时可能共享内存)
na_value=None # 缺失值替换值
)
# 纯数值 DataFrame → 数值数组
numeric_df = df[['id', 'score']]
arr_num = numeric_df.to_numpy() # dtype=float64
# 指定 dtype
arr = df[['id', 'score']].to_numpy(dtype=np.float32)
# 缺失值处理
arr = df[['score']].to_numpy(na_value=-1)注意
- 混合类型(数值 + 字符串)时,
to_numpy()默认返回dtype=objectcopy=False时返回的数组可能与 DataFrame 共享内存,修改可能影响原数据
与 .values 对比
df.values # 旧 API,不推荐
df.to_numpy() # 推荐 API四、to_list() — 转换为列表
import pandas as pd
s = pd.Series([1, 2, 3, 4])
lst = s.to_list()
# [1, 2, 3, 4]
# Series 转列表
df['id'].to_list() # [1, 2, 3]
# 等效于 list(s)
list(s)说明
to_list()是 Series 的方法(DataFrame 无to_list());DataFrame 转列表可先to_numpy()再.tolist():
# DataFrame → 嵌套列表
df[['id', 'score']].to_numpy().tolist()
# [[1, 95.5], [2, 88.0], [3, 76.25]]对象转换对比
| 方法 | 输出类型 | 适用场景 |
|---|---|---|
to_dict() | dict | API 数据、配置、JSON 前处理 |
to_records() | numpy.recarray | 数值计算、SQL 插入 |
to_numpy() | numpy.ndarray | 大规模数值计算 |
to_list() | list | 简单序列、迭代 |
相关笔记
- 23.6 文本格式导出 - to_json 与 to_dict 关系密切
- 23.5 通用字符串与剪贴板 - 字符串级导出