核心内容

将 DataFrame 转换为内存中的 Python 对象:字典、记录数组、NumPy 数组、列表。

一、to_dict() — 转换为字典

import pandas as pd
import numpy as np
 
df = pd.DataFrame({
    'id': [1, 2, 3],
    'name': ['Alice', 'Bob', 'Charlie'],
    'score': [95.5, 88.0, 76.25]
})
 
# 默认 orient='dict':{列名: {索引: 值}}
d1 = df.to_dict()
# {'id': {0: 1, 1: 2, 2: 3}, 'name': {0: 'Alice', ...}}
 
# 记录列表:[{列名: 值}, ...]
d2 = df.to_dict(orient='records')
# [{'id': 1, 'name': 'Alice', 'score': 95.5}, ...]
 
# 按行:{索引: {列名: 值}}
d3 = df.to_dict(orient='index')
# {0: {'id': 1, 'name': 'Alice', ...}, ...}
 
# 列表:{列名: [值, ...]}
d4 = df.to_dict(orient='list')
# {'id': [1, 2, 3], 'name': ['Alice', 'Bob', 'Charlie'], ...}
 
# 分割:{'index': [...], 'columns': [...], 'data': [[...]]}
d5 = df.to_dict(orient='split')
 
# 元组:{(行, 列): 值}
d6 = df.to_dict(orient='tight')
 
# 系列:{列名: Series}
d7 = df.to_dict(orient='series')
orient结构
dict(默认){列: {行: 值}}
list{列: [值]}
series{列: Series}
split{'index', 'columns', 'data'}
records[{列: 值}]
index{行: {列: 值}}
tight{'index','columns','data','index_names','column_names'}
# 复杂数据类型处理
df2 = pd.DataFrame({'a': [1, 2], 'b': [pd.Timestamp('2025-01-01'), pd.Timestamp('2025-01-02')]})
 
# into 参数指定输出类型
from collections import OrderedDict, defaultdict
df2.to_dict(into=OrderedDict)
df2.to_dict(into=defaultdict(list), orient='list')

二、to_records() — 转换为记录数组

# 转换为 NumPy 结构化数组(record array)
rec = df.to_records()
# rec[0] → (0, 1, 'Alice', 95.5)
 
# 参数
df.to_records(
    index=True,        # 是否包含索引(默认包含,作为第一列)
    column_dtypes=None,  # 列 dtype 映射
    index_dtypes=None    # 索引 dtype
)
 
# 不带索引
rec_no_index = df.to_records(index=False)
 
# 指定 dtype
rec = df.to_records(
    index=False,
    column_dtypes={'id': np.int32, 'score': np.float32}
)
 
# 可以像 ndarray 一样操作
rec[0]              # 第一行
rec['name']         # 字段访问
type(rec)           # numpy.recarray
# 从记录数组创建 DataFrame
df_back = pd.DataFrame.from_records(rec)

三、to_numpy() — 转换为 NumPy 数组

# 基础转换
arr = df.to_numpy()
# array([[1, 'Alice', 95.5], [2, 'Bob', 88.0]], dtype=object)
 
# 参数
df.to_numpy(
    dtype=None,      # 指定 dtype(混合类型时可用)
    copy=False,      # 是否复制(False 时可能共享内存)
    na_value=None    # 缺失值替换值
)
 
# 纯数值 DataFrame → 数值数组
numeric_df = df[['id', 'score']]
arr_num = numeric_df.to_numpy()          # dtype=float64
 
# 指定 dtype
arr = df[['id', 'score']].to_numpy(dtype=np.float32)
 
# 缺失值处理
arr = df[['score']].to_numpy(na_value=-1)

注意

  • 混合类型(数值 + 字符串)时,to_numpy() 默认返回 dtype=object
  • copy=False 时返回的数组可能与 DataFrame 共享内存,修改可能影响原数据

与 .values 对比

df.values              # 旧 API,不推荐
df.to_numpy()          # 推荐 API

四、to_list() — 转换为列表

import pandas as pd
 
s = pd.Series([1, 2, 3, 4])
lst = s.to_list()
# [1, 2, 3, 4]
 
# Series 转列表
df['id'].to_list()     # [1, 2, 3]
 
# 等效于 list(s)
list(s)

说明

to_list() 是 Series 的方法(DataFrame 无 to_list());DataFrame 转列表可先 to_numpy() 再 .tolist():

# DataFrame → 嵌套列表
df[['id', 'score']].to_numpy().tolist()
# [[1, 95.5], [2, 88.0], [3, 76.25]]

对象转换对比

方法输出类型适用场景
to_dict()dictAPI 数据、配置、JSON 前处理
to_records()numpy.recarray数值计算、SQL 插入
to_numpy()numpy.ndarray大规模数值计算
to_list()list简单序列、迭代

相关笔记