### 2.2.1. 读取数据集

In [1]:
import os

os.makedirs(os.path.join('..', 'data'), exist_ok=True)
data_file = os.path.join('..', 'data', 'house_tiny.csv')
with open(data_file, 'w') as f:
    f.write('NumRooms,Alley,Price\n')  # 列名
    f.write('NA,Pave,127500\n')  # 每行表示一个数据样本
    f.write('2,街道,106000\n')
    f.write('4,NA,178100\n')
    f.write('NA,NA,140000\n')


In [2]:
import pandas as pd

data = pd.read_csv(data_file)
data

Unnamed: 0,NumRooms,Alley,Price
0,,Pave,127500
1,2.0,街道,106000
2,4.0,,178100
3,,,140000


### 2.2.2. 处理缺失值

In [3]:
inputs, outputs = data.iloc[:, 0:2], data.iloc[:, 2]
inputs = inputs.fillna(inputs.mean())
print(inputs)

   NumRooms Alley
0       3.0  Pave
1       2.0    街道
2       4.0   NaN
3       3.0   NaN


In [4]:
inputs = pd.get_dummies(inputs, dummy_na=True)
print(inputs)

   NumRooms  Alley_Pave  Alley_街道  Alley_nan
0       3.0           1         0          0
1       2.0           0         1          0
2       4.0           0         0          1
3       3.0           0         0          1


### 2.2.3. 转换为张量格式

In [5]:
import torch

X = torch.tensor(inputs.to_numpy(dtype=float))
y = torch.tensor(outputs.to_numpy(dtype=float))
X, y

(tensor([[3., 1., 0., 0.],
         [2., 0., 1., 0.],
         [4., 0., 0., 1.],
         [3., 0., 0., 1.]], dtype=torch.float64),
 tensor([127500., 106000., 178100., 140000.], dtype=torch.float64))

### Practice

In [6]:
# 1
import os
data_file = os.path.join('..', 'data', 'house_tiny.csv')
data = pd.read_csv(data_file)
data_isna_max = data.isna().sum().idxmax()
new_data= data.drop(data_isna_max, axis=1)
new_data

Unnamed: 0,Alley,Price
0,Pave,127500
1,街道,106000
2,,178100
3,,140000


In [7]:
# 2
import torch
inputs, outputs = new_data.iloc[:,0:-1], new_data.iloc[:,-1]
inputs = inputs.fillna(inputs.mean())
inputs = pd.get_dummies(inputs, dummy_na = True)
print(inputs)
print(outputs)
X, Y = torch.tensor(inputs.to_numpy(dtype=int)), torch.tensor(outputs.to_numpy(dtype=int))
X, Y

   Alley_Pave  Alley_街道  Alley_nan
0           1         0          0
1           0         1          0
2           0         0          1
3           0         0          1
0    127500
1    106000
2    178100
3    140000
Name: Price, dtype: int64


(tensor([[1, 0, 0],
         [0, 1, 0],
         [0, 0, 1],
         [0, 0, 1]]),
 tensor([127500, 106000, 178100, 140000]))