In [42]:
import os
# import this package to handle csv files
os.makedirs(os.path.join('data'), exist_ok=True) 
# create a directory called 'data' if it doesn't exist
data_csv = os.path.join('data', 'data.csv')

with open(data_csv, 'w') as f:
    f.write('NumRooms,Alley,Price\n')  # 列名
    f.write('NA,Pave,127500\n')  # 每行表示一个数据样本
    f.write('2,NA,106000\n')
    f.write('4,NA,178100\n')
    f.write('NA,NA,140000\n')
    

# create a csv file called 'data.csv' and write some data to it
# the data includes four columns: name, age, gender, and some sample data
# the data is written to the file using the 'w' mode, which means write
# if the file already exists, it will be overwritten

In [43]:
import pandas as pd
data = pd.read_csv(data_csv)
print(data)

   NumRooms Alley   Price
0       NaN  Pave  127500
1       2.0   NaN  106000
2       4.0   NaN  178100
3       NaN   NaN  140000


In [44]:
inputs,outputs = data.iloc[:,0:2],data.iloc[:,2]
print(inputs)
print('\n')
print(outputs)
print('\n')

inputs['NumRooms'] = inputs['NumRooms'].fillna(inputs['NumRooms'].mean())

print(inputs)

   NumRooms Alley
0       NaN  Pave
1       2.0   NaN
2       4.0   NaN
3       NaN   NaN


0    127500
1    106000
2    178100
3    140000
Name: Price, dtype: int64


   NumRooms Alley
0       3.0  Pave
1       2.0   NaN
2       4.0   NaN
3       3.0   NaN


In [45]:
inputs = pd.get_dummies(inputs,dummy_na=True)
# this methods will convert categorical variables into dummy variables
# dummy_na=True will include a column for missing values

# dummy variables are binary variables that take on a value of 0 or 1, 
# based on whether the category is present or absent in the data
# this is useful for algorithms that can only deal with numerical data
# and it's usually used in machine learning algorithms that require categorical data

print(inputs)

   NumRooms  Alley_Pave  Alley_nan
0       3.0        True      False
1       2.0       False       True
2       4.0       False       True
3       3.0       False       True


In [46]:
# transformation of data into tensors for pytorch
import torch

x = torch.tensor(inputs.to_numpy(dtype=float))
y = torch.tensor(outputs.to_numpy(dtype=float))

x,y

(tensor([[3., 1., 0.],
         [2., 0., 1.],
         [4., 0., 1.],
         [3., 0., 1.]], dtype=torch.float64),
 tensor([127500., 106000., 178100., 140000.], dtype=torch.float64))

## 小结

* `pandas`软件包是Python中常用的数据分析工具中，`pandas`可以与张量兼容。
* 用`pandas`处理缺失的数据时，我们可根据情况选择用插值法和删除法。