# Create Numpy Array

In [2]:
import numpy as np

np_array_1d = np.array([1,2,3]) 
print(type(np_array_1d))
print(np_array_1d.shape)
print(np_array_1d)

np_array_2d = np.array([[1, 2], [3, 4]]) 
print(type(np_array_2d))
print(np_array_2d.shape)
print(np_array_2d)
print(np_array_2d.ndim)
print(np_array_2d.size)

<class 'numpy.ndarray'>
(3,)
[1 2 3]
<class 'numpy.ndarray'>
(2, 2)
[[1 2]
 [3 4]]
2
4


In [10]:
import numpy as np

x = np.array([1, 2])   # Let numpy choose the datatype
print(x.dtype)         # Prints "int64"

x = np.array([1.0, 2.0])   # Let numpy choose the datatype
print(x.dtype)             # Prints "float64"

x = np.array([1, 2], dtype=np.int64)   # Force a particular datatype
print(x.dtype)                         # Prints "int64"

int64
float64
int64


In [3]:
import numpy as np
np_array_zeros = np.zeros([3,4])
np_array_ones = np.ones([2,3])
np_array_full = np.full([3,2], 5)
np_array_eye = np.eye(2)   
np_random = np.random.random((2,2))
np_arange = np.arange(4) # given number from 0~3
np_linspace = np.linspace(0, 2*np.pi, 5) # given 5 value start from 0 in steps: 2pi

print(np_array_zeros)
print(np_array_ones)
print(np_array_full)
print(np_array_eye)
print(np_random)
print(np_arange)
print(np_linspace)

#arr2[1,1] = np.nan  # not a number
#arr2[1,2] = np.inf  # infinite

[[0. 0. 0. 0.]
 [0. 0. 0. 0.]
 [0. 0. 0. 0.]]
[[1. 1. 1.]
 [1. 1. 1.]]
[[5 5]
 [5 5]
 [5 5]]
[[1. 0.]
 [0. 1.]]
[[0.24742217 0.75976284]
 [0.72244539 0.11792284]]
[0 1 2 3]
[0.         1.57079633 3.14159265 4.71238898 6.28318531]


# Basic Arithmetic

In [4]:
import numpy as np

x = np.array([[1,2],[3,4]], dtype=np.float64)
y = np.array([[5,6],[7,8]], dtype=np.float64)

# Elementwise sum; both produce the array
# [[ 6.0  8.0]
#  [10.0 12.0]]
print(x + y)
print(np.add(x, y))
print('==============')
# Elementwise difference; both produce the array
# [[-4.0 -4.0]
#  [-4.0 -4.0]]
print(x - y)
print(np.subtract(x, y))
print('==============')

# Elementwise product; both produce the array
# [[ 5.0 12.0]
#  [21.0 32.0]]
print(x * y)
print(np.multiply(x, y))
print('==============')

# Elementwise division; both produce the array
# [[ 0.2         0.33333333]
#  [ 0.42857143  0.5       ]]
print(x / y)
print(np.divide(x, y))
print('==============')

# Elementwise square root; produces the array
# [[ 1.          1.41421356]
#  [ 1.73205081  2.        ]]
print(np.sqrt(x))


[[ 6.  8.]
 [10. 12.]]
[[ 6.  8.]
 [10. 12.]]
[[-4. -4.]
 [-4. -4.]]
[[-4. -4.]
 [-4. -4.]]
[[ 5. 12.]
 [21. 32.]]
[[ 5. 12.]
 [21. 32.]]
[[0.2        0.33333333]
 [0.42857143 0.5       ]]
[[0.2        0.33333333]
 [0.42857143 0.5       ]]
[[1.         1.41421356]
 [1.73205081 2.        ]]


# Matrix Algebra

In [1]:
import numpy as np

x = np.array([[1,2],[3,4]])
y = np.array([[5,6],[7,8]])

v = np.array([9,10])
w = np.array([11, 12])

# Inner product of vectors; both produce 219
print(v.dot(w))
print(np.dot(v, w))
print('==============')
# Matrix / vector product; both produce the rank 1 array [29 67]
print(x.dot(v))
print(np.matmul(x, v))
print('==============')
# Matrix / matrix product; both produce the rank 2 array
# [[19 22]
#  [43 50]]
print(x.dot(y))
print(np.dot(x, y))
print('==============')

# Matrix transpose
print(x)
print(x.T)
print('==============')

# Matrix inverse
print(x)
print(np.linalg.inv(x))
print('==============')

# Matrix determinant
print(x)
print(np.linalg.det(x))

print('==============')
# SVD
U, S, V = np.linalg.svd(x)
print(U)
print(S)
print(V)


219
219
[29 67]
[29 67]
[[19 22]
 [43 50]]
[[19 22]
 [43 50]]
[[1 2]
 [3 4]]
[[1 3]
 [2 4]]
[[1 2]
 [3 4]]
[[-2.   1. ]
 [ 1.5 -0.5]]
[[1 2]
 [3 4]]
-2.0000000000000004
[[-0.40455358 -0.9145143 ]
 [-0.9145143   0.40455358]]
[5.4649857  0.36596619]
[[-0.57604844 -0.81741556]
 [ 0.81741556 -0.57604844]]


# Comparison

In [33]:
a = np.array([[1, 1, 2], [2, 3, 5]])
b = np.array([[1, 3, 2], [3, 3, 1]])

print(a == b)
print(a >= b)

[[ True False  True]
 [False  True False]]
[[ True False  True]
 [False  True  True]]


# Sorting

In [9]:
a = np.array([[1, 5, 2], [7, 1, 3]])

print(np.sort(a, axis=0))
print(np.sort(a, axis=1))
print('======')

b = np.array([[1, 5, 2], [7, 1, 3]])
b.sort(axis=0)
print(b)

[[1 1 2]
 [7 5 3]]
[[1 2 5]
 [1 3 7]]
[[1 1 2]
 [7 5 3]]


# Statistical Computations

In [17]:
x = np.array([[1,2],[3,4]], dtype=np.float64)

print("Mean value is: {}".format(x.mean()))
print("Max value is: {}".format(x.max()))
print("Min value is: {}".format(x.min()))
print("Std value is: {}".format(x.std()))
print("Var value is: {}".format(x.var()))
print("Sum value is: {}".format(x.sum()))

# Row wise and column wise min
print("column wise sum: {}".format(np.sum(x, axis=0)))
print("row wise sum: {}".format(np.sum(x, axis=1)))
print("column wise minimum: {}".format(np.min(x, axis=0)))
print("row wise minimum: {}".format(np.min(x, axis=1)))

print("Cumulative sum: {}".format(np.cumsum(x)))

Mean value is: 2.5
Max value is: 4.0
Min value is: 1.0
Std value is: 1.118033988749895
Var value is: 1.25
Sum value is: 10.0
column wise sum: [4. 6.]
row wise sum: [3. 7.]
column wise minimum: [1. 2.]
row wise minimum: [1. 3.]
Cumulative sum: [ 1.  3.  6. 10.]


In [46]:
x = np.array([[1,4],[2,3]], dtype=np.float64)
print("argmax(axis=0): {}".format(np.argmax(x, axis=0)))
print("argmax(axis=1): {}".format(np.argmax(x, axis=1)))


argmax(axis=0): [1 0]
argmax(axis=1): [1 1]


#  Indexing & Slicing

In [22]:
import numpy as np

# Create the following rank 2 array with shape (3, 4)
# [[ 1  2  3  4]
#  [ 5  6  7  8]
#  [ 9 10 11 12]]
a = np.array([[1,2,3,4], [5,6,7,8], [9,10,11,12]])
#a = np.arange(1, 13).reshape(3,4)


# Use slicing to pull out the subarray consisting of the first 2 rows
# and columns 1 and 2; b is the following array of shape (2, 2):
# [[2 3]
#  [6 7]]
print(a[:2, 1:3])

# A slice of an array is a view into the same data, so modifying it
# will modify the original array.
print(a[0, 1])   # Prints "2"
a[0, 1] = 77     # b[0, 0] is the same piece of data as a[0, 1]
print(a[0, 1])   # Prints "77

[[2 3]
 [6 7]]
2
77


In [8]:
import numpy as np

# Create the following rank 2 array with shape (3, 4)
# [[ 1  2  3  4]
#  [ 5  6  7  8]
#  [ 9 10 11 12]]
a = np.array([[1,2,3,4], [5,6,7,8], [9,10,11,12]])

# Two ways of accessing the data in the middle row of the array.
# Mixing integer indexing with slices yields an array of lower rank,
# while using only slices yields an array of the same rank as the
# original array:
row_r1 = a[1, :]    # Rank 1 view of the second row of a
row_r2 = a[1:2, :]  # Rank 2 view of the second row of a
print(row_r1, row_r1.shape)  # Prints "[5 6 7 8] (4,)"
print(row_r2, row_r2.shape)  # Prints "[[5 6 7 8]] (1, 4)"

# We can make the same distinction when accessing columns of an array:
col_r1 = a[:, 1]
col_r2 = a[:, 1:2]
print(col_r1, col_r1.shape)  # Prints "[ 2  6 10] (3,)"
print(col_r2, col_r2.shape)  # Prints "[[ 2]
                             #          [ 6]
                             #          [10]] (3, 1)"


[5 6 7 8] (4,)
[[5 6 7 8]] (1, 4)
[ 2  6 10] (3,)
[[ 2]
 [ 6]
 [10]] (3, 1)


In [23]:
import numpy as np 
a = np.arange(10) 
print(a)
b = a[2:7:2] 
print(b)
print(a[2:])


a = np.array([[1,2,3],[3,4,5],[4,5,6]]) 
print(a)  

# slice items starting from index
print('Now we will slice the array from the index a[1:]') 
print(a[1:])


print('Our array is:') 
print(a) 

# this returns array of items in the second column 
print('The items in the second column are:')  
print(a[:,1]) 

# Now we will slice all items from the second row 
print('The items in the second row are:') 
print(a[1,:]) 

# Now we will slice all items from column 1 onwards 
print('The items column 1 onwards are:') 
print(a[:,1:])

[0 1 2 3 4 5 6 7 8 9]
[2 4 6]
[2 3 4 5 6 7 8 9]
[[1 2 3]
 [3 4 5]
 [4 5 6]]
Now we will slice the array from the index a[1:]
[[3 4 5]
 [4 5 6]]
Our array is:
[[1 2 3]
 [3 4 5]
 [4 5 6]]
The items in the second column are:
[2 4 5]
The items in the second row are:
[3 4 5]
The items column 1 onwards are:
[[2 3]
 [4 5]
 [5 6]]


# Boolean index

In [11]:
import numpy as np

a = np.array([[1,2], [3, 4], [5, 6]])

bool_idx = (a > 2)   # Find the elements of a that are bigger than 2;
                     # this returns a numpy array of Booleans of the same
                     # shape as a, where each slot of bool_idx tells
                     # whether that element of a is > 2.

print(bool_idx)      # Prints "[[False False]
                     #          [ True  True]
                     #          [ True  True]]"

# We use boolean array indexing to construct a rank 1 array
# consisting of the elements of a corresponding to the True values
# of bool_idx
print(a[bool_idx])  # Prints "[3 4 5 6]"

# We can do all of the above in a single concise statement:
print(a[a > 2])     # Prints "[3 4 5 6]"


[[False False]
 [ True  True]
 [ True  True]]
[3 4 5 6]
[3 4 5 6]


# Add/Remove Elements

In [43]:
a = np.array([0, 1, 2])
b = np.array([3, 4, 5])

a = np.append(a, 10)
print(a)

a = np.append(a, b)
print(a)



[ 0  1  2 10]
[ 0  1  2 10  3  4  5]


# Conversion between Python List/Tuple and Numpy Array

In [15]:
my_list = [1,2,3]
my_tuple = (1,2,3) 
numpy_a = np.asarray(my_tuple) 
numpy_b = np.asarray(my_list) 

print(numpy_a)
print(numpy_b)

list_a = list(numpy_a)
list_b = list(numpy_b)
print(list_a)
print(list_b)


[1 2 3]
[1 2 3]
[1, 2, 3]
[1, 2, 3]


# Conversion between Numpy and Pandas

In [13]:
import pandas as pd

# load csv file
df = pd.read_csv('na_demo.csv')

print('change dataframe to numpy array')
numpy_array = np.array(df)
print(numpy_array)

print('change numpy array to dataframe')
df_from_numpy = pd.DataFrame(numpy_array)
print(df_from_numpy)

change dataframe to numpy array
[[40.0 3.0 800 'old']
 [29.0 5.0 700 'young']
 [33.0 2.0 670 'young']
 [nan 2.0 770 'old']
 [nan nan 870 'young']]
change numpy array to dataframe
     0    1    2      3
0   40    3  800    old
1   29    5  700  young
2   33    2  670  young
3  NaN    2  770    old
4  NaN  NaN  870  young


# Broadcasting  
ref: https://jakevdp.github.io/PythonDataScienceHandbook/02.05-computation-on-arrays-broadcasting.html

In [25]:
import numpy as np
a = np.array([0, 1, 2])
b = a + 3
print(b)

M = np.ones((3, 3))
print(M + a)


[3 4 5]
[[1. 2. 3.]
 [1. 2. 3.]
 [1. 2. 3.]]


In [26]:
a = np.arange(3)
b = np.arange(3)[:, np.newaxis]

print(a)
print(b)
print(a+b)

[0 1 2]
[[0]
 [1]
 [2]]
[[0 1 2]
 [1 2 3]
 [2 3 4]]


# Reshape & Flatten

In [29]:
import numpy as np
a=np.array([1,2,3,4,5,6,7,8,9,10,11,12])  
b=np.reshape(a,(2,-1))  
c=np.reshape(a,(2,2,-1))  
d=np.reshape(a,(2,3,-1))
print(a)
print('=========')
print(b)
print('=========')
print(c)
print('=========')
print(d)

[ 1  2  3  4  5  6  7  8  9 10 11 12]
[[ 1  2  3  4  5  6]
 [ 7  8  9 10 11 12]]
[[[ 1  2  3]
  [ 4  5  6]]

 [[ 7  8  9]
  [10 11 12]]]
[[[ 1  2]
  [ 3  4]
  [ 5  6]]

 [[ 7  8]
  [ 9 10]
  [11 12]]]


In [31]:
a=np.array([[1,2],[3,4],[5,6]])
print(a.flatten())

[1 2 3 4 5 6]


# Stack  
ref: https://blog.csdn.net/qq_17550379/article/details/78934529

In [2]:
import numpy as np
a = np.array([1, 2, 3])
b = np.array([2, 3, 4])

print('a shape {}, b shape {}'.format(a.shape, b.shape))

print(np.stack((a, b), axis=0).shape)
print(np.stack((a, b), axis=1).shape)

a shape (3,), b shape (3,)
(2, 3)
(3, 2)


# Copy & Deep Copy

In [49]:
import numpy as np

a = np.arange(4)
b = a
c = a
d = b

d[1:3] = [22, 33]   # array([11, 22, 33,  3])
print(a)            # array([11, 22, 33,  3])
print(b)            # array([11, 22, 33,  3])
print(c)            # array([11, 22, 33,  3])
print(b is a)

[ 0 22 33  3]
[ 0 22 33  3]
[ 0 22 33  3]
True


In [50]:
b = a.copy()    # deep copy
print(b)        # array([11, 22, 33,  3])
a[3] = 44
print(a)        # array([11, 22, 33, 44])
print(b)        # array([11, 22, 33,  3])

[ 0 22 33  3]
[ 0 22 33 44]
[ 0 22 33  3]
