In [3]:
import numpy as np
import pandas as pd

In [9]:
# 1. Creating Data
# Create a NumPy array of 10 random integers between 1 and 100, then convert it to a Pandas Series.
arr = np.random.randint(101, size=(10,))
s = pd.Series(arr)
print(f"NumPy Array: {arr}")
print(f"Pandas Series:\n{s}")

# Create a DataFrame with columns: Name, Age, City (at least 5 rows).
data = {
    "Name": ["Alice", "Bob", "Charlie", "Diana", "Ethan"],
    "Age": [25, 30, 22, 28, 35],
    "City": ["New York", "Los Angeles", "Chicago", "Houston", "Phoenix"]
}
df = pd.DataFrame(data)
print(f"DataFrame:\n{df}")

# Convert a list of dictionaries into a DataFrame.
data = [{'Name': 'Alice', 'Age': 25}, {'Name': 'Bob', 'Age': 30}]
df = pd.DataFrame(data)
print(f"DataFrame:\n{df}")

NumPy Array: [ 96  16 100  26  48  11  79  92  79  82]
Pandas Series:
0     96
1     16
2    100
3     26
4     48
5     11
6     79
7     92
8     79
9     82
dtype: int32
DataFrame:
      Name  Age         City
0    Alice   25     New York
1      Bob   30  Los Angeles
2  Charlie   22      Chicago
3    Diana   28      Houston
4    Ethan   35      Phoenix
DataFrame:
    Name  Age
0  Alice   25
1    Bob   30


In [22]:
# 2. Exploring Data
# Print the first 5 rows and last 5 rows of a DataFrame.
df=pd.read_csv("../datasets/titanic/train.csv")
print(f"First 5 rows: {df.head()}")
print(f"Last 5 rows: {df.tail()}")

# Display summary statistics of numeric columns (mean, std, min, max).
numeric_cols=[]

# Get list of numeric columns
from pandas.api.types import is_numeric_dtype
numeric_cols = [col for col in df.columns if is_numeric_dtype(df[col])] 
print(numeric_cols)

# Display numerical statistics for numeric columns
for index, col in enumerate(numeric_cols, start=1):
    print(f"{index}. {col}")
    print(f"Mean: {df[col].mean()}")
    print(f"Std: {df[col].std()}")
    print(f"Min: {df[col].min()}")
    print(f"Max: {df[col].max()}")
    print()

# Show the column names, data types, and shape of a DataFrame.
print(df.info())

First 5 rows:    PassengerId  Survived  Pclass  \
0            1         0       3   
1            2         1       1   
2            3         1       3   
3            4         1       1   
4            5         0       3   

                                                Name     Sex   Age  SibSp  \
0                            Braund, Mr. Owen Harris    male  22.0      1   
1  Cumings, Mrs. John Bradley (Florence Briggs Th...  female  38.0      1   
2                             Heikkinen, Miss. Laina  female  26.0      0   
3       Futrelle, Mrs. Jacques Heath (Lily May Peel)  female  35.0      1   
4                           Allen, Mr. William Henry    male  35.0      0   

   Parch            Ticket     Fare Cabin Embarked  
0      0         A/5 21171   7.2500   NaN        S  
1      0          PC 17599  71.2833   C85        C  
2      0  STON/O2. 3101282   7.9250   NaN        S  
3      0            113803  53.1000  C123        S  
4      0            373450   8.0500   NaN

In [49]:
# 3. Selecting Data
# Select a single column from a DataFrame as a Series.
df=pd.read_csv("../datasets/titanic/train.csv")
s=pd.Series(df.loc[:, 'PassengerId'])
# print(s)

# Select multiple columns as a new DataFrame.
df_first_three_cols=pd.DataFrame(df.iloc[:, :3])
# print(df_first_three_cols)

# Select rows by index position using .iloc.
df_rows=pd.DataFrame(df.iloc[:6, :])
# print(df_rows)

# Select rows by condition, e.g., Age > 30.
df_age_greater_than_30=pd.DataFrame(df[df['Age'] > 30])
# print(df_age_greater_than_30)

# Select rows and columns together, e.g., Name and Sex for rows where Age > 25.
df_age_greater_than_25=df.loc[df['Age'] > 25, ['Name', 'Age']]
print(df_age_greater_than_25)

                                                  Name   Age
1    Cumings, Mrs. John Bradley (Florence Briggs Th...  38.0
2                               Heikkinen, Miss. Laina  26.0
3         Futrelle, Mrs. Jacques Heath (Lily May Peel)  35.0
4                             Allen, Mr. William Henry  35.0
6                              McCarthy, Mr. Timothy J  54.0
..                                                 ...   ...
883                      Banfield, Mr. Frederick James  28.0
885               Rice, Mrs. William (Margaret Norton)  39.0
886                              Montvila, Rev. Juozas  27.0
889                              Behr, Mr. Karl Howell  26.0
890                                Dooley, Mr. Patrick  32.0

[413 rows x 2 columns]


In [50]:
# 4. Modifying Data
# Add a new column Salary with random values between 30000 and 100000.
df['Salary']=np.random.randint(30000, 100001, size=(df.shape[0]))
# print(df)

# Increase all Age values by 1.
df['Age']+=1

# Rename columns to FullName, AgeYears, CityName.
df.rename(columns={'Name':'FullName', 'Age':'AgeYears', 'Cabin':'CityName'}, inplace=True)
# print(df)

# Drop a column from the DataFrame.
df=df.drop(columns=['SibSp', 'Parch', 'Ticket', 'CityName', 'Salary'])
print(df)

     PassengerId  Survived  Pclass  \
0              1         0       3   
1              2         1       1   
2              3         1       3   
3              4         1       1   
4              5         0       3   
..           ...       ...     ...   
886          887         0       2   
887          888         1       1   
888          889         0       3   
889          890         1       1   
890          891         0       3   

                                              FullName     Sex  AgeYears  \
0                              Braund, Mr. Owen Harris    male      23.0   
1    Cumings, Mrs. John Bradley (Florence Briggs Th...  female      39.0   
2                               Heikkinen, Miss. Laina  female      27.0   
3         Futrelle, Mrs. Jacques Heath (Lily May Peel)  female      36.0   
4                             Allen, Mr. William Henry    male      36.0   
..                                                 ...     ...       ...   
886        

In [72]:
# 5. Missing Data
# Introduce NaN values in your DataFrame and detect them with .isna().
df=pd.read_csv("../datasets/titanic/train.csv")
# print(df.isna().sum())

# Fill missing numeric values with the mean of that column.
df.fillna(df.mean(numeric_only=True), inplace=True)
# print(df.select_dtypes(include='number').isna().sum())

# Drop rows that contain missing values.
print(df.isna().sum())
df.dropna(inplace=True)
print(df.isna().sum())

PassengerId      0
Survived         0
Pclass           0
Name             0
Sex              0
Age              0
SibSp            0
Parch            0
Ticket           0
Fare             0
Cabin          687
Embarked         2
dtype: int64
PassengerId    0
Survived       0
Pclass         0
Name           0
Sex            0
Age            0
SibSp          0
Parch          0
Ticket         0
Fare           0
Cabin          0
Embarked       0
dtype: int64


In [94]:
# 6. Sorting and Indexing
# Sort the DataFrame by Age descending.
df=pd.read_csv("../datasets/titanic/train.csv")
df.sort_values(by=['Age'], ascending=False, inplace=True)
# print(df['Age'].to_string())

# Sort by multiple columns: first by Fare, then by Age.
df.sort_values(by=['Fare', 'Age'], inplace=True)
fare_and_age=df[['Fare', 'Age']]
# print(fare_and_age.head(10))

# Set Name as the index of the DataFrame.
df.set_index(df['Name'], inplace=True)
# print(df)

In [103]:
# 7. Grouping and Aggregation
# Group data by Sex and find average Age.
df=pd.read_csv("../datasets/titanic/train.csv")
print(df.groupby('Sex')['Age'].mean())

# Group by Sex and find count of people in each sex.
print(df.groupby('Sex')['Sex'].count())

# Group by Sex and compute multiple aggregations: mean Age, max Fare.
print(df.groupby('Sex')[['Age', 'Fare']].mean())


Sex
female    27.915709
male      30.726645
Name: Age, dtype: float64
Sex
female    314
male      577
Name: Sex, dtype: int64
              Age       Fare
Sex                         
female  27.915709  44.479818
male    30.726645  25.523893


In [18]:
# 8. Combining DataFrames
# Concatenate two DataFrames vertically (stack rows).
data1=pd.DataFrame({
    'Name': ['Juan dela Cruz'],
    'Age': [20],
    'Gender': ['M']
})

data2=pd.DataFrame({
    'Name': ['Marie Doe'],
    'Age': [19],
    'Gender': ['F']
})

df=pd.concat([data1, data2], ignore_index=True) # Ignores original row indices
# print(df)

# Concatenate two DataFrames horizontally (add columns).
data1=pd.DataFrame({
    'Name': ['Juan dela Cruz'],
    'Age': [20],
    'Gender': ['M']
})

data2=pd.DataFrame({
    'Name': ['Marie Doe'],
    'Age': [19],
    'Gender': ['F']
})
df=pd.concat([data1, data2], axis=1)
# print(df)

# Merge two DataFrames on a common column (Name) using inner join.
data1 = pd.DataFrame({
    'Name': ['Juan dela Cruz', 'Marie Doe', 'Alice'],
    'Age': [20, 19, 25]
})

data2 = pd.DataFrame({
    'Name': ['Juan dela Cruz', 'Marie Doe', 'Bob'],
    'Gender': ['M', 'F', 'M']
})
df=pd.concat([data1, data2], ignore_index=True) # Ignores original row indices
print(f"Original:\n{df}\n")

inner_merged = pd.merge(data1, data2, on='Name', how='inner') # Keep intersection
print(f"Inner Join:\n{inner_merged}\n")
outer_merged = pd.merge(data1, data2, on='Name', how='outer') # Keep all rows from both
print(f"Outer Join:\n{outer_merged}\n")

left_merged = pd.merge(data1, data2, on='Name', how='left') # Keep all rows from data1
print(f"Left Join:\n{left_merged}\n")
right_merged = pd.merge(data1, data2, on='Name', how='right')# Keep all rows from data2
print(f"Right Join:\n{right_merged}")

Original:
             Name   Age Gender
0  Juan dela Cruz  20.0    NaN
1       Marie Doe  19.0    NaN
2           Alice  25.0    NaN
3  Juan dela Cruz   NaN      M
4       Marie Doe   NaN      F
5             Bob   NaN      M

Inner Join:
             Name  Age Gender
0  Juan dela Cruz   20      M
1       Marie Doe   19      F

Outer Join:
             Name   Age Gender
0           Alice  25.0    NaN
1             Bob   NaN      M
2  Juan dela Cruz  20.0      M
3       Marie Doe  19.0      F

Left Join:
             Name  Age Gender
0  Juan dela Cruz   20      M
1       Marie Doe   19      F
2           Alice   25    NaN

Right Join:
             Name   Age Gender
0  Juan dela Cruz  20.0      M
1       Marie Doe  19.0      F
2             Bob   NaN      M


In [21]:
# 9. Working with Dates
# Create a column DateOfJoining with random dates in 2025.
df = pd.DataFrame({
    'Name': ['Juan dela Cruz', 'Marie Doe', 'Alice', 'Bob'],
    'Age': [20, 19, 25, 30]
})
dates = pd.date_range(start='2025-01-01', end='2025-12-31')
df['DateOfJoining'] = np.random.choice(dates, size=len(df))
print(df)

# Extract Year, Month, and Day from the date column.
df['Year'] = df['DateOfJoining'].dt.year
df['Month'] = df['DateOfJoining'].dt.month
df['Day'] = df['DateOfJoining'].dt.day
print(df)

# Filter rows where DateOfJoining is after June 1, 2025.
filtered_df = df[df['DateOfJoining'] > '2025-06-01']
print(filtered_df)

             Name  Age DateOfJoining
0  Juan dela Cruz   20    2025-01-17
1       Marie Doe   19    2025-12-10
2           Alice   25    2025-01-30
3             Bob   30    2025-03-16
             Name  Age DateOfJoining  Year  Month  Day
0  Juan dela Cruz   20    2025-01-17  2025      1   17
1       Marie Doe   19    2025-12-10  2025     12   10
2           Alice   25    2025-01-30  2025      1   30
3             Bob   30    2025-03-16  2025      3   16
        Name  Age DateOfJoining  Year  Month  Day
1  Marie Doe   19    2025-12-10  2025     12   10
