In [1]:
from pyspark.sql import SparkSession

In [3]:
spark = SparkSession.builder.appName("Dataframe").getOrCreate()

In [4]:
spark

In [13]:
# read dataset
df_pyspark = spark.read.option("header", "true").csv("test1.csv", inferSchema =True)

In [7]:
spark.read.option("header", "true").csv("test1.csv").show()

+---------+---+----------+------+
|     Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
|     Paul| 24|         3| 20000|
|   Harsha| 21|         1| 15000|
|  Shubham| 23|         2| 18000|
+---------+---+----------+------+



In [9]:
df_pyspark

DataFrame[Name: string, age: string, Experience: string, Salary: string]

In [11]:
df_pyspark.show()

+---------+---+----------+------+
|     Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
|     Paul| 24|         3| 20000|
|   Harsha| 21|         1| 15000|
|  Shubham| 23|         2| 18000|
+---------+---+----------+------+



In [14]:
# check schema(datatypes of column)
df_pyspark.printSchema()

root
 |-- Name: string (nullable = true)
 |-- age: integer (nullable = true)
 |-- Experience: integer (nullable = true)
 |-- Salary: integer (nullable = true)



In [36]:
# different method to make schema inference
df_pyspark = spark.read.csv("test1.csv", header = True, inferSchema =True)
df_pyspark
df_pyspark.printSchema()

root
 |-- Name: string (nullable = true)
 |-- age: integer (nullable = true)
 |-- Experience: integer (nullable = true)
 |-- Salary: integer (nullable = true)



In [18]:
#get column names
df_pyspark.columns

['Name', 'age', 'Experience', 'Salary']

In [20]:
df_pyspark.head(3)

[Row(Name='Krish', age=31, Experience=10, Salary=30000),
 Row(Name='Sudhanshu', age=30, Experience=8, Salary=25000),
 Row(Name='Sunny', age=29, Experience=4, Salary=20000)]

In [21]:
df_pyspark.show(3)

+---------+---+----------+------+
|     Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
+---------+---+----------+------+
only showing top 3 rows



In [22]:
# only pick up "Name" column
df_pyspark.select("Name")

DataFrame[Name: string]

In [23]:
df_pyspark.select("Name").show()

+---------+
|     Name|
+---------+
|    Krish|
|Sudhanshu|
|    Sunny|
|     Paul|
|   Harsha|
|  Shubham|
+---------+



In [25]:
# for 2 specific columns
df_pyspark.select(["Name", "Experience"]).show()

+---------+----------+
|     Name|Experience|
+---------+----------+
|    Krish|        10|
|Sudhanshu|         8|
|    Sunny|         4|
|     Paul|         3|
|   Harsha|         1|
|  Shubham|         2|
+---------+----------+



In [26]:
df_pyspark["Name"]

Column<'Name'>

In [27]:
df_pyspark.dtypes #similar to pandas

[('Name', 'string'), ('age', 'int'), ('Experience', 'int'), ('Salary', 'int')]

In [29]:
df_pyspark.describe().show()

+-------+------+------------------+-----------------+------------------+
|summary|  Name|               age|       Experience|            Salary|
+-------+------+------------------+-----------------+------------------+
|  count|     6|                 6|                6|                 6|
|   mean|  null|26.333333333333332|4.666666666666667|21333.333333333332|
| stddev|  null| 4.179314138308661|3.559026084010437| 5354.126134736337|
|    min|Harsha|                21|                1|             15000|
|    max| Sunny|                31|               10|             30000|
+-------+------+------------------+-----------------+------------------+



In [37]:
# add columns in dataframe
df_pyspark = df_pyspark.withColumn("Experience After 2 Years",df_pyspark["Experience"]+2)

In [38]:
df_pyspark.show()

+---------+---+----------+------+------------------------+
|     Name|age|Experience|Salary|Experience After 2 Years|
+---------+---+----------+------+------------------------+
|    Krish| 31|        10| 30000|                      12|
|Sudhanshu| 30|         8| 25000|                      10|
|    Sunny| 29|         4| 20000|                       6|
|     Paul| 24|         3| 20000|                       5|
|   Harsha| 21|         1| 15000|                       3|
|  Shubham| 23|         2| 18000|                       4|
+---------+---+----------+------+------------------------+



In [41]:
# drop columns in dataframe
df_pyspark = df_pyspark.drop("Experience After 2 Years")

In [42]:
df_pyspark.show()

+---------+---+----------+------+
|     Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
|     Paul| 24|         3| 20000|
|   Harsha| 21|         1| 15000|
|  Shubham| 23|         2| 18000|
+---------+---+----------+------+



In [43]:
# Renaming Columns
df_pyspark.withColumnRenamed("age", "Current Age at 2023").show()

+---------+-------------------+----------+------+
|     Name|Current Age at 2023|Experience|Salary|
+---------+-------------------+----------+------+
|    Krish|                 31|        10| 30000|
|Sudhanshu|                 30|         8| 25000|
|    Sunny|                 29|         4| 20000|
|     Paul|                 24|         3| 20000|
|   Harsha|                 21|         1| 15000|
|  Shubham|                 23|         2| 18000|
+---------+-------------------+----------+------+

