# Pyspark tutorial 

https://www.youtube.com/watch?v=_C8kWso4ne4&t=10s

In [40]:
# Start a file session 

from pyspark.sql import SparkSession

spark=SparkSession.builder.appName('Dataframe').getOrCreate()
spark

In [41]:
## read dataset

df_pyspark = spark.read.option('header','true').csv("pyspark_tuto_files/test1.csv", inferSchema=True)

# inferschema allows to identify the type automatically

In [42]:
df_pyspark.printSchema()

root
 |-- Name: string (nullable = true)
 |-- age: integer (nullable = true)
 |-- Experience: integer (nullable = true)
 |-- Salary: integer (nullable = true)



In [43]:
# Other syntax

df_pyspark = spark.read.csv("pyspark_tuto_files/test1.csv",header=True, inferSchema=True)

In [44]:
df_pyspark.show()

+---------+---+----------+------+
|     Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
|     Paul| 24|         3| 20000|
|   Harsha| 21|         1| 15000|
|  Shubham| 23|         2| 18000|
+---------+---+----------+------+



In [45]:
df_pyspark.columns

['Name', 'age', 'Experience', 'Salary']

In [46]:
df_pyspark.head(3)

[Row(Name='Krish', age=31, Experience=10, Salary=30000),
 Row(Name='Sudhanshu', age=30, Experience=8, Salary=25000),
 Row(Name='Sunny', age=29, Experience=4, Salary=20000)]

In [47]:
df_pyspark.select('Name').show()

+---------+
|     Name|
+---------+
|    Krish|
|Sudhanshu|
|    Sunny|
|     Paul|
|   Harsha|
|  Shubham|
+---------+



In [48]:
df_pyspark.select('Name','Age').show()

+---------+---+
|     Name|Age|
+---------+---+
|    Krish| 31|
|Sudhanshu| 30|
|    Sunny| 29|
|     Paul| 24|
|   Harsha| 21|
|  Shubham| 23|
+---------+---+



In [49]:
df_pyspark['Name']

# Not working

Column<'Name'>

In [50]:
df_pyspark.dtypes

[('Name', 'string'), ('age', 'int'), ('Experience', 'int'), ('Salary', 'int')]

In [51]:
df_pyspark.describe().show()

+-------+------+------------------+-----------------+------------------+
|summary|  Name|               age|       Experience|            Salary|
+-------+------+------------------+-----------------+------------------+
|  count|     6|                 6|                6|                 6|
|   mean|  NULL|26.333333333333332|4.666666666666667|21333.333333333332|
| stddev|  NULL| 4.179314138308661|3.559026084010437| 5354.126134736337|
|    min|Harsha|                21|                1|             15000|
|    max| Sunny|                31|               10|             30000|
+-------+------+------------------+-----------------+------------------+



In [52]:
# Adding columns 

df_pyspark = df_pyspark.withColumn('Experience + 2 years',df_pyspark['Experience']+2)

In [53]:
df_pyspark.show()

+---------+---+----------+------+--------------------+
|     Name|age|Experience|Salary|Experience + 2 years|
+---------+---+----------+------+--------------------+
|    Krish| 31|        10| 30000|                  12|
|Sudhanshu| 30|         8| 25000|                  10|
|    Sunny| 29|         4| 20000|                   6|
|     Paul| 24|         3| 20000|                   5|
|   Harsha| 21|         1| 15000|                   3|
|  Shubham| 23|         2| 18000|                   4|
+---------+---+----------+------+--------------------+



In [54]:
# Deleting column 

df_pyspark = df_pyspark.drop('Experience + 2 years')

In [55]:
df_pyspark.show()

+---------+---+----------+------+
|     Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
|     Paul| 24|         3| 20000|
|   Harsha| 21|         1| 15000|
|  Shubham| 23|         2| 18000|
+---------+---+----------+------+



In [57]:
# Rename column 

df_pyspark = df_pyspark.withColumnRenamed('Name','New Name')

In [59]:
df_pyspark.show()

+---------+---+----------+------+
| New Name|age|Experience|Salary|
+---------+---+----------+------+
|    Krish| 31|        10| 30000|
|Sudhanshu| 30|         8| 25000|
|    Sunny| 29|         4| 20000|
|     Paul| 24|         3| 20000|
|   Harsha| 21|         1| 15000|
|  Shubham| 23|         2| 18000|
+---------+---+----------+------+

