1. Import Libraries

In [1]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder

2. Reading File

In [2]:
LM = pd.read_csv("LM_position.csv")

3. Selecting Features

In [3]:

# Drop unnecessary columns
LM = LM.drop(['Nationality', 'Overall', 'Club', 'Work Rate', 'Body Type',
                  'Jersey Number', 'Joined', 'Loaned From', 'Contract Valid Until',
                  'GKDiving', 'GKHandling', 'GKKicking', 'GKPositioning', 
                  'GKReflexes', 'Release Clause', 'Positioning'], axis=1)

# Convert categorical features
label_encoder = LabelEncoder()
LM['Preferred Foot'] = label_encoder.fit_transform(LM['Preferred Foot'])
LM['Position'] = label_encoder.fit_transform(LM['Position'])
    


4. Feature Engineering Functions

In [4]:
import NormalizeValue

LM['Height'] = LM['Height'].apply(NormalizeValue.convert_height_to_cm)
LM['Weight'] = LM['Weight'].apply(NormalizeValue.convert_weight_to_kg)
LM['Value'] = LM['Value'].apply(NormalizeValue.convert_value_wage).astype(int)
LM['Wage'] = LM['Wage'].apply(NormalizeValue.convert_value_wage).astype(int)
    
    # Create additional features
LM['Fitness'] = LM[['Acceleration', 'SprintSpeed', 'Agility', 'Reactions',
                         'Balance', 'Jumping', 'Stamina', 'Strength', 
                         'Aggression', 'Vision']].sum(axis=1)
    



Potential Prediction Model

In [5]:
import CombinedModle

# Define features and targets for potential prediction
x_potential = LM.drop(['ID', 'Potential'], axis=1)
y_potential = LM['Potential']

# Ensure x_potential contains only numeric values
x_potential = x_potential.select_dtypes(include=['number'])

# Split data
x_train_p, x_test_p, y_train_p, y_test_p = train_test_split(x_potential, y_potential, test_size=0.25, random_state=42)
print("Player Potential : ")
type='potential'
Combine_test_p, Combine_train_p , LM = CombinedModle.train_and_evaluate(LM, type, x_train_p, y_train_p, x_test_p, y_test_p)



Player Potential : 


  return linalg.solve(A, Xy, assume_a="pos", overwrite_a=True).T


Combined R^2 Test: 0.9201062665005812
Combined R^2 Train: 0.9632602934637504


Wage Prediction Model

In [6]:
import CombinedModle


x_Wage = LM[['International Reputation', 'Potential', 'Fitness', 'Skill Moves','Value']]
y_Wage = LM['Wage']

# Ensure x_Wage contains only numeric values
x_Wage = x_Wage.select_dtypes(include=['number'])

# Split data
x_train_w, x_test_w, y_train_w, y_test_w = train_test_split(x_Wage, y_Wage, test_size=0.25, random_state=42)
print("\nPlayer Wage : ")
type='wage'
Combine_test_w, Combine_train_w, LM = CombinedModle.train_and_evaluate(LM, type, x_train_w, y_train_w, x_test_w, y_test_w)




Player Wage : 
Combined R^2 Test: 0.6578707518278537
Combined R^2 Train: 0.9200965068848909


Filtering Top 10 Players

In [8]:

# Get the top 10 players based on predicted potential
top_players = LM.nlargest(10, 'PredictedPotential')
    
# Display the top players' information in a table format
print("\nTop 10 Players' Information:")
print(top_players.to_string(index=False))
data= top_players


Top 10 Players' Information:
    ID            Name  Age  Potential    Value   Wage  Preferred Foot  International Reputation  Weak Foot  Skill Moves  Position  Height  Weight  Crossing  Finishing  HeadingAccuracy  ShortPassing  Volleys  Dribbling  Curve  FKAccuracy  LongPassing  BallControl  Acceleration  SprintSpeed  Agility  Reactions  Balance  ShotPower  Jumping  Stamina  Strength  LongShots  Aggression  Interceptions  Vision  Penalties  Composure  Marking  StandingTackle  SlidingTackle  Fitness  PredictedPotential  PredictedWage
215914        N. Kanté   27         90 63000000 225000               1                       3.0        3.0          2.0         2  167.64   72.12      68.0       65.0             54.0          86.0     56.0       79.0   49.0        49.0         81.0         80.0          82.0         78.0     82.0       93.0     92.0       71.0     77.0     96.0      76.0       69.0        90.0           92.0    79.0       54.0       85.0     90.0            91.0        