In [None]:
import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from tensorflow.keras.preprocessing.text import Tokenizer
from tensorflow.keras.preprocessing.sequence import pad_sequences
from tensorflow.keras.models import Sequential
from tensorflow.keras.layers import Embedding, GlobalAveragePooling1D, Dense
import tensorflow as tf

In [None]:
df = pd.read_csv('drive/MyDrive/postings.csv')

In [None]:
def preprocess_text(text):
    if pd.isna(text):
        return ""
    return str(text).lower()

df['description'] = df['description'].apply(preprocess_text)
df = df.dropna(subset=['max_salary', 'description'])
df = df[df['max_salary'] > 0]

In [None]:
df['log_salary'] = np.log1p(df['max_salary'])

In [None]:
max_words = 10000
max_len = 200
tokenizer = Tokenizer(num_words=max_words, oov_token='<OOV>')
tokenizer.fit_on_texts(df['description'])
X = tokenizer.texts_to_sequences(df['description'])
X = pad_sequences(X, maxlen=max_len)
y = df['log_salary'].values

In [None]:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=69)

In [None]:
scaler = StandardScaler()
y_train_scaled = scaler.fit_transform(y_train.reshape(-1, 1)).flatten()
y_test_scaled = scaler.transform(y_test.reshape(-1, 1)).flatten()

In [None]:
model = Sequential([
    Embedding(max_words, 50, input_length=max_len),
    GlobalAveragePooling1D(),
    Dense(16, activation='relu'),
    Dense(1)
])

In [None]:
optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001)
model.compile(optimizer=optimizer, loss='mse', metrics=['mae'])

In [None]:
history = model.fit(
    X_train, y_train_scaled,
    epochs=100,
    batch_size=32,
    validation_split=0.2,
    verbose=1
)

Epoch 1/100
Epoch 2/100
Epoch 3/100
Epoch 4/100
Epoch 5/100
Epoch 6/100
Epoch 7/100
Epoch 8/100
Epoch 9/100
Epoch 10/100
Epoch 11/100
Epoch 12/100
Epoch 13/100
Epoch 14/100
Epoch 15/100
Epoch 16/100
Epoch 17/100
Epoch 18/100
Epoch 19/100
Epoch 20/100
Epoch 21/100
Epoch 22/100
Epoch 23/100
Epoch 24/100
Epoch 25/100
Epoch 26/100
Epoch 27/100
Epoch 28/100
Epoch 29/100
Epoch 30/100
Epoch 31/100
Epoch 32/100
Epoch 33/100
Epoch 34/100
Epoch 35/100
Epoch 36/100
Epoch 37/100
Epoch 38/100
Epoch 39/100
Epoch 40/100
Epoch 41/100
Epoch 42/100
Epoch 43/100
Epoch 44/100
Epoch 45/100
Epoch 46/100
Epoch 47/100
Epoch 48/100
Epoch 49/100
Epoch 50/100
Epoch 51/100
Epoch 52/100
Epoch 53/100
Epoch 54/100
Epoch 55/100
Epoch 56/100
Epoch 57/100
Epoch 58/100
Epoch 59/100
Epoch 60/100
Epoch 61/100
Epoch 62/100
Epoch 63/100
Epoch 64/100
Epoch 65/100
Epoch 66/100
Epoch 67/100
Epoch 68/100
Epoch 69/100
Epoch 70/100
Epoch 71/100
Epoch 72/100
Epoch 73/100
Epoch 74/100
Epoch 75/100
Epoch 76/100
Epoch 77/100
Epoch 78

In [None]:
loss, mae = model.evaluate(X_test, y_test_scaled, verbose=0)
print(f"\nTest MAE: {mae}")


Test MAE: 0.4297606348991394


In [None]:
def predict_salary(job_description):
    processed_text = preprocess_text(job_description)
    sequence = tokenizer.texts_to_sequences([processed_text])
    padded = pad_sequences(sequence, maxlen=max_len)
    prediction = model.predict(padded)
    log_salary = scaler.inverse_transform(prediction)[0][0]
    return np.expm1(log_salary)

In [None]:
example_job = "Data Scientist with 5 years of experience in machine learning and big data technologies."
predicted_salary = predict_salary(example_job)
print(f"\nPredicted salary: ${predicted_salary:.2f}")


Predicted salary: $6847.87


In [None]:
example_job = "School director based in Missisipi state, 10 years of experience"
predicted_salary = predict_salary(example_job)
print(f"\nPredicted salary: ${predicted_salary:.2f}")


Predicted salary: $2652.85


In [None]:
!pip install joblib



In [None]:
import joblib

model.save('salary_prediction_model.keras')
joblib.dump(tokenizer, 'tokenizer.joblib')
joblib.dump(scaler, 'scaler.joblib')

print("Model, tokenizer, and scaler have been saved.")

Model, tokenizer, and scaler have been saved.
