# AI-Based Autism Screening - Model Training Notebook

In [None]:


# Import required libraries
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns

from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder, StandardScaler
from sklearn.metrics import classification_report, confusion_matrix, accuracy_score
from sklearn.linear_model import LogisticRegression
from sklearn.ensemble import RandomForestClassifier
import xgboost as xgb

import joblib


In [None]:
# Load the UCI Autism Screening Dataset
df = pd.read_csv('../data/autism_screening.csv')

# Display basic info
print("Dataset Shape:", df.shape)
df.head()


In [None]:
# Preprocessing: Handle missing values (if any)
df = df.dropna()

# Encode categorical columns
categorical_cols = ['gender', 'ethnicity', 'jundice', 'austim', 'used_app_before', 'relation']
le = LabelEncoder()

for col in categorical_cols:
    df[col] = le.fit_transform(df[col])

# Encode target column
df['Class/ASD'] = df['Class/ASD'].apply(lambda x: 1 if x == 'YES' else 0)

# Check updated dataframe
df.head()


In [None]:
# Features and Labels
X = df.drop(['Class/ASD', 'age_desc', 'country_of_res'], axis=1, errors='ignore')  # drop unnecessary if exist
y = df['Class/ASD']

# Split into train and test sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

print("Training Samples:", X_train.shape[0])
print("Testing Samples:", X_test.shape[0])
