In [None]:
%pip install -U scikit-learn scipy matplotlib

In [None]:
import numpy as np
from sklearn.compose import ColumnTransformer
from sklearn.datasets import fetch_openml
from sklearn.pipeline import Pipeline
from sklearn.impute import SimpleImputer
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.linear_model import LogisticRegression
from sklearn.model_selection import train_test_split, GridSearchCV


X = df_fishing_2017_2020
y = df_fishing_2017_2020['PESO VIVO KILOGRAMOS']

numeric_features = ["PESO DESEMBARCADO KILOGRAMOS", "PESO VIVO KILOGRAMOS","VALOR PESOS"]
numeric_transformer = Pipeline(
    steps=[("imputer", SimpleImputer(strategy="median")), ("scaler", StandardScaler())]
)

categorical_features = ["FECHA", "ESTADO", "OFICINA","MES DE CORTE","ORIGEN","ESPECIE","FAMILIA"]
categorical_transformer = OneHotEncoder(handle_unknown="ignore")

preprocessor = ColumnTransformer(
    transformers=[
        ("num", numeric_transformer, numeric_features),
        ("cat", categorical_transformer, categorical_features),
    ]
)
clf = Pipeline(
    steps=[("preprocessor", preprocessor), ("classifier", LogisticRegression())]
)

X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0)

clf.fit(X_train, y_train)
print("model score: %.3f" % clf.score(X_test, y_test))

In [None]:
from sklearn import set_config

set_config(display="diagram")
clf