{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\nHey, thanks for viewing my Kernel!\n\nIf you like my work, please, leave an upvote: it will be really appreciated and it will motivate me in offering more content to the Kaggle community ! :)\n\nYou can access more detail about featimp [here](https://github.com/Hasan-Basri-Akcay/featimp). If you like the featimp library, don't forget to give a star to the repository. I will continue to develop the featimp according to your liking. :)\n\nEDA was done in this [notebook](https://www.kaggle.com/code/hasanbasriakcay/tpsaug22-insightful-eda-new-lib-featimp).","metadata":{}},{"cell_type":"code","source":"pip install featimp","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-05T06:20:38.135493Z","iopub.execute_input":"2022-08-05T06:20:38.135932Z","iopub.status.idle":"2022-08-05T06:20:51.852667Z","shell.execute_reply.started":"2022-08-05T06:20:38.135897Z","shell.execute_reply":"2022-08-05T06:20:51.851376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport warnings\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom featimp import get_feature_importances\nfrom featimp import get_fi_plots\n\nsns.set()\nwarnings.simplefilter(\"ignore\")\ncm = sns.color_palette(\"coolwarm\", as_cmap=True)\n\nAGG_GROUPS = ['product_code']\nNFOLD = 5\nN_TRAILS = 100","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:06:27.717671Z","iopub.execute_input":"2022-08-05T08:06:27.718074Z","iopub.status.idle":"2022-08-05T08:06:27.728033Z","shell.execute_reply.started":"2022-08-05T08:06:27.718041Z","shell.execute_reply":"2022-08-05T08:06:27.726736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\")\ntest = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\")\nsub = pd.read_csv(\"../input/tabular-playground-series-aug-2022/sample_submission.csv\")\n\ndisplay(train.head())\ndisplay(test.head())\ndisplay(sub.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:53.587207Z","iopub.execute_input":"2022-08-05T06:20:53.587564Z","iopub.status.idle":"2022-08-05T06:20:53.928617Z","shell.execute_reply.started":"2022-08-05T06:20:53.587533Z","shell.execute_reply":"2022-08-05T06:20:53.927235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_features = list(test.select_dtypes(np.float).columns)\nint_features = list(test.select_dtypes(np.int).columns)\nobject_features = list(test.select_dtypes(np.object).columns)\n\nobject_features += ['attribute_2', 'attribute_3']\nint_features.remove('attribute_2')\nint_features.remove('attribute_3')\n\nprint(\"float_features:\",float_features)\nprint(\"int_features:\",int_features)\nprint(\"object_features:\",object_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:58:14.285934Z","iopub.execute_input":"2022-08-05T06:58:14.286381Z","iopub.status.idle":"2022-08-05T06:58:14.302421Z","shell.execute_reply.started":"2022-08-05T06:58:14.286346Z","shell.execute_reply":"2022-08-05T06:58:14.300795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering\n\nIn this part, I create a feature engineering function and test it.","metadata":{}},{"cell_type":"code","source":"def add_agg_features_test(df, features, group, inplace=False):\n    from scipy.stats import iqr, kurtosis\n    \n    agg_funcs = {'mean', 'min', 'max', 'var', iqr, kurtosis}\n    \n    if inplace:\n        for agg_func in agg_funcs:\n            agg_df = df.groupby(group)[features].transform(kurtosis)\n            col_names = list(agg_df.columns)\n            if agg_func is iqr:\n                col_names = [col+'_iqr' for col in col_names]\n            elif agg_func is kurtosis:\n                col_names = [col+'_kurt' for col in col_names]\n            else:\n                col_names = [col+'_'+agg_func for col in col_names]\n            agg_df.columns = col_names\n            df = pd.concat([df, agg_df], axis=1)\n    else:\n        df_temp = df.copy()\n        for agg_func in agg_funcs:\n            agg_df = df.groupby(group)[features].transform(kurtosis)\n            col_names = list(agg_df.columns)\n            if agg_func is iqr:\n                col_names = [col+'_iqr' for col in col_names]\n            elif agg_func is kurtosis:\n                col_names = [col+'_kurt' for col in col_names]\n            else:\n                col_names = [col+'_'+agg_func for col in col_names]\n            agg_df.columns = col_names\n            df_temp = pd.concat([df_temp, agg_df], axis=1)\n        return df_temp","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-05T08:14:58.228110Z","iopub.execute_input":"2022-08-05T08:14:58.228562Z","iopub.status.idle":"2022-08-05T08:14:58.239877Z","shell.execute_reply.started":"2022-08-05T08:14:58.228525Z","shell.execute_reply":"2022-08-05T08:14:58.238741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_float_features = ['measurement_3', 'measurement_4', 'measurement_5', 'measurement_6', 'measurement_7', 'measurement_8', 'measurement_9', 'measurement_10', \n                           'measurement_11', 'measurement_12', 'measurement_13', 'measurement_14', 'measurement_15', 'measurement_16', 'measurement_17']\ntt = add_agg_features_test(train, selected_float_features, AGG_GROUPS)\ntt.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:15:01.193526Z","iopub.execute_input":"2022-08-05T08:15:01.193950Z","iopub.status.idle":"2022-08-05T08:15:01.529019Z","shell.execute_reply.started":"2022-08-05T08:15:01.193906Z","shell.execute_reply":"2022-08-05T08:15:01.527462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_fe = train.copy()\ntest_fe = test.copy()\ntrain_fe['measurement_2_greater'] = train_fe['measurement_2'] >= 12\ntest_fe['measurement_2_greater'] = test_fe['measurement_2'] >= 12\ntrain_fe['attribute_0'] = train_fe['attribute_0'] == 'material_7'\ntest_fe['attribute_0'] = test_fe['attribute_0'] == 'material_7'","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:19:25.770922Z","iopub.execute_input":"2022-08-05T07:19:25.771324Z","iopub.status.idle":"2022-08-05T07:19:25.788244Z","shell.execute_reply.started":"2022-08-05T07:19:25.771293Z","shell.execute_reply":"2022-08-05T07:19:25.787218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline\nWe need a class to add our feature engineering function to the pipeline. You can see the class below.","metadata":{}},{"cell_type":"code","source":"class FE():\n    def __init__ (self, features=[], group_cols=[] ,groups=[]):\n        self.features = features\n        self.group_cols = group_cols\n        self.groups = groups\n    def _add_agg_features(self, array, features, group_cols, groups):\n        from scipy.stats import iqr, kurtosis\n\n        df = pd.DataFrame(array, columns=features)\n        df[group_cols] = groups\n\n        agg_funcs = {'mean', 'min', 'max', 'var', iqr, kurtosis}\n\n        new_col_names =[]\n        for agg_func in agg_funcs:\n                agg_df = df.groupby(group_cols)[features].transform(kurtosis)\n                col_names = list(agg_df.columns)\n                if agg_func is iqr:\n                    col_names = [col+'_iqr' for col in col_names]\n                elif agg_func is kurtosis:\n                    col_names = [col+'_kurt' for col in col_names]\n                else:\n                    col_names = [col+'_'+agg_func for col in col_names]\n                agg_df.columns = col_names\n                new_col_names.extend(col_names)\n                df = pd.concat([df, agg_df], axis=1)\n\n        # If don't add this part, the old feature will drop.\n        df[features] = array\n\n        return df[new_col_names+features].to_numpy()\n    def fit(self):\n        pass\n    def transform(self, array, y=None):\n        array_fe = self._add_agg_features(array, self.features, self.group_cols, self.groups)\n        return array_fe\n    def fit_transform(self, array, y=None):\n        array_fe = self._add_agg_features(array, self.features, self.group_cols, self.groups)\n        return array_fe","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:12.586578Z","iopub.execute_input":"2022-08-05T09:23:12.586989Z","iopub.status.idle":"2022-08-05T09:23:12.600132Z","shell.execute_reply.started":"2022-08-05T09:23:12.586956Z","shell.execute_reply":"2022-08-05T09:23:12.598718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\nselected_float_features = ['measurement_3', 'measurement_4', 'measurement_5', 'measurement_6', 'measurement_7', 'measurement_8', 'measurement_9', 'measurement_10', \n                           'measurement_11', 'measurement_12', 'measurement_13', 'measurement_14', 'measurement_15', 'measurement_16', 'measurement_17']\nselected_int_features = ['measurement_0', 'measurement_1', 'measurement_2']\nselected_cat_features = ['attribute_1', 'attribute_2', 'attribute_3']\nadditional_features = ['loading', 'attribute_0', 'measurement_2_greater']\n\nnumeric_transformer_float = Pipeline(\n    steps=[(\"imputer\", IterativeImputer()), (\"scaler\", StandardScaler()), ('fe_float', FE(features=selected_float_features, group_cols=AGG_GROUPS, \n                                                                                                      groups=train_fe[AGG_GROUPS]))]\n)\nnumeric_transformer_int = Pipeline(\n    steps=[(\"imputer\", SimpleImputer(strategy=\"mean\")), (\"scaler\", StandardScaler()), ('fe_float', FE(features=selected_int_features, group_cols=AGG_GROUPS, \n                                                                                                      groups=train_fe[AGG_GROUPS]))]\n)\ncategorical_transformer = OneHotEncoder(handle_unknown=\"ignore\")\nadditional_transformer = SimpleImputer(strategy=\"most_frequent\")\n\n\npreprocessor_own = ColumnTransformer(\n    transformers=[\n        (\"num_float\", numeric_transformer_float, selected_float_features),\n        (\"num_int\", numeric_transformer_int, selected_int_features),\n        (\"cat\", categorical_transformer, selected_cat_features),\n        ('add', additional_transformer, additional_features)\n    ]\n)\n\npipe = Pipeline(\n    steps=[('preprocessor', preprocessor_own), ('model', LogisticRegression())]\n)\n\nX = train_fe.drop(['failure', 'product_code', 'id'], axis=1)\ny = train_fe[['failure']]\n\nX_tr = preprocessor_own.fit_transform(X)\nprint('X_tr.shape:', X_tr.shape)\n\ndisplay(pipe.fit(X, y))\n\nprint(\"model score: %.3f\" % pipe.score(X, y))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T10:04:54.082130Z","iopub.execute_input":"2022-08-05T10:04:54.082545Z","iopub.status.idle":"2022-08-05T10:05:07.149743Z","shell.execute_reply.started":"2022-08-05T10:04:54.082498Z","shell.execute_reply":"2022-08-05T10:05:07.148829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = pipe.predict_proba(X)[:,1]\npreds","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:26:23.269819Z","iopub.execute_input":"2022-08-05T09:26:23.270349Z","iopub.status.idle":"2022-08-05T09:26:24.213270Z","shell.execute_reply.started":"2022-08-05T09:26:23.270313Z","shell.execute_reply":"2022-08-05T09:26:24.211949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import dump, load\ndump(pipe, 'pipe.joblib')\nsaved_pipe = load('pipe.joblib') ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:26:26.353439Z","iopub.execute_input":"2022-08-05T09:26:26.354391Z","iopub.status.idle":"2022-08-05T09:26:26.386270Z","shell.execute_reply.started":"2022-08-05T09:26:26.354347Z","shell.execute_reply":"2022-08-05T09:26:26.385148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_preds = saved_pipe.predict_proba(X)[:,1]\nnew_preds","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:26:28.077416Z","iopub.execute_input":"2022-08-05T09:26:28.077886Z","iopub.status.idle":"2022-08-05T09:26:29.003212Z","shell.execute_reply.started":"2022-08-05T09:26:28.077836Z","shell.execute_reply":"2022-08-05T09:26:29.001990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.testing.assert_array_equal(preds, new_preds)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:27:30.697450Z","iopub.execute_input":"2022-08-05T09:27:30.698189Z","iopub.status.idle":"2022-08-05T09:27:30.705172Z","shell.execute_reply.started":"2022-08-05T09:27:30.698148Z","shell.execute_reply":"2022-08-05T09:27:30.703855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameter Tuning\nYou can also use pipeline model with optuna and cross_val_score.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nimport optuna\n\ndef objective_logistic(trial):\n    from sklearn.model_selection import StratifiedGroupKFold    \n    cv = StratifiedGroupKFold(n_splits=NFOLD)\n    \n    params = {\n        'solver': trial.suggest_categorical(\"solver\", [\"newton-cg\", \"lbfgs\", \"liblinear\", \"sag\", \"saga\"]),\n        'C': trial.suggest_float(\"C\", 1e-10, 1e10, log=True),\n        'max_iter': trial.suggest_int('max_iter', 100, 1000),\n    }\n    \n    if params['solver'] in ['liblinear', 'saga']:\n        params['penalty'] = trial.suggest_categorical(\"penalty\", ['l1', 'l2'])\n    \n    params['multi_class'] = 'ovr'\n    \n    pipe_model = Pipeline(\n        steps=[('preprocessor', preprocessor_own), ('model', LogisticRegression(**params))]\n    )\n    \n    score = cross_val_score(pipe_model, X, y, groups=train_fe['product_code'], cv=cv, scoring='roc_auc').mean()\n    \n    return score","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:27:43.113095Z","iopub.execute_input":"2022-08-05T09:27:43.113508Z","iopub.status.idle":"2022-08-05T09:27:43.123692Z","shell.execute_reply.started":"2022-08-05T09:27:43.113476Z","shell.execute_reply":"2022-08-05T09:27:43.122363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction='maximize')\nstudy.optimize(objective_logistic, n_trials=N_TRAILS)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:28:35.566418Z","iopub.execute_input":"2022-08-05T09:28:35.566924Z","iopub.status.idle":"2022-08-05T10:03:49.930678Z","shell.execute_reply.started":"2022-08-05T09:28:35.566877Z","shell.execute_reply":"2022-08-05T10:03:49.927946Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of finished trials: ', len(study.trials))\nprint('Best trial: ', study.best_trial.params)\nprint('Best value: ', study.best_value)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_logistic = {'solver': 'liblinear', 'C': 0.010138546022235769, 'max_iter': 462, 'penalty': 'l1'}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"pipe_model = Pipeline(\n        steps=[('preprocessor', preprocessor_own), ('model', LogisticRegression(**study.best_trial.params))]\n    )\npipe_model.fit(X,y)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = pipe_model.predict_proba(test_fe[X.columns])[:,1]\nsub['failure'] = preds\nsub.to_csv(\"submission.csv\", index=False)\nsub.describe()","metadata":{},"execution_count":null,"outputs":[]}]}