{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Coupling unsupervised learning with supervised learning to increase model performance. \n\nAfter pre-processing the data using SimpleImputer and IterativeImputer, using two types of models: \n1) to predict the missing label\n2) to make the prediction\n\n## Label prediction using unsupervised learning:\nUsing cluster algorithm for mix data type: K-prototypes.\n\n## Prediction using supervised learning:\nUsing an ensemble model stacking classifier with XGboost, LightGBT and Catboost.\nThe hyperparameters of these models have been optimized using optuna.","metadata":{}},{"cell_type":"code","source":"%%capture\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:42.716105Z","iopub.execute_input":"2024-11-22T18:11:42.716592Z","iopub.status.idle":"2024-11-22T18:11:44.274796Z","shell.execute_reply.started":"2024-11-22T18:11:42.716547Z","shell.execute_reply":"2024-11-22T18:11:44.273702Z"},"_kg_hide-input":false,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install kmodes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:44.276362Z","iopub.execute_input":"2024-11-22T18:11:44.276807Z","iopub.status.idle":"2024-11-22T18:11:54.318793Z","shell.execute_reply.started":"2024-11-22T18:11:44.276774Z","shell.execute_reply":"2024-11-22T18:11:54.317384Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kmodes.kmodes import KModes\nfrom kmodes.kprototypes import KPrototypes\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score\n\nfrom sklearn.model_selection import (\n    train_test_split,\n    cross_val_score,\n    StratifiedKFold,\n)\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer, SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\n\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.linear_model import LogisticRegression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:54.320553Z","iopub.execute_input":"2024-11-22T18:11:54.320939Z","iopub.status.idle":"2024-11-22T18:11:55.717850Z","shell.execute_reply.started":"2024-11-22T18:11:54.320902Z","shell.execute_reply":"2024-11-22T18:11:55.716789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_category(df: pd.DataFrame)->pd.DataFrame:\n    categoric_c = df.select_dtypes(include=['object']).columns.tolist()\n    df[categoric_c] = df[categoric_c].astype('category')\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:55.720789Z","iopub.execute_input":"2024-11-22T18:11:55.721504Z","iopub.status.idle":"2024-11-22T18:11:55.727608Z","shell.execute_reply.started":"2024-11-22T18:11:55.721455Z","shell.execute_reply":"2024-11-22T18:11:55.726312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df: pd.DataFrame, verbose: bool=True)->pd.DataFrame:\n    numerics = ['int16', 'int32', 'int64', \n                'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: \n        print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:55.728863Z","iopub.execute_input":"2024-11-22T18:11:55.729215Z","iopub.status.idle":"2024-11-22T18:11:55.741070Z","shell.execute_reply.started":"2024-11-22T18:11:55.729182Z","shell.execute_reply":"2024-11-22T18:11:55.739889Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsubmission = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:55.742931Z","iopub.execute_input":"2024-11-22T18:11:55.743380Z","iopub.status.idle":"2024-11-22T18:11:55.812267Z","shell.execute_reply.started":"2024-11-22T18:11:55.743330Z","shell.execute_reply":"2024-11-22T18:11:55.810424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_keep = test.columns.tolist()\ncolumns_to_keep.append('sii')\ntrain = train[columns_to_keep]\ntrain = train.replace({'NaN':np.nan})\ntest = test.replace({'NaN':np.nan})\nif 'id' in train.columns:\n    train.drop(columns=['id'], inplace=True)\nif 'id' in test.columns:\n    test.drop(columns=['id'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:55.813984Z","iopub.execute_input":"2024-11-22T18:11:55.814444Z","iopub.status.idle":"2024-11-22T18:11:55.834876Z","shell.execute_reply.started":"2024-11-22T18:11:55.814395Z","shell.execute_reply":"2024-11-22T18:11:55.833554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_mem_usage(train)\ntest = reduce_mem_usage(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:55.836174Z","iopub.execute_input":"2024-11-22T18:11:55.836468Z","iopub.status.idle":"2024-11-22T18:11:56.020496Z","shell.execute_reply.started":"2024-11-22T18:11:55.836441Z","shell.execute_reply":"2024-11-22T18:11:56.019274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = to_category(train)\ntest = to_category(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:56.022143Z","iopub.execute_input":"2024-11-22T18:11:56.022480Z","iopub.status.idle":"2024-11-22T18:11:56.051016Z","shell.execute_reply.started":"2024-11-22T18:11:56.022448Z","shell.execute_reply":"2024-11-22T18:11:56.049953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_c = test.select_dtypes(include=['int8','float32']).columns.tolist()\ncat_columns = test.select_dtypes(include=['category']).columns.tolist()\n\nscaler_train = MinMaxScaler()\ntrain[numeric_c] = scaler_train.fit_transform(train[numeric_c])\ntest[numeric_c] = scaler_train.transform(test[numeric_c])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:56.054100Z","iopub.execute_input":"2024-11-22T18:11:56.054457Z","iopub.status.idle":"2024-11-22T18:11:56.081920Z","shell.execute_reply.started":"2024-11-22T18:11:56.054424Z","shell.execute_reply":"2024-11-22T18:11:56.080718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Replacing by Most frequent value","metadata":{}},{"cell_type":"code","source":"object_imputer = SimpleImputer(strategy='constant', fill_value='missing').fit(train[cat_columns])\n# Transform ONLY categorical columns\ntrain[cat_columns] = object_imputer.transform(train[cat_columns])\ntest[cat_columns] = object_imputer.transform(test[cat_columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:56.083233Z","iopub.execute_input":"2024-11-22T18:11:56.083523Z","iopub.status.idle":"2024-11-22T18:11:56.110199Z","shell.execute_reply.started":"2024-11-22T18:11:56.083496Z","shell.execute_reply":"2024-11-22T18:11:56.109276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.replace({'NaN':np.nan})\ntrain[numeric_c]=train[numeric_c].astype('float16')\nimputer = IterativeImputer(max_iter=50, random_state=42).fit(train[numeric_c])\ntrain[numeric_c] = imputer.transform(train[numeric_c])\ntest[numeric_c] = imputer.transform(test[numeric_c])\ntrain.fillna('NaN',inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:11:56.111590Z","iopub.execute_input":"2024-11-22T18:11:56.111975Z","iopub.status.idle":"2024-11-22T18:15:06.888441Z","shell.execute_reply.started":"2024-11-22T18:11:56.111938Z","shell.execute_reply":"2024-11-22T18:15:06.887115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"da_A = train[train['sii'] == 'NaN']\nda_B = train[train['sii'] != 'NaN']\ncolumn_to_trainon = da_A.columns.tolist()\ncolumn_to_trainon.remove('sii')\nda_A[column_to_trainon] = to_category(da_A[column_to_trainon])\nda_B[column_to_trainon] = to_category(da_B[column_to_trainon])\ncategorical = da_A.select_dtypes(include=['category']).columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:06.890016Z","iopub.execute_input":"2024-11-22T18:15:06.890373Z","iopub.status.idle":"2024-11-22T18:15:06.943555Z","shell.execute_reply.started":"2024-11-22T18:15:06.890338Z","shell.execute_reply":"2024-11-22T18:15:06.942287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train model K-protorype to predict 'sii' for da_A Using Kmodes","metadata":{}},{"cell_type":"code","source":"list_index = []\nfor name in categorical:\n    list_index.append(da_A[column_to_trainon].columns.get_loc(name))\nlist_index\nkp = KPrototypes(\n    n_clusters=4, \n    max_iter=1000,\n    random_state=10,\n    verbose=0\n)\n\nclusters = kp.fit_predict(da_A[column_to_trainon], categorical=list_index)\nda_A['sii'] = clusters\nda_B['sii'] = da_B['sii'].astype('uint16')\ntrain_copy = train.copy()\ntrain_copy['sii'].update(da_A['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:06.944911Z","iopub.execute_input":"2024-11-22T18:15:06.945250Z","iopub.status.idle":"2024-11-22T18:15:27.434827Z","shell.execute_reply.started":"2024-11-22T18:15:06.945220Z","shell.execute_reply":"2024-11-22T18:15:27.433599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reverse MinMaxScaler / Use labelencoder for sii","metadata":{}},{"cell_type":"code","source":"label_encoder = LabelEncoder() \n# Encode labels in column 'Depression'. \ntrain_copy['sii'] = label_encoder.fit_transform(train_copy['sii'])\nda_B['sii'] = label_encoder.transform(da_B['sii'])\n\ntrain_copy['sii'] = train_copy['sii'].astype('int8')\nda_B['sii'] = da_B['sii'].astype('int8')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.436199Z","iopub.execute_input":"2024-11-22T18:15:27.436639Z","iopub.status.idle":"2024-11-22T18:15:27.445897Z","shell.execute_reply.started":"2024-11-22T18:15:27.436586Z","shell.execute_reply":"2024-11-22T18:15:27.444801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy = to_category(train_copy)\ntest = to_category(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.447505Z","iopub.execute_input":"2024-11-22T18:15:27.447998Z","iopub.status.idle":"2024-11-22T18:15:27.479297Z","shell.execute_reply.started":"2024-11-22T18:15:27.447941Z","shell.execute_reply":"2024-11-22T18:15:27.478227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy[numeric_c] = train_copy[numeric_c].astype('float32')\ntest[numeric_c] = test[numeric_c].astype('float32')\nda_B[numeric_c] = da_B[numeric_c].astype('float32')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.480625Z","iopub.execute_input":"2024-11-22T18:15:27.481004Z","iopub.status.idle":"2024-11-22T18:15:27.513657Z","shell.execute_reply.started":"2024-11-22T18:15:27.480970Z","shell.execute_reply":"2024-11-22T18:15:27.512417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy[numeric_c] = scaler_train.inverse_transform(train_copy[numeric_c])\nda_B[numeric_c] = scaler_train.inverse_transform(da_B[numeric_c])\ntest[numeric_c] = scaler_train.inverse_transform(test[numeric_c])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.514817Z","iopub.execute_input":"2024-11-22T18:15:27.515181Z","iopub.status.idle":"2024-11-22T18:15:27.548085Z","shell.execute_reply.started":"2024-11-22T18:15:27.515149Z","shell.execute_reply":"2024-11-22T18:15:27.546358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split dataset to train and test","metadata":{}},{"cell_type":"code","source":"X = train_copy.drop('sii', axis=1)\ny = train_copy['sii']\n\nX_Train, X_Test, Y_Train, Y_Test = train_test_split(X, y, test_size=0.3, random_state=42,stratify=y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.549688Z","iopub.execute_input":"2024-11-22T18:15:27.550087Z","iopub.status.idle":"2024-11-22T18:15:27.568650Z","shell.execute_reply.started":"2024-11-22T18:15:27.550050Z","shell.execute_reply":"2024-11-22T18:15:27.567463Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Parameters optimise for models: Catboost, LightGB, XGBoost","metadata":{}},{"cell_type":"code","source":"param_cat = {\n    'learning_rate': 0.3281465930790786,\n    'depth': 7,\n    'random_strength': 0.09001722295974743,\n    'colsample_bylevel': 0.9258899403741545,\n    'bagging_temperature': 0.013024697027863666,\n    'border_count': 5,\n    'l2_leaf_reg': 3,\n    'min_data_in_leaf': 5,\n    'loss_function': 'MultiClass',\n    'iterations': 100,\n    'silent':True,\n    'random_state':42,\n    'thread_count':1,\n}\n\nparam_xgb = {\n    \"eval_metric\": \"mlogloss\",\n    'num_class':4,\n    'objective': 'multi:softmax',\n    'tree_method': 'hist',\n    \"iterations\": 1000,\n    'learning_rate': 0.031749645724399714,\n    'colsample_bytree': 0.944717228468839,\n    'max_depth': 4,\n    'min_child_weight': 1,\n    'subsample': 0.7360783585682544,\n    'reg_lambda': 0.0030351576804894865,\n    \"verbosity\": 0,\n    'random_state':42,\n    'nthread':3,\n} \n\nparam_lgb = {\n    'reg_alpha': 0.0016968406162651858,\n    'reg_lambda': 0.011536519314739676,\n    'colsample_bytree': 0.7068192974267937,\n    'subsample': 0.780076961667283,\n    'learning_rate': 0.006434545432136151,\n    'max_depth': 5,\n    'num_leaves': 100,\n    'min_child_samples': 27,\n    'cat_smooth': 15,\n    'objective': 'multiclass', \n    'num_class': 4, \n    'metric': 'multi_logloss',\n    'n_estimators': 500,\n    'force_col_wise':True,\n    'random_state':42,\n    'num_threads':3,\n    'verbose':-1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.691399Z","iopub.execute_input":"2024-11-22T18:15:27.691694Z","iopub.status.idle":"2024-11-22T18:15:27.700006Z","shell.execute_reply.started":"2024-11-22T18:15:27.691667Z","shell.execute_reply":"2024-11-22T18:15:27.698788Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Stacking Classifier","metadata":{}},{"cell_type":"code","source":"model_XGBoost = xgb.XGBClassifier(\n    **param_xgb, \n    enable_categorical=True, \n)\nmodel_LGMB = lgb.LGBMClassifier(\n    **param_lgb, \n)\nmodel_Cat = CatBoostClassifier(\n    **param_cat,\n    cat_features=cat_columns, \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.701456Z","iopub.execute_input":"2024-11-22T18:15:27.702261Z","iopub.status.idle":"2024-11-22T18:15:27.718757Z","shell.execute_reply.started":"2024-11-22T18:15:27.702214Z","shell.execute_reply":"2024-11-22T18:15:27.717487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"estimators = [\n    ('lgb', model_LGMB),\n    ('cat', model_Cat),\n    ('xgb', model_XGBoost),\n]\nstack = StackingClassifier(\n    estimators=estimators, \n    final_estimator=LogisticRegression(),\n    cv=StratifiedKFold(\n        n_splits=5, \n        shuffle=True, \n        random_state=3\n    ),\n    n_jobs=5,\n)\n\nstack.fit(X_Train, Y_Train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:15:27.720380Z","iopub.execute_input":"2024-11-22T18:15:27.720877Z","iopub.status.idle":"2024-11-22T18:16:27.587401Z","shell.execute_reply.started":"2024-11-22T18:15:27.720806Z","shell.execute_reply":"2024-11-22T18:16:27.585996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, X_train, y_train, X_test, y_test):\n    train_pred = model.predict(X_train)\n    print('----------------------------------------------\\n')\n    print('Train Cohen Kappa: ', cohen_kappa_score(y_train, train_pred))\n    print('----------------------------------------------\\n')\n    test_pred = model.predict(X_test)\n    print('Test Cohen Kappa: ', cohen_kappa_score(y_test, test_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:16:27.589238Z","iopub.execute_input":"2024-11-22T18:16:27.589651Z","iopub.status.idle":"2024-11-22T18:16:27.596252Z","shell.execute_reply.started":"2024-11-22T18:16:27.589610Z","shell.execute_reply":"2024-11-22T18:16:27.594991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_model(stack, X_Train, Y_Train, X_Test, Y_Test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:16:27.598351Z","iopub.execute_input":"2024-11-22T18:16:27.600039Z","iopub.status.idle":"2024-11-22T18:16:28.017889Z","shell.execute_reply.started":"2024-11-22T18:16:27.599971Z","shell.execute_reply":"2024-11-22T18:16:28.016302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_stack = stack.predict(test)\n\nSub = pd.DataFrame({\n    'id': submission.id,\n    'sii': pred_stack\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:16:28.019387Z","iopub.execute_input":"2024-11-22T18:16:28.019946Z","iopub.status.idle":"2024-11-22T18:16:28.062736Z","shell.execute_reply.started":"2024-11-22T18:16:28.019891Z","shell.execute_reply":"2024-11-22T18:16:28.061542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Sub.to_csv(\n    './submission_lgtm_xgboost_catboost_stackingclassifier.csv', \n    index=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T18:16:28.064211Z","iopub.execute_input":"2024-11-22T18:16:28.065179Z","iopub.status.idle":"2024-11-22T18:16:28.074764Z","shell.execute_reply.started":"2024-11-22T18:16:28.065132Z","shell.execute_reply":"2024-11-22T18:16:28.073619Z"}},"outputs":[],"execution_count":null}]}