{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this attempt, I am just going to use CatBoost to engineer features in the train dataset based on their importance, and then train the CatBoost model with aforementioned features and optimize the model using Optuna","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\nfrom itertools import combinations\nfrom scipy.special import softmax\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedKFold, KFold, cross_val_predict\nimport datetime\nfrom sklearn.linear_model import RidgeClassifier, RidgeClassifierCV, LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.preprocessing import MinMaxScaler,StandardScaler, OneHotEncoder, RobustScaler, LabelBinarizer\nfrom sklearn.pipeline import make_pipeline, Pipeline,  FunctionTransformer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.metrics import roc_auc_score, accuracy_score\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.base import clone, TransformerMixin, ClassifierMixin, BaseEstimator\nfrom sklearn import set_config\nfrom tensorflow import keras\nimport catboost as cb\nimport scipy\nfrom colorama import Style, Fore\nimport optuna\nset_config(transform_output='pandas')\n\n# to make the code run faster\ndef reduce_mem(df: pd.DataFrame):\n    \"This method reduces memory for numeric columns in the dataframe\";\n\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64', \"uint16\", \"uint32\", \"uint64\"];\n    start_mem = df.memory_usage().sum() / 1024**2;\n\n    for col in df.columns:\n        col_type = df[col].dtypes\n\n        if col_type in numerics:\n            c_min = df[col].min();\n            c_max = df[col].max();\n\n            if \"int\" in str(col_type):\n                if c_min >= np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min >= np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min >= np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min >= np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min >= np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                if c_min >= np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)  \n\n    end_mem = df.memory_usage().sum() / 1024**2\n\n    print(f\"Start - end memory:- {start_mem:5.2f} - {end_mem:5.2f} Mb\");\n    return df;","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:01:15.504024Z","iopub.execute_input":"2024-11-24T20:01:15.504395Z","iopub.status.idle":"2024-11-24T20:01:20.762724Z","shell.execute_reply.started":"2024-11-24T20:01:15.504358Z","shell.execute_reply":"2024-11-24T20:01:20.761729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Data","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:01:28.188125Z","iopub.execute_input":"2024-11-24T20:01:28.189407Z","iopub.status.idle":"2024-11-24T20:01:28.242107Z","shell.execute_reply.started":"2024-11-24T20:01:28.189370Z","shell.execute_reply":"2024-11-24T20:01:28.241413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:01:30.256127Z","iopub.execute_input":"2024-11-24T20:01:30.256552Z","iopub.status.idle":"2024-11-24T20:01:30.283035Z","shell.execute_reply.started":"2024-11-24T20:01:30.256519Z","shell.execute_reply":"2024-11-24T20:01:30.282180Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The dataset is a huge mess, and we need to remove rows that have NaN values","metadata":{}},{"cell_type":"code","source":"train_data.shape, test_data.shape\n# the number of columns in test and train data is different\ntest_features = list(test_data.columns)\ntrain_features = list(train_data.columns)\nset(train_features)-set(test_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:01:32.867245Z","iopub.execute_input":"2024-11-24T20:01:32.867578Z","iopub.status.idle":"2024-11-24T20:01:32.874246Z","shell.execute_reply.started":"2024-11-24T20:01:32.867551Z","shell.execute_reply":"2024-11-24T20:01:32.873449Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"PCIAT features should not be used for modeling since SII is derived from PCIAT, which means including them will lead to data leakage ","metadata":{}},{"cell_type":"markdown","source":"Calculate the number of nan values in each column","metadata":{}},{"cell_type":"code","source":"train_clean = train_data[test_features+['sii']].drop(columns=['id']).dropna(subset=['sii'])\ntrain_clean.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:01:36.667232Z","iopub.execute_input":"2024-11-24T20:01:36.667975Z","iopub.status.idle":"2024-11-24T20:01:36.679817Z","shell.execute_reply.started":"2024-11-24T20:01:36.667940Z","shell.execute_reply":"2024-11-24T20:01:36.678921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_percentage = (train_clean.isna().sum()/len(train_clean))*100\nprint(\"Percentage of NaN values in each column:\")\nprint(nan_percentage)\nplt.figure(figsize=(8,5))\n\nplt.hist(nan_percentage,bins=5,color='skyblue')\nplt.title('Distribution of Missing Values Percentage Across Columns',fontsize=14)\nplt.xlabel('Percentage of Missing Values',fontsize=12)\nplt.ylabel('Frequency',fontsize=12)\nplt.grid(axis='y',linestyle='--',alpha=0.7)\n\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/missing_value_distribution.png\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:01:43.667531Z","iopub.execute_input":"2024-11-24T20:01:43.668270Z","iopub.status.idle":"2024-11-24T20:01:44.145886Z","shell.execute_reply.started":"2024-11-24T20:01:43.668233Z","shell.execute_reply":"2024-11-24T20:01:44.145053Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are more than 10 features that have more than 50% of missing values. We need to remove them all","metadata":{}},{"cell_type":"code","source":"def classify_feature(col):\n    if pd.api.types.is_bool_dtype(col):\n        return 'boolean'\n    elif pd.api.types.is_numeric_dtype(col) and col.nunique() == 2 and set(col.unique()).issubset({0,1}):\n        return 'categorical' #0/1 encoded as categorical\n    elif pd.api.types.is_integer_dtype(col):\n        return 'categorical'\n    elif pd.api.types.is_numeric_dtype(col):\n        return 'numeric'\n    else:\n        return 'categorical'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T20:03:57.292730Z","iopub.execute_input":"2024-11-24T20:03:57.293638Z","iopub.status.idle":"2024-11-24T20:03:57.308891Z","shell.execute_reply.started":"2024-11-24T20:03:57.293601Z","shell.execute_reply":"2024-11-24T20:03:57.308096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nthreshold = 50\ncolumns_to_keep = nan_percentage[nan_percentage<=threshold].index\ncolumns_to_keep = columns_to_keep.difference(['sii'])\ntrain_final = train_clean[list(columns_to_keep)+['sii']]\nfeature_types = train_final.apply(classify_feature)\ncat_features = [col for col in feature_types[feature_types.isin(['categorical'])].index]\nencoder = OrdinalEncoder()\ntrain_final[cat_features]=encoder.fit_transform(train_final[cat_features])\ntest_data[cat_features]=encoder.fit_transform(test_data[cat_features])\n\n\ntest = test_data[columns_to_keep]\nimputer = KNNImputer(n_neighbors=5)\ntrain_imputed = pd.DataFrame(imputer.fit_transform(train_final),columns=train_final.columns)\ntest_imputed = pd.DataFrame(imputer.fit_transform(test),columns=test.columns)\n\n#train_imputed[cat_features]=encoder.inverse_transform(train_imputed[cat_features])\n#test_imputed[cat_features] = encoder.inverse_transform(test_imputed[cat_features])\n\nfor col in cat_features:\n    train_imputed[col] = train_imputed[col].astype(int)\n    test_imputed[col] = test_imputed[col].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T21:40:04.267789Z","iopub.execute_input":"2024-11-24T21:40:04.268419Z","iopub.status.idle":"2024-11-24T21:40:05.923956Z","shell.execute_reply.started":"2024-11-24T21:40:04.268382Z","shell.execute_reply":"2024-11-24T21:40:05.923230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_imputed[cat_features].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T21:39:13.524005Z","iopub.execute_input":"2024-11-24T21:39:13.524825Z","iopub.status.idle":"2024-11-24T21:39:13.546469Z","shell.execute_reply.started":"2024-11-24T21:39:13.524786Z","shell.execute_reply":"2024-11-24T21:39:13.545544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CatBoost for Features Ranking","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport random\nimport os\nfrom catboost import CatBoostClassifier,Pool\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import cohen_kappa_score\nimport matplotlib.pyplot as plt\nfrom sklearn.multiclass import OneVsRestClassifier\n\nseed =42\nnp.random.seed(seed)\nrandom.seed(seed)\nos.environ['PYTHONHASHSEED'] = str(seed)\n\nX = train_imputed.drop('sii',axis=1)\ny = train_imputed['sii']\n\nX_train,X_test,y_train,y_test = train_test_split(X,y,test_size=0.2,random_state=42)\ntrain_pool = Pool(X_train,y_train,cat_features = cat_features)\ntest_pool = Pool(X_test,y_test,cat_features=cat_features)\n#def cohen_kappa_metric(y_true,y_pred): # turns out cohen_kappa is not built-in\n    #y_pred_classes = np.round(y_pred)\n    #return cohen_kappa_score(y_true,y_pred_classes)\n    \nmodel = CatBoostClassifier(\n    iterations = 5000,\n    learning_rate = 0.1,\n    task_type = 'GPU',\n    thread_count = -1,\n    eval_metric='Kappa',\n    random_seed = seed\n)\n\nmodel.fit(train_pool,eval_set=test_pool,verbose=100)\nfeature_importances = model.get_feature_importance(train_pool,type='PredictionValuesChange')\nfeature_names = X.columns.tolist()\n\nsorted_idx = np.argsort(feature_importances)\nsorted_features = [feature_names[i] for i in sorted_idx]\n\nplt.figure(figsize=(10,20))\nplt.barh(range(len(feature_importances)),feature_importances[sorted_idx])\nplt.yticks(range(len(feature_importances)),[feature_names[i] for i in sorted_idx])\nplt.xlabel('Prediction Values Change')\nplt.title('Feature Importances')\nplt.tight_layout()\nplt.show()\n\ncohen_kappa_scores = []\nfeature_counts = list(range(12,len(feature_names)+1,6))\nfeature_sets = []\n\nfor n_features in feature_counts:\n    selected_features = sorted_features[-n_features:]\n    feature_sets.append(selected_features)\n    X_train_selected = X_train[selected_features]\n    X_test_selected = X_test[selected_features]\n    train_pool_selected = Pool(X_train_selected,y_train)\n    test_pool_selected = Pool(X_test_selected,y_test)\n    model = CatBoostClassifier(\n        iterations=100,\n        learning_rate=0.1,\n        task_type='GPU',\n        thread_count = -1,\n        eval_metric='Kappa',\n        random_seed = 42\n    )\n    model.fit(train_pool_selected,eval_set=test_pool_selected,verbose=0)\n    y_pred = model.predict(X_test_selected)\n    kappa_score = cohen_kappa_score(y_test,y_pred)\n    print(f\"Number of Features: {n_features}, Cohen's Kappa: {kappa_score}\")\n    cohen_kappa_scores.append({'n_features':n_features,'kappa_score':kappa_score})\n\nkappa_values = [entry['kappa_score'] for entry in cohen_kappa_scores]\nbest_index = np.argmax(kappa_values)\nbest_num_features = feature_counts[best_index]\nbest_features = feature_sets[best_index]\n\nprint(f\"The highest kappa score of {kappa_values[best_index]:.4f} was achieved with {best_num_features} features.\")\nprint(\"The feature that resulted in the highest kappa score are:\")\nfor i, feature in enumerate(best_features,1):\n    print(f\"{i}.{feature}\")\n\nplt.figure(figsize=(10,6))\nplt.plot(feature_counts,kappa_values,marker='o')\nplt.xlabel('Number of Features')\nplt.ylabel('Cohen_Kappa Score')\nplt.title('Cohen_Kappa Score vs Number of Features')\nplt.grid(True)\nplt.tight_layout()\nplt.savefig('kappa_vs_features.png')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T21:40:11.244404Z","iopub.execute_input":"2024-11-24T21:40:11.244797Z","iopub.status.idle":"2024-11-24T21:41:20.539089Z","shell.execute_reply.started":"2024-11-24T21:40:11.244763Z","shell.execute_reply":"2024-11-24T21:41:20.538206Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now use the best sets of features for model training and optimization","metadata":{}},{"cell_type":"code","source":"og_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\nid = og_test['id']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T22:49:04.935784Z","iopub.execute_input":"2024-11-24T22:49:04.936884Z","iopub.status.idle":"2024-11-24T22:49:04.950661Z","shell.execute_reply.started":"2024-11-24T22:49:04.936831Z","shell.execute_reply":"2024-11-24T22:49:04.949403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimport optuna\n\ntrain_best = train_imputed[best_features]\ntest_best = test_imputed[best_features]\nbest_feature_types = train_best.apply(classify_feature)\nbest_cat_features = [col for col in best_feature_types[best_feature_types.isin(['categorical'])].index]\nX = train_best\ny = train_imputed['sii']\n\n#def evaluate_kappa(y_true,y_pred):\n    #return cohen_kappa_score(y_test,y_pred)\n\ndef catboost_objective(trial):\n    params = {\n        'iterations': trial.suggest_int('iterations',100,1000),\n        'depth':trial.suggest_int('depth',4,10),\n        'learning_rate': trial.suggest_float('learning_rate',1e-3,1.0,log=True),\n        'l2_leaf_reg':trial.suggest_float('l2_leaf_reg',1e-3,10.0,log=True),\n        'border_count':trial.suggest_int('border_count',32,255),\n        'bagging_temperature':trial.suggest_float('bagging_temperature',1e-1,10.0,log=True),\n        'task_type':'GPU',\n        'devices':'0'\n    }\n    cv = StratifiedKFold(n_splits=5,shuffle=True,random_state=42)\n    cv_scores = []\n    for train_idx,val_idx in cv.split(X,y):\n        #print(f\"val_idx length: {len(val_idx)}, y_val length: {len(y_val)}, y_pred length: {len(y_pred)}\")\n        X_train,X_val = X.iloc[train_idx],X.iloc[val_idx]\n        y_train,y_val = y.iloc[train_idx],y.iloc[val_idx]\n        model = CatBoostClassifier(**params,random_seed=42,verbose=0)\n        model.fit(\n            X_train,y_train,cat_features=best_cat_features,eval_set=(X_val,y_val),\n                 early_stopping_rounds=50,verbose=0\n        )\n        y_pred = model.predict(X_val)\n        cv_scores.append(cohen_kappa_score(y_val,y_pred))\n    return np.mean(cv_scores)\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T01:33:15.710387Z","iopub.status.idle":"2024-11-25T01:33:15.710756Z","shell.execute_reply.started":"2024-11-25T01:33:15.710594Z","shell.execute_reply":"2024-11-25T01:33:15.710612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    catboost_study = optuna.create_study(direction='maximize')\n    catboost_study.optimize(catboost_objective,n_trials=50)\n    best_catboost_params=catboost_study.best_params\n    # prepare for cross-validation\n    cv = StratifiedKFold(n_splits=5,shuffle=True,random_state=42)\n    catboost_oof = np.zeros((len(X),1)) #note: there are only 4 classes in the data\n    #test_catboost = np.zeros((len(test_imputed),4))\n\n    for fold,(train_idx,val_idx) in enumerate(cv.split(X,y)):\n        print(f\"Training fold {fold+1}\")\n        X_train,X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train,y_val = y.iloc[train_idx],y.iloc[val_idx]\n\n        catboost_model = CatBoostClassifier(**best_catboost_params,random_seed=42,verbose=0)\n        catboost_model.fit(X_train,y_train,eval_set=(X_val,y_val),early_stopping_rounds=50,verbose=0)\n        catboost_oof[val_idx] = catboost_model.predict(X_val)\n        #test_catboost += catboost_model.predict_proba(test_imputed)/5\n\n    # Train final model on all training data\n    catboost_model_final = CatBoostClassifier(**best_catboost_params,random_seed=42,verbose=0)\n    catboost_model_final.fit(X,y,cat_features=best_cat_features)\n    test_catboost_final = catboost_model_final.predict(test_best)\n    # Evaluate\n    catboost_kappa = cohen_kappa_score(y,catboost_oof)\n\n    print(f\"CatBoost Kappa: {catboost_kappa:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T01:33:53.020392Z","iopub.execute_input":"2024-11-25T01:33:53.021158Z","iopub.status.idle":"2024-11-25T01:51:07.062396Z","shell.execute_reply.started":"2024-11-25T01:33:53.021123Z","shell.execute_reply":"2024-11-25T01:51:07.061304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")\ns['sii'] = test_catboost_final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T01:54:11.214960Z","iopub.execute_input":"2024-11-25T01:54:11.215448Z","iopub.status.idle":"2024-11-25T01:54:11.222668Z","shell.execute_reply.started":"2024-11-25T01:54:11.215410Z","shell.execute_reply":"2024-11-25T01:54:11.221644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s.head()\ns.to_csv('/kaggle/working/submission.csv',index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T01:54:26.275690Z","iopub.execute_input":"2024-11-25T01:54:26.276339Z","iopub.status.idle":"2024-11-25T01:54:26.288585Z","shell.execute_reply.started":"2024-11-25T01:54:26.276300Z","shell.execute_reply":"2024-11-25T01:54:26.287487Z"}},"outputs":[],"execution_count":null}]}