{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://media.giphy.com/media/PAqjdPkJLDsmBRSYUp/giphy.gif\" width=80%>","metadata":{}},{"cell_type":"code","source":"\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T17:16:55.845789Z","iopub.execute_input":"2022-07-17T17:16:55.846449Z","iopub.status.idle":"2022-07-17T17:16:55.874825Z","shell.execute_reply.started":"2022-07-17T17:16:55.846356Z","shell.execute_reply":"2022-07-17T17:16:55.873971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != 'object':\n            c_min = df[col].min()\n            c_max = df[col].max()\n            \n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    pass\n        else:\n            df[col] = df[col].astype('category')\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:16:55.876437Z","iopub.execute_input":"2022-07-17T17:16:55.876759Z","iopub.status.idle":"2022-07-17T17:16:55.889096Z","shell.execute_reply.started":"2022-07-17T17:16:55.876723Z","shell.execute_reply":"2022-07-17T17:16:55.887951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv(\"../input/tabular-playground-series-dec-2021/train.csv\")\nreduce_memory_usage(df)\ntest=pd.read_csv(\"../input/tabular-playground-series-dec-2021/test.csv\")\nreduce_memory_usage(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:16:55.890714Z","iopub.execute_input":"2022-07-17T17:16:55.891254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://media.giphy.com/media/l4RKhOL0xiBdbgglFi/giphy.gif\" width=50%>","metadata":{}},{"cell_type":"code","source":"df.describe().T","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking for NULLs in the data","metadata":{}},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking for data types in the DataFrame","metadata":{"execution":{"iopub.status.busy":"2021-12-05T07:09:56.392479Z","iopub.execute_input":"2021-12-05T07:09:56.394024Z","iopub.status.idle":"2021-12-05T07:09:56.400535Z","shell.execute_reply.started":"2021-12-05T07:09:56.393954Z","shell.execute_reply":"2021-12-05T07:09:56.399472Z"}}},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in df.columns:\n    print(f\"The total unique values in {col} are {len(df[col].unique())}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As **Soil_Type7** and **Soil_Type15** are  having only 1 type of data need to be removed from the data frame","metadata":{}},{"cell_type":"code","source":"df.drop([\"Soil_Type7\",\"Soil_Type15\"],axis=1,inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://media.giphy.com/media/XfnuZsoKN5VCjyynHn/giphy.gif\">","metadata":{}},{"cell_type":"markdown","source":"# Data Description\n- Elevation - Elevation in meters\n- Aspect - Aspect in degrees azimuth\n- Slope - Slope in degrees\n- Horizontal_Distance_To_Hydrology - Horz Dist to nearest surface water features\n- Vertical_Distance_To_Hydrology - Vert Dist to nearest surface water features\n- Horizontal_Distance_To_Roadways - Horz Dist to nearest roadway\n- Hillshade_9am (0 to 255 index) - Hillshade index at 9am, summer solstice\n- Hillshade_Noon (0 to 255 index) - Hillshade index at noon, summer solstice\n- Hillshade_3pm (0 to 255 index) - Hillshade index at 3pm, summer solstice\n- Horizontal_Distance_To_Fire_Points - Horz Dist to nearest wildfire ignition points\n- Wilderness_Area (4 binary columns, 0 = absence or 1 = presence) - Wilderness area designation\n- Soil_Type (40 binary columns, 0 = absence or 1 = presence) - Soil Type designation\n- Cover_Type (7 types, integers 1 to 7) - Forest Cover Type designation","metadata":{}},{"cell_type":"code","source":"import matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nplt.style.use('ggplot')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nsns.countplot(df.Cover_Type)\nplt.plot()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see there is imbalance in the dataset","metadata":{}},{"cell_type":"code","source":"df['Cover_Type'].value_counts(ascending=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"try:\n    fig, axes=plt.subplots(2,5,figsize=(30,15))\n    j=0\n    i=0\n    for k in range(1,11):\n        if j==5:\n            i+=1\n            j=0\n        sns.kdeplot(df.loc[:,df.columns[k]],ax=axes[i,j])\n        plt.gca().set_title(f\"{df.columns[k]}\")\n        j+=1\nexcept:\n    print(\"Got all the columns\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting Outliers\n\n<img src=\"https://media.giphy.com/media/OSUuEuaz0imBy/giphy.gif\">","metadata":{}},{"cell_type":"code","source":"def outlier_function(df, col_name):\n    first_quartile = np.percentile(np.array(df[col_name].tolist()), 25)\n    third_quartile = np.percentile(np.array(df[col_name].tolist()), 75)\n    IQR = third_quartile - first_quartile\n    \n    upper_limit = third_quartile+(3*IQR)\n    lower_limit = first_quartile-(3*IQR)\n    outlier_count = 0\n    \n    for value in df[col_name].tolist():\n        if (value < lower_limit) | (value > upper_limit):\n            outlier_count += 1\n    return lower_limit, upper_limit, outlier_count","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in  df.columns[:10]:\n    out=outlier_function(df,col)\n    if out[2]>0:\n        print(f\"There are {out[2]} outliers in {col}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    fig_out, axes_out=plt.subplots(2,5,figsize=(30,15))\n    j=0\n    i=0\n    for k in range(1,11):\n        if j==5:\n            i+=1\n            j=0\n        sns.boxplot(y=df.columns[k],x=df.columns[-1],data=df,ax=axes_out[i,j])\n        plt.gca().set_title(f\"{df.columns[k]}\")\n        j+=1\nexcept:\n    print(\"Got all the columns\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.corr())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"cb_params = {'iterations': 10000,\n             'learning_rate': 0.218904169525507,\n             'loss_function': 'MultiClass',\n             'eval_metric': 'Accuracy',\n             'l2_leaf_reg': 1.6163189485316596,\n             'bagging_temperature': 0.14353551008899088,\n             'random_strength': 1.29,\n             'depth': 10,\n             'grow_policy': 'SymmetricTree',\n             'leaf_estimation_method': 'Gradient',\n             'od_type': 'Iter',\n             'early_stopping_rounds': 300,\n             'border_count': 254,\n             'use_best_model': True,\n             'min_data_in_leaf': 150,\n             'task_type': 'GPU',\n             'random_seed': 42}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix,classification_report,accuracy_score,roc_auc_score\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nimport pickle\nfrom sklearn import model_selection\nfrom catboost import CatBoostClassifier","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Data for Minority Classes","metadata":{}},{"cell_type":"code","source":"df=pd.concat([df,\n              df[df[\"Cover_Type\"]==5],\n              df[df[\"Cover_Type\"]==5],\n              df[df[\"Cover_Type\"]==5],\n              df[df[\"Cover_Type\"]==5],\n              df[df[\"Cover_Type\"]==5],\n              df[df[\"Cover_Type\"]==5],\n              df[df[\"Cover_Type\"]==4],\n              df[df[\"Cover_Type\"]==4],\n              df[df[\"Cover_Type\"]==4],\n              df[df[\"Cover_Type\"]==4],\n              df[df[\"Cover_Type\"]==4],\n              df[df[\"Cover_Type\"]==4]],ignore_index=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop([\"Soil_Type7\",\"Soil_Type15\"],axis=1,inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Thanks for the [Discussion](https://www.kaggle.com/c/tabular-playground-series-dec-2021/discussion/293373) , Providing such usefull insites of the data ","metadata":{}},{"cell_type":"code","source":"df[\"Aspect\"][df[\"Aspect\"] < 0] += 360\ndf[\"Aspect\"][df[\"Aspect\"] > 359] -= 360\n\ntest[\"Aspect\"][test[\"Aspect\"] < 0] += 360\ntest[\"Aspect\"][test[\"Aspect\"] > 359] -= 360","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df[\"Hillshade_9am\"] < 0, \"Hillshade_9am\"] = 0\ntest.loc[test[\"Hillshade_9am\"] < 0, \"Hillshade_9am\"] = 0\n\ndf.loc[df[\"Hillshade_Noon\"] < 0, \"Hillshade_Noon\"] = 0\ntest.loc[test[\"Hillshade_Noon\"] < 0, \"Hillshade_Noon\"] = 0\n\ndf.loc[df[\"Hillshade_3pm\"] < 0, \"Hillshade_3pm\"] = 0\ntest.loc[test[\"Hillshade_3pm\"] < 0, \"Hillshade_3pm\"] = 0\n\ndf.loc[df[\"Hillshade_9am\"] > 255, \"Hillshade_9am\"] = 255\ntest.loc[test[\"Hillshade_9am\"] > 255, \"Hillshade_9am\"] = 255\n\ndf.loc[df[\"Hillshade_Noon\"] > 255, \"Hillshade_Noon\"] = 255\ntest.loc[test[\"Hillshade_Noon\"] > 255, \"Hillshade_Noon\"] = 255\n\ndf.loc[df[\"Hillshade_3pm\"] > 255, \"Hillshade_3pm\"] = 255\ntest.loc[test[\"Hillshade_3pm\"] > 255, \"Hillshade_3pm\"] = 255","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_col=df.columns[1:-1]\nX=df[feature_col]\ny=df[\"Cover_Type\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test=test[feature_col]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 👍 Building the CatBoost Model\n\n\n<img src=\"https://media.giphy.com/media/lJNoBCvQYp7nq/giphy.gif\">","metadata":{}},{"cell_type":"code","source":"%%time\n# Setting up fold parameters\nsplits = 5\nskf = model_selection.StratifiedKFold(n_splits=splits, shuffle=True, random_state=42)\n\n# Creating an array of zeros for storing \"out of fold\" predictions\noof_preds = np.zeros((X.shape[0],))\npreds = np.zeros((X_test.shape[0],len(np.unique(y))))\nmodel_fi = 0\ntotal_mean_acc = 0\n\n# Generating folds and making training and prediction for each of 10 folds\nfor num, (train_idx, valid_idx) in enumerate(skf.split(X, y)):\n    X_train, X_valid = X.loc[train_idx], X.loc[valid_idx]\n    y_train, y_valid = y.loc[train_idx], y.loc[valid_idx]\n    \n    model = CatBoostClassifier(**cb_params)\n    model.fit(X_train, y_train,\n              verbose=False,\n              eval_set=(X_valid, y_valid),\n              )\n    \n    # Getting mean test data predictions (i.e. devided by number of splits)\n    preds += model.predict_proba(X_test) / splits\n    \n    # Getting mean feature importances (i.e. devided by number of splits)\n    model_fi += model.feature_importances_ / splits\n    \n    # Getting validation data predictions. Each fold model makes predictions on an unseen data.\n    # So in the end it will be completely filled with unseen data predictions.\n    # It will be used to evaluate hyperparameters performance only.\n    \n    oof_preds[valid_idx] = model.predict(X_valid).flatten()\n    \n    # Getting score for a fold model\n    fold_acc = accuracy_score(y_valid, oof_preds[valid_idx])\n    \n    print(f\"Fold {num} accuracy: {fold_acc}\")\n    print(classification_report(y_valid,oof_preds[valid_idx]))\n    \n    # Getting mean score of all fold models (i.e. devided by number of splits)\n    total_mean_acc += fold_acc / splits\n    \nprint(f\"\\nOverall ROC AUC: {total_mean_acc}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(45,30))\nplt.rcParams.update({'font.size': 30})\nidxs = np.argsort(model_fi)\nplt.title(\"Feature Importance\")\nplt.barh(range(len(idxs)),model_fi[idxs],align=\"center\")\nplt.yticks(range(len(idxs)),[feature_col[i] for i in idxs])\nplt.xlabel(\"Random Forest Feature Importance\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result=pd.DataFrame(model.predict(X_test))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit=pd.concat([pd.DataFrame(test[\"Id\"]),result],axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.columns=[\"Id\",\"Cover_Type\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Please Upvote if you Liked what you saw!! Helps a lot😁","metadata":{}},{"cell_type":"markdown","source":"<img src=\"https://media.giphy.com/media/4LM3elgbccSje/giphy.gif\">","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}