{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T18:37:51.665054Z","iopub.execute_input":"2022-08-02T18:37:51.666077Z","iopub.status.idle":"2022-08-02T18:37:51.675710Z","shell.execute_reply.started":"2022-08-02T18:37:51.666028Z","shell.execute_reply":"2022-08-02T18:37:51.674476Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***In this Notebook I have included the following main steps***\n\n***- Data Preparation***\n\n***- Distribution of Features***\n\n***- Measures to choose the number of clusters***\n\n***- Preliminary investigations for features selection***\n\n***- Gaussian Mixture model***\n\n***- Bayesian Gaussian Mixture model***\n\n***- XGB classifier***\n\n***- Support Vector Machines***\n\n***- KNN classifier***\n\n***- Soft Voting***","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler,RobustScaler,PowerTransformer\nfrom sklearn.decomposition import PCA\nfrom sklearn.mixture import BayesianGaussianMixture,GaussianMixture\nfrom sklearn.metrics import silhouette_score, balanced_accuracy_score, roc_auc_score\nfrom sklearn.cluster import KMeans\nimport xgboost\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import RandomizedSearchCV\n\n# setting some globl config\nplt.style.use('fivethirtyeight')\ncust_color = ['#fdc029',\n'#f7c14c',\n'#f0c268',\n'#e8c381',\n'#dfc498',\n'#d4c5af',\n'#c6c6c6',\n'#a6a6a8',\n'#86868a',\n'#68686d',\n'#4b4c52',\n'#303138',\n'#171820',\n]\n\nplt.rcParams['figure.figsize'] = (18,14)\nplt.rcParams['figure.dpi'] = 300\nplt.rcParams[\"axes.grid\"] = True\nplt.rcParams[\"grid.color\"] = cust_color[3]\nplt.rcParams[\"grid.alpha\"] = 0.5\nplt.rcParams[\"grid.linestyle\"] = '--'\nplt.rcParams[\"font.family\"] = \"monospace\"\n\nplt.rcParams['axes.edgecolor'] = 'black'\nplt.rcParams['figure.frameon'] = True\nplt.rcParams['axes.spines.left'] = True\nplt.rcParams['axes.spines.bottom'] = True\nplt.rcParams['axes.spines.top'] = False\nplt.rcParams['axes.spines.right'] = False\nplt.rcParams['axes.linewidth'] = 1.5","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-04T13:51:06.981781Z","iopub.execute_input":"2022-08-04T13:51:06.982279Z","iopub.status.idle":"2022-08-04T13:51:06.994813Z","shell.execute_reply.started":"2022-08-04T13:51:06.982244Z","shell.execute_reply":"2022-08-04T13:51:06.993586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Reading Data </h1><a id=0></a>","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ndf=df.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:00:53.981081Z","iopub.execute_input":"2022-08-04T12:00:53.981493Z","iopub.status.idle":"2022-08-04T12:00:54.875904Z","shell.execute_reply.started":"2022-08-04T12:00:53.981460Z","shell.execute_reply":"2022-08-04T12:00:54.874624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Data Types and Missing Values </h1><a id=0></a>","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:10:14.759827Z","iopub.execute_input":"2022-08-03T17:10:14.760214Z","iopub.status.idle":"2022-08-03T17:10:14.794508Z","shell.execute_reply.started":"2022-08-03T17:10:14.760182Z","shell.execute_reply":"2022-08-03T17:10:14.793301Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Data is clean, there are no columns with missing values***\n\n***There are total 29 columns, out of which 7 are categorical columns and 22 are continuous columns***\n","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Distribution of Features </h1><a id=0>   \n    \n***Let's check the distribution of all columns in the data***","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(28, 18))\nplt.subplots_adjust(hspace=0.6)\nplt.suptitle(\"Feature Distributions\", fontsize=18, y=0.95)\n\nfor n, feature in enumerate(df.columns):\n    ax = plt.subplot(6, 5, n + 1)\n    sns.histplot(x=df[feature], color=cust_color[0])\n    ax.set_xlabel(f'Feature: {feature}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:37:52.869290Z","iopub.execute_input":"2022-08-02T18:37:52.869748Z","iopub.status.idle":"2022-08-02T18:38:07.306722Z","shell.execute_reply.started":"2022-08-02T18:37:52.869706Z","shell.execute_reply":"2022-08-02T18:38:07.305455Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***All the continuous columns are normally distributed with the mean at 0***\n\n***Categorical columns have poisson like distributions***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Data Scaling </h1><a id=0>\n    \n***For the clustering all the features need to be on the same scale, so let's scale the data using Power Transformer***","metadata":{}},{"cell_type":"code","source":"scaler = PowerTransformer()\n\nscaler.fit(df)\n\nX_scaled = scaler.transform(df)\n\ndf_scaled = pd.DataFrame(X_scaled, columns=df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:01:05.354549Z","iopub.execute_input":"2022-08-04T12:01:05.355062Z","iopub.status.idle":"2022-08-04T12:01:09.492050Z","shell.execute_reply.started":"2022-08-04T12:01:05.355024Z","shell.execute_reply":"2022-08-04T12:01:09.490814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***I also tried the standard scaler but it led to poor score on the LB***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Dimensionality Reduction </h1><a id=0>\n    \n***Most of the clustering algorithms suffer higher dimensions of features, so let's try to minimize the features with the help of principal component analysis***","metadata":{}},{"cell_type":"code","source":"pca = PCA(n_components=29)\npca.fit(df_scaled)\nvariance = pca.explained_variance_ratio_\nvar = np.cumsum(np.round(variance, 3)*100)\nfeatures = np.arange(1,30,1)\n\npca_variance = pd.DataFrame({'features': features, 'variance': var})\n\ncust_color_pca = ['#806000', '#997300', '#b38600', '#cc9900', '#e6ac00', '#ffbf00', '#fdc029', '#f7c14c', '#f0c268', '#e8c381','#dfc498','#d4c5af',\n'#B8B8B8', '#A9A9A9', '#A0A0A0', '#909090', '#808080', '#707070', '#696969', '#606060', '#585858', '#484848', '#404040',\n'#383838', '#303030', '#282828', '#202020', '#101010', '#000000']\n\nplt.figure(figsize=(12,6))\nsns.barplot(x=pca_variance.features, y=pca_variance.variance, data=pca_variance, edgecolor='black', linewidth=1.5, saturation=1.5, palette=cust_color_pca);\nplt.ylabel('% Variance Explained')\nplt.xlabel('# of Features')\nplt.title('PCA Analysis')\nplt.ylim(0,100.5);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-04T13:26:19.384438Z","iopub.execute_input":"2022-08-04T13:26:19.385479Z","iopub.status.idle":"2022-08-04T13:26:20.481084Z","shell.execute_reply.started":"2022-08-04T13:26:19.385440Z","shell.execute_reply":"2022-08-04T13:26:20.479432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***PCA analysis show that in order to explain 80% variance in the data, 22 principal components are required which is quite a large number for PCA. We normally try to reduce the dimensions in the range of 3-5. We will do the further analysis with all the features without PCA***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Number of Clusters </h1><a id=0>\n\n***Let's try to find the optimal clusters by using the elbow method for K-means clustering***","metadata":{}},{"cell_type":"code","source":"clusters = np.arange(3, 20)\n\ninertias = []\nsil_score = []\n\nfor k in clusters:\n    km = KMeans(n_clusters=k)\n    km.fit(df_scaled.iloc[:10000,:])\n    inertias.append(km.inertia_)\n    sil_score.append(silhouette_score(df_scaled.iloc[:10000,:], km.labels_))\n    \nkmeans = list(zip(clusters, inertias, sil_score))\nkmeans_df = pd.DataFrame(kmeans, columns=['clusters',  'interias', 'silhouette_score'])  ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T18:49:29.202601Z","iopub.execute_input":"2022-08-03T18:49:29.203628Z","iopub.status.idle":"2022-08-03T18:50:48.221106Z","shell.execute_reply.started":"2022-08-03T18:49:29.203589Z","shell.execute_reply":"2022-08-03T18:50:48.219577Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.lineplot(x='clusters', y='interias', data=kmeans_df, color='black', marker='o', markerfacecolor=cust_color[0], markersize=12);\nplt.ylabel('Interias')\nplt.xlabel('No. of Clusters')\nplt.title('Elbow Method', fontsize=20)\nplt.xticks(np.arange(3, 20, step=1));","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T19:49:37.278592Z","iopub.execute_input":"2022-08-03T19:49:37.279482Z","iopub.status.idle":"2022-08-03T19:49:38.157074Z","shell.execute_reply.started":"2022-08-03T19:49:37.279439Z","shell.execute_reply":"2022-08-03T19:49:38.156140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is not very clear that how many clusters are optimal based on the elbow method. So I would experiment with different values to check how they perform on the LB","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Priliminary Clustering Analysis </h1><a id=0>\n    \n***K-means and DBSCAN algorithms didn't perform well for this dataset. So I will go for Gaussian mixture model and Bayesian Gaussian mixture models. I will first train the Gaussian model for different number of clusters and plot the curves for AIC and BIC scores to identify the number of clusters based on these scores***","metadata":{}},{"cell_type":"code","source":"clusters = np.arange(3,20)\naic = []\nbic = []\n\nfor i in clusters:\n    gmm = GaussianMixture(n_components=i, random_state=3, verbose=0, n_init=1)\n    gmm.fit(df_scaled)\n    aic.append(gmm.aic(df_scaled))\n    bic.append(gmm.bic(df_scaled))\n\n\ngm = list(zip(clusters, aic, bic))\ngm_df = pd.DataFrame(gm, columns=['clusters', 'aic_score', 'bic_score'])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:03:49.748320Z","iopub.execute_input":"2022-08-02T20:03:49.748813Z","iopub.status.idle":"2022-08-02T20:17:13.096305Z","shell.execute_reply.started":"2022-08-02T20:03:49.748755Z","shell.execute_reply":"2022-08-02T20:17:13.094615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Akaike Information Criterion (AIC Score) </h1><a id=0>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,12))\nsns.lineplot(x='clusters', y='aic_score', data=gm_df, color= 'black', marker='o', markerfacecolor=cust_color[0], markersize=12);\nplt.xticks(np.arange(3, 20, step=1))\nplt.ylabel('AIC Score')\nplt.xlabel('Clusters');","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:17:13.100499Z","iopub.execute_input":"2022-08-02T20:17:13.100955Z","iopub.status.idle":"2022-08-02T20:17:14.973860Z","shell.execute_reply.started":"2022-08-02T20:17:13.100899Z","shell.execute_reply":"2022-08-02T20:17:14.972086Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Bayesian Information Criterion (BIC Score)</h1><a id=0>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,12))\nsns.lineplot(x='clusters', y='bic_score', data=gm_df, color='black', marker='o', markerfacecolor=cust_color[0], markersize=12);\nplt.xticks(np.arange(3, 20, step=1))\nplt.ylabel('BIC Score')\nplt.xlabel('Clusters');","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:17:14.975828Z","iopub.execute_input":"2022-08-02T20:17:14.976368Z","iopub.status.idle":"2022-08-02T20:17:16.714592Z","shell.execute_reply.started":"2022-08-02T20:17:14.976315Z","shell.execute_reply":"2022-08-02T20:17:16.713458Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Based on the AIC and BIC scores 11 seem to be optimal clusters and also a managable number from business point of view but most people are taking 7 clusters for the competition. I will experiment with both 7 and 11 to see where I land on the LB***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Preliminary Analysis </h1><a id=0>\n    \n***Let's train the Gaussian mixture model for 7 clusters to study the effect of features on the clusters***","metadata":{}},{"cell_type":"code","source":"gmm = GaussianMixture(n_components=7, random_state=3, verbose=3, n_init=10)\ngmm_preds = gmm.fit_predict(df_scaled)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-03T17:26:10.326759Z","iopub.execute_input":"2022-08-03T17:26:10.327301Z","iopub.status.idle":"2022-08-03T17:29:04.128799Z","shell.execute_reply.started":"2022-08-03T17:26:10.327257Z","shell.execute_reply":"2022-08-03T17:29:04.127105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Cluster Means </h1><a id=0>\n    \n***Let's plot the cluster means for all the features***","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nfor i in range(gmm.means_.shape[0]):\n    plt.scatter(np.arange(df_scaled.shape[1]), gmm.means_[i])\nplt.xticks(ticks=np.arange(df_scaled.shape[1]), labels=df_scaled.columns, rotation=90)\nplt.title('Cluster Means')\nplt.ylabel('Cluster Centres')\nplt.xlabel('Features');\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:30:16.174434Z","iopub.execute_input":"2022-08-03T17:30:16.174820Z","iopub.status.idle":"2022-08-03T17:30:17.202869Z","shell.execute_reply.started":"2022-08-03T17:30:16.174789Z","shell.execute_reply":"2022-08-03T17:30:17.201729Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***All clusters have all zero means for Features f_00 to f_06 and f_14 to f_21, so these features don't have to separate the clusters and don't give any meaningful information***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Cluster Covariances </h1><a id=0>\n    \n***Let's also check the cluster covariances. Covariances matrices are 3-dimensional but we can slice the data for each cluster and plot it separately***","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:50:50.980359Z","iopub.execute_input":"2022-08-02T18:50:50.980768Z","iopub.status.idle":"2022-08-02T18:50:50.988125Z","shell.execute_reply.started":"2022-08-02T18:50:50.980735Z","shell.execute_reply":"2022-08-02T18:50:50.987130Z"}}},{"cell_type":"code","source":"for i in range(len(gmm.covariances_)):\n    plt.figure(figsize=(20, 20))\n    sns.heatmap(gmm.covariances_[i], annot=True, cmap='YlOrBr', annot_kws={\"fontsize\":12}, fmt='.2f')\n    plt.title(f'Cluster {i}', fontname = 'monospace', weight='bold')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:20:05.602068Z","iopub.execute_input":"2022-08-02T20:20:05.602394Z","iopub.status.idle":"2022-08-02T20:20:46.557503Z","shell.execute_reply.started":"2022-08-02T20:20:05.602364Z","shell.execute_reply":"2022-08-02T20:20:46.556338Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***The covariance matrices of the clusters show that the features f_00 to f_07 and f_14 to f_21 have zero covariances, these features aren't partcipating in the cluster analysis. Perhaps we should drop them from our analysis***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Useful Features </h1><a id=0>\n    \n***Let's drop out the features with zero mean and covariance and keep only the useful features for the analysis***","metadata":{}},{"cell_type":"code","source":"useful_features= ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\ndf_scaled_useful = df_scaled[useful_features]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:01:39.431668Z","iopub.execute_input":"2022-08-04T12:01:39.433271Z","iopub.status.idle":"2022-08-04T12:01:39.444155Z","shell.execute_reply.started":"2022-08-04T12:01:39.433217Z","shell.execute_reply":"2022-08-04T12:01:39.442903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Clustering </h1><a id=0>\n                \n***After the preliminary analysis and features selection, I will test the follwoing techniques and make different submissions to see how the results perform on the LB***\n    \n***1. Gaussian Mixture Model***\n    \n***2. Bayesian Gaussian Mixture Model***\n    \n***3. XGB Classifier***\n    \n***4. Soft voting with XGB Classifier, Support Vector Machines, KNN***","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:54:36.270117Z","iopub.execute_input":"2022-08-04T11:54:36.270771Z","iopub.status.idle":"2022-08-04T11:54:36.305061Z","shell.execute_reply.started":"2022-08-04T11:54:36.270646Z","shell.execute_reply":"2022-08-04T11:54:36.303751Z"}}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Gaussian Mixture Model </h1><a id=0>\n    \n***Let's train the Gaussian Mixture model for 7 and 11 clusters and see which perform the better***","metadata":{}},{"cell_type":"code","source":"gmm = GaussianMixture(n_components=7, random_state=3, verbose=0, n_init=10, max_iter=500)\ngmm_preds = gmm.fit_predict(df_scaled_useful)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:13:51.293420Z","iopub.execute_input":"2022-08-04T12:13:51.293886Z","iopub.status.idle":"2022-08-04T12:16:19.190393Z","shell.execute_reply.started":"2022-08-04T12:13:51.293852Z","shell.execute_reply":"2022-08-04T12:16:19.188443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\n\nsub['Predicted'] = gmm_preds\nsub.to_csv(\"submission_gmm_7.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:16:51.161040Z","iopub.execute_input":"2022-08-04T12:16:51.161576Z","iopub.status.idle":"2022-08-04T12:16:51.289077Z","shell.execute_reply.started":"2022-08-04T12:16:51.161519Z","shell.execute_reply":"2022-08-04T12:16:51.287562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***11 clusters produced a score of 0.584 on the LB***","metadata":{}},{"cell_type":"code","source":"gmm_11 = GaussianMixture(n_components=11, random_state=3, verbose=0, n_init=10, max_iter=500)\ngmm_preds_11 = gmm_11.fit_predict(df_scaled_useful)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:03:34.737302Z","iopub.execute_input":"2022-08-04T12:03:34.738285Z","iopub.status.idle":"2022-08-04T12:07:44.430393Z","shell.execute_reply.started":"2022-08-04T12:03:34.738225Z","shell.execute_reply":"2022-08-04T12:07:44.428751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\n\nsub['Predicted'] = gmm_preds_11\nsub.to_csv(\"submission_gmm_11.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T12:11:02.963245Z","iopub.execute_input":"2022-08-04T12:11:02.964072Z","iopub.status.idle":"2022-08-04T12:11:03.120515Z","shell.execute_reply.started":"2022-08-04T12:11:02.964029Z","shell.execute_reply":"2022-08-04T12:11:03.119504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***11 clusters produced a score of 0.474 on the LB***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Bayesian Gaussian Mixture Model </h1><a id=0>\n    \n***Let's train the Bayesian Gaussian Mixture model for 7 and 11 clusters***","metadata":{}},{"cell_type":"code","source":"bgmm = BayesianGaussianMixture(n_components=7, random_state=3, verbose=0, covariance_type='full', max_iter=500, n_init=10, init_params='random')\nbgmm_preds_7 = bgmm.fit_predict(df_scaled_useful)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:35:14.170624Z","iopub.execute_input":"2022-08-03T17:35:14.172538Z","iopub.status.idle":"2022-08-03T17:45:42.895903Z","shell.execute_reply.started":"2022-08-03T17:35:14.172457Z","shell.execute_reply":"2022-08-03T17:45:42.894075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsub['Predicted'] = bgmm_preds_7\nsub.to_csv(\"submission_bgmm_7.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:38:52.931238Z","iopub.status.idle":"2022-08-02T18:38:52.931958Z","shell.execute_reply.started":"2022-08-02T18:38:52.931746Z","shell.execute_reply":"2022-08-02T18:38:52.931768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***7 clusters with BGMM model produced a score of 0.601 on the LB***","metadata":{}},{"cell_type":"code","source":"bgmm_11 = BayesianGaussianMixture(n_components=11, random_state=3, verbose=0, covariance_type='full', max_iter=500, n_init=10, init_params='random')\nbgmm_preds_11 = bgmm_11.fit_predict(df_scaled_useful)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:45:42.900087Z","iopub.execute_input":"2022-08-03T17:45:42.900744Z","iopub.status.idle":"2022-08-03T18:16:59.317187Z","shell.execute_reply.started":"2022-08-03T17:45:42.900681Z","shell.execute_reply":"2022-08-03T18:16:59.315454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsub['Predicted'] = bgmm_preds_11\nsub.to_csv(\"submission_bgmm_11.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:38:52.935005Z","iopub.status.idle":"2022-08-02T18:38:52.935420Z","shell.execute_reply.started":"2022-08-02T18:38:52.935191Z","shell.execute_reply":"2022-08-02T18:38:52.935208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***11 clusters with BGMM model produced a score of 0.526 on the LB***\n\n***So the BGMM model for 7 clusters has the best score and will be considered for durther analysis***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\"> Classification after Clustering </h1><a id=0>\n    \n*** The main idea here is to get the labels for those clusters which have the probability of more than 80% and then train a classification model to predict the cluster lables based on the useful features and labels with higher probabilities***\n    \n***The code below is inspired from @aldparis. Thanks to him***","metadata":{}},{"cell_type":"code","source":"pp=bgmm.predict_proba(df_scaled_useful)# Calcualting the probabilities of each prediction\ndf_new=pd.DataFrame(df_scaled_useful,columns=useful_features) \ndf_new[[f'predict_proba_{i}' for i in range(7)]]=pp # creating new dataframe columns of probabilites \ndf_new['preds']=bgmm_preds_7\ndf_new['predict_proba']=np.max(pp,axis=1)\ndf_new['predict']=np.argmax(pp,axis=1)\n    \ntrain_index=np.array([])\nfor n in range(7):\n    n_inx=df_new[(df_new.preds==n) & (df_new.predict_proba > 0.8)].index\n    train_index = np.concatenate((train_index, n_inx))\n    \nX=df_new.loc[train_index][useful_features]\ny=df_new.loc[train_index]['preds']\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:11:18.582990Z","iopub.execute_input":"2022-08-03T14:11:18.583507Z","iopub.status.idle":"2022-08-03T14:11:18.947951Z","shell.execute_reply.started":"2022-08-03T14:11:18.583471Z","shell.execute_reply":"2022-08-03T14:11:18.946706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Hyperparameter Tuning for XGBoost </h1><a id=0>\n    \n***I will first train the XGB classifier to predict the cluster labels as discussed above. Before that I will tune some important hyperparameters using the randomized search***","metadata":{}},{"cell_type":"code","source":"params = {\n    'learning_rate': [0.05, 0.1, 0.15, 0.2, 0.25, 0.3],\n    'max_depth': [3, 4, 5, 6, 8, 10, 12, 15],\n    'min_child_weight': [1, 3, 5, 7],\n    'gamma': [0.0, 0.1, 0.2, 0.3, 0.4],\n    'subsample': [0, 0.25, 0.5, 0.75, 1],\n    'colsample_bytree': [0, 0.25, 0.5, 0.75, 1]\n}\nclassifier = xgboost.XGBClassifier(objective = 'multi:softprob', num_class=7)\nrandom_search = RandomizedSearchCV(classifier, param_distributions = params, scoring='neg_log_loss', n_iter=10,n_jobs=-1, cv=10)\nrandom_search.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:50:50.077989Z","iopub.execute_input":"2022-08-03T18:50:50.078752Z","iopub.status.idle":"2022-08-03T19:11:07.728641Z","shell.execute_reply.started":"2022-08-03T18:50:50.078715Z","shell.execute_reply":"2022-08-03T19:11:07.727402Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_search.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:26:52.078144Z","iopub.execute_input":"2022-08-03T19:26:52.080499Z","iopub.status.idle":"2022-08-03T19:26:52.093098Z","shell.execute_reply.started":"2022-08-03T19:26:52.080401Z","shell.execute_reply":"2022-08-03T19:26:52.091431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  XGBoost Classifier</h1><a id=0>\n    \n***After getting the tuned hyperparameters, XGB classifier was trained. For the cross validation 10 fold Stratifiedkfold strategy was used and balanced accuracy score and roc auc score were used as a creteria to judge the performance of the classifier on the validation data. Finally the results from all the 10 folds were averaged to get the probabilities of the clsuter labels***","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nparams_xgb = {'objective':'multi:softprob', 'num_class':7, 'subsample':0.75, 'min_child_weight': 1,\n              'max_depth' : 15, 'learning_rate': 0.1, 'gamma':0.2, 'colsample_bytree':0.5, 'n_jobs':-1} \n\nxgb_acc_scores = []\nxgb_roc_scores = []\nxgb_predict_proba = 0\n\n\ngkf = StratifiedKFold(10, shuffle=True, random_state = 3)\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(X,y)):   \n\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n    dtrain = xgboost.DMatrix(X_train, label=y_train)\n    dvalid = xgboost.DMatrix(X_val, label=y_val)\n    watchlist = [(dtrain, 'train'), (dvalid, 'eval')]                  \n    \n    model = xgboost.train(params = params_xgb, \n                dtrain=dtrain, \n                evals=watchlist, \n                num_boost_round = 1000, \n                early_stopping_rounds=30,\n                verbose_eval = 200)\n                          \n    y_pred_proba = model.predict(xgboost.DMatrix(X.iloc[val_idx]))\n    y_pred = np.argmax(y_pred_proba, axis=1)\n    \n    xgb_acc_scores.append(balanced_accuracy_score(y.iloc[val_idx], y_pred))\n    xgb_roc_scores.append(roc_auc_score(y.iloc[val_idx], y_pred_proba, average=\"weighted\", multi_class=\"ovo\"))                      \n    \n    xgb_predict_proba += model.predict(xgboost.DMatrix(df_new[useful_features]))/10\n                          \nxgb_scores = pd.DataFrame(list(zip(xgb_acc_scores, xgb_roc_scores)), columns=['balanced_accuracy_score', 'roc_auc_score'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:27:41.369965Z","iopub.execute_input":"2022-08-03T15:27:41.370518Z","iopub.status.idle":"2022-08-03T16:37:43.623171Z","shell.execute_reply.started":"2022-08-03T15:27:41.370476Z","shell.execute_reply":"2022-08-03T16:37:43.622240Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_scores","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:38:55.230055Z","iopub.execute_input":"2022-08-03T16:38:55.230533Z","iopub.status.idle":"2022-08-03T16:38:55.243169Z","shell.execute_reply.started":"2022-08-03T16:38:55.230498Z","shell.execute_reply":"2022-08-03T16:38:55.241847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***The XGB calssifier performed very well on the validation data as canbe seen from the performace metrices***","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsub['Predicted'] = labels[0]\nsub.to_csv(\"submission_xgb.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:38:52.972892Z","iopub.status.idle":"2022-08-02T18:38:52.973318Z","shell.execute_reply.started":"2022-08-02T18:38:52.973096Z","shell.execute_reply":"2022-08-02T18:38:52.973115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***The submission with XGB classifier has 0.615 score on the LB, a marginal improvement as compared to the BGMM model***","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Support Vector Machines </h1><a id=0>\n    \n***Next I trained the Support Vector Mchines and KNN classifiers using the 10 fold StratifiedKFold methodolgy. Both the classifiers showed good performance on the validation data***","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nsvm_acc_scores = []\nsvm_roc_scores = []\nsvm_predict_proba = 0\n\ngkf = StratifiedKFold(10, shuffle=True, random_state = 3)\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(X,y)):   \n\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n    \n    model = SVC(probability=True)\n    model.fit(X_train, y_train)\n    \n    y_pred = model.predict(X_val)\n    y_pred_proba = model.predict_proba(X_val)\n    \n    svm_acc_scores.append(balanced_accuracy_score(y_val, y_pred))\n    svm_roc_scores.append(roc_auc_score(y_val, y_pred_proba, average=\"weighted\", multi_class=\"ovo\"))\n    \n    svm_predict_proba += model.predict_proba(df_new[useful_features])/10\n    \nsvm_scores = pd.DataFrame(list(zip(svm_acc_scores, svm_roc_scores)), columns=['balanced_accuracy_score', 'roc_auc_score'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:16:59.319882Z","iopub.execute_input":"2022-08-03T18:16:59.320948Z","iopub.status.idle":"2022-08-03T18:39:49.817976Z","shell.execute_reply.started":"2022-08-03T18:16:59.320864Z","shell.execute_reply":"2022-08-03T18:39:49.816574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svm_scores","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:23:17.812250Z","iopub.execute_input":"2022-08-03T15:23:17.812655Z","iopub.status.idle":"2022-08-03T15:23:17.823876Z","shell.execute_reply.started":"2022-08-03T15:23:17.812616Z","shell.execute_reply":"2022-08-03T15:23:17.822900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  KNN Classifier </h1><a id=0>","metadata":{}},{"cell_type":"code","source":"knn_acc_scores = []\nknn_roc_scores = []\nknn_predict_proba = 0\n\ngkf = StratifiedKFold(10, shuffle=True, random_state = 3)\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(X,y)):   \n\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n    \n    model = KNeighborsClassifier(n_neighbors=20)\n    model.fit(X_train, y_train)\n    \n    y_pred = model.predict(X_val)\n    y_pred_proba = model.predict_proba(X_val)\n    \n    knn_acc_scores.append(balanced_accuracy_score(y_val, y_pred))\n    knn_roc_scores.append(roc_auc_score(y_val, y_pred_proba, average=\"weighted\", multi_class=\"ovo\"))\n    \n    knn_predict_proba += model.predict_proba(df_new[useful_features])/10\n                          \nknn_scores = pd.DataFrame(list(zip(knn_acc_scores, knn_roc_scores)), columns=['balanced_accuracy_score', 'roc_auc_score'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:50:13.581190Z","iopub.execute_input":"2022-08-03T14:50:13.581529Z","iopub.status.idle":"2022-08-03T15:19:24.684541Z","shell.execute_reply.started":"2022-08-03T14:50:13.581501Z","shell.execute_reply":"2022-08-03T15:19:24.683552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn_scores","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:23:54.197662Z","iopub.execute_input":"2022-08-03T15:23:54.198060Z","iopub.status.idle":"2022-08-03T15:23:54.209839Z","shell.execute_reply.started":"2022-08-03T15:23:54.198029Z","shell.execute_reply":"2022-08-03T15:23:54.208941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Soft Voting </h1><a id=0>\n    \n***Soft voting relies on the probabilities of the predicted classes. Probabilities of each predicted classes from all the classifiers are averaged and then the class label is predicted based on the highest averaged probaility***\n    \n***Code below is inspired from @ricopue. Thanks to him***","metadata":{}},{"cell_type":"code","source":"def soft_voting(preds_probs):\n\n    values = list(range(7))\n    pred_test = pd.DataFrame(np.zeros((df.shape[0], 7)), columns = values)\n\n    for i, p in enumerate(preds_probs):\n    \n        MAX = np.argmax(p, axis=1)\n        df[f'pred_{i}'] = MAX\n    \n        # Sort of the prediction by same value of cluster\n        pred_keys = df[f'pred_{i}'].value_counts().index.tolist()\n        pred_dict = dict(zip(pred_keys, values))\n        df[f'pred_{i}'] = df[f'pred_{i}'].map(pred_dict)\n\n        pred_new = pd.DataFrame(p).rename(columns = pred_dict)\n        pred_new = pred_new.reindex(sorted(pred_new.columns), axis=1)\n        pred_test += pred_new # Soft voting by probabiliy addition\n\n    return np.argmax(np.array(pred_test), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:52.759361Z","iopub.execute_input":"2022-08-03T16:45:52.759751Z","iopub.status.idle":"2022-08-03T16:45:52.769649Z","shell.execute_reply.started":"2022-08-03T16:45:52.759720Z","shell.execute_reply":"2022-08-03T16:45:52.768056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sv_predict = soft_voting([xgb_predict_proba, svm_predict_proba, knn_predict_proba])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:58.706618Z","iopub.execute_input":"2022-08-03T16:45:58.707052Z","iopub.status.idle":"2022-08-03T16:45:58.760944Z","shell.execute_reply.started":"2022-08-03T16:45:58.707017Z","shell.execute_reply":"2022-08-03T16:45:58.759659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsub['Predicted'] = sv_predict\nsub.to_csv(\"submission_softvoting3.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:46:11.304777Z","iopub.execute_input":"2022-08-03T16:46:11.305181Z","iopub.status.idle":"2022-08-03T16:46:11.435793Z","shell.execute_reply.started":"2022-08-03T16:46:11.305144Z","shell.execute_reply":"2022-08-03T16:46:11.434866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Final predictions after the soft voting from the 3 classifiers produced a score of 0.626 on the LB, which is again a marginal improvement as compared to the results of only XGB classifier***","metadata":{}},{"cell_type":"code","source":"clustering_scores =  {'GMM':0.584, 'BGMM':0.601, 'XGB Classifier':0.615, 'Soft Voting':0.626}\nclustering_scores_df = pd.DataFrame()\n\nclustering_scores_df['Method'] = clustering_scores.keys()\nclustering_scores_df['LB Score'] = clustering_scores.values()\n\nclustering_scores_df\n\nplt.figure(figsize=(12,6))\nsns.barplot(x= 'Method', y= 'LB Score', data=clustering_scores_df, edgecolor='black', linewidth=1.5, saturation=1.5, palette=cust_color[::4]);\nplt.ylabel('LB Score')\nplt.xlabel('Method')\nplt.title('Algorithms Performance');\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-04T13:29:02.448575Z","iopub.execute_input":"2022-08-04T13:29:02.450483Z","iopub.status.idle":"2022-08-04T13:29:03.148316Z","shell.execute_reply.started":"2022-08-04T13:29:02.450381Z","shell.execute_reply":"2022-08-04T13:29:03.147268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:#171820\n;font-family:newtimeroman;font-size:400%;text-align:left; color:#fdc029\">  Conclusions and Recommendations </h1><a id=0>\n    \n***1. K-means and DBSCAN clustering algorithms are not suited for this data***\n    \n***2. Gaussain Mixture model and Bayesian Gaussian mixture models produced robust results***\n    \n***3. Performance of Bayesian Gaussian mixture model is better than the Gaussian Mixture model***\n    \n***4. Only the clustering algorith it was possible to acheive only certain level of performance***\n    \n***5. Classifiers help improved the performance of predictions but only marginally***\n    \n***6. SVM and KNN classifiers can be tuned to check if they perform better***\n    \n***7. More classifiers like Bayesian Gaussain classifier, Naive Bayes classifier to check if they produce better performance***","metadata":{}}]}