{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular Playground Series - Jul 2022","metadata":{}},{"cell_type":"markdown","source":"**unsupervised clustering challenge!*\n* K-means resulted in poor score.\n* Yellowbrick is used for guessing initial clusters requiremnets.\n* BayesianGaussianMixture used for predicting cluster for a given data points.\n* Power Transformer followed by RobustScaler is used for scaling  beacuse of its adavantage in dealing with outliers.","metadata":{}},{"cell_type":"code","source":"\n\nimport pandas as pd\nfrom sklearn.preprocessing import*\nfrom sklearn.cluster import KMeans\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis, LinearDiscriminantAnalysis\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:34:58.628667Z","iopub.execute_input":"2022-07-31T20:34:58.629184Z","iopub.status.idle":"2022-07-31T20:35:01.155971Z","shell.execute_reply.started":"2022-07-31T20:34:58.629090Z","shell.execute_reply":"2022-07-31T20:35:01.154463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install sklego","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-31T20:35:01.158333Z","iopub.execute_input":"2022-07-31T20:35:01.158690Z","iopub.status.idle":"2022-07-31T20:35:18.529750Z","shell.execute_reply.started":"2022-07-31T20:35:01.158658Z","shell.execute_reply":"2022-07-31T20:35:18.528720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklego.mixture import BayesianGMMClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:18.531443Z","iopub.execute_input":"2022-07-31T20:35:18.531827Z","iopub.status.idle":"2022-07-31T20:35:18.547592Z","shell.execute_reply.started":"2022-07-31T20:35:18.531779Z","shell.execute_reply":"2022-07-31T20:35:18.546260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:18.549603Z","iopub.execute_input":"2022-07-31T20:35:18.550147Z","iopub.status.idle":"2022-07-31T20:35:19.731602Z","shell.execute_reply.started":"2022-07-31T20:35:18.550094Z","shell.execute_reply":"2022-07-31T20:35:19.730188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:19.736663Z","iopub.execute_input":"2022-07-31T20:35:19.737176Z","iopub.status.idle":"2022-07-31T20:35:19.778578Z","shell.execute_reply.started":"2022-07-31T20:35:19.737144Z","shell.execute_reply":"2022-07-31T20:35:19.777137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# No Missing Values!","metadata":{}},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:19.780078Z","iopub.execute_input":"2022-07-31T20:35:19.780582Z","iopub.status.idle":"2022-07-31T20:35:19.797622Z","shell.execute_reply.started":"2022-07-31T20:35:19.780551Z","shell.execute_reply":"2022-07-31T20:35:19.796461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=df.drop(columns=['id'],axis=1)\ndataframe=train.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:19.799204Z","iopub.execute_input":"2022-07-31T20:35:19.799524Z","iopub.status.idle":"2022-07-31T20:35:19.823105Z","shell.execute_reply.started":"2022-07-31T20:35:19.799494Z","shell.execute_reply":"2022-07-31T20:35:19.821449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Outliers!\n* There is huge difference between the values at 75th percentile and max value.\n* Power transformer can be used for the dealing with outliers.\n* Robust scaler will be used because it will scale that data.\n* Power transformer is failing to scale the data for the given problem (some values in the features are having high magnitude compared to others).   \n\n","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:19.824898Z","iopub.execute_input":"2022-07-31T20:35:19.826741Z","iopub.status.idle":"2022-07-31T20:35:20.049110Z","shell.execute_reply.started":"2022-07-31T20:35:19.826697Z","shell.execute_reply":"2022-07-31T20:35:20.047594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set style\nsns.set(style='whitegrid')\nfig,axs = plt.subplots(figsize=(20,6))\ng= sns.boxplot(data=train,width = 0.7)\n\n# title\nplt.title('Box plots for features',fontsize=20)\n\n# x label\ng.set_xlabel('Data Features', fontsize=15)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:20.050999Z","iopub.execute_input":"2022-07-31T20:35:20.051491Z","iopub.status.idle":"2022-07-31T20:35:21.252059Z","shell.execute_reply.started":"2022-07-31T20:35:20.051443Z","shell.execute_reply":"2022-07-31T20:35:21.250835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col=list(train.columns)\ncol","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:21.253745Z","iopub.execute_input":"2022-07-31T20:35:21.254116Z","iopub.status.idle":"2022-07-31T20:35:21.262191Z","shell.execute_reply.started":"2022-07-31T20:35:21.254083Z","shell.execute_reply":"2022-07-31T20:35:21.260955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features distribution!\n* Columns f_07 to f_13 are not having mean equal to zero.\n* Remaining columns are having gaussian distribution.","metadata":{}},{"cell_type":"code","source":"# fix figure size\nfig=plt.figure(figsize=(13,15))\n# enumerate will give a tuples assigns index\nfor i,f in enumerate(train.columns):\n    plt.subplot(6,5,i+1)\n    sns.histplot(train[f])\n    plt.xlabel(f)\n\nplt.tight_layout() \n#plt.title(\"Column wise distribution\",fontsize=20) \nplt.show()    ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:21.264366Z","iopub.execute_input":"2022-07-31T20:35:21.264689Z","iopub.status.idle":"2022-07-31T20:35:40.759808Z","shell.execute_reply.started":"2022-07-31T20:35:21.264657Z","shell.execute_reply":"2022-07-31T20:35:40.758854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scaling \n* Values in the each feature are having different magnitudes.\n* Scaling is required, since there are lot of outliers in the original data and using RobustScaler is better option.\n* The reason for using PowerTransformer is because of its better score in leader board.","metadata":{}},{"cell_type":"code","source":"data=train.copy()\ntrain = PowerTransformer().fit_transform(train)\n#train=RobustScaler().fit_transform(transformer)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:40.761187Z","iopub.execute_input":"2022-07-31T20:35:40.762241Z","iopub.status.idle":"2022-07-31T20:35:44.353592Z","shell.execute_reply.started":"2022-07-31T20:35:40.762185Z","shell.execute_reply":"2022-07-31T20:35:44.351951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.DataFrame(train,columns=data.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:44.355674Z","iopub.execute_input":"2022-07-31T20:35:44.356210Z","iopub.status.idle":"2022-07-31T20:35:44.364506Z","shell.execute_reply.started":"2022-07-31T20:35:44.356162Z","shell.execute_reply":"2022-07-31T20:35:44.363029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:44.371944Z","iopub.execute_input":"2022-07-31T20:35:44.372933Z","iopub.status.idle":"2022-07-31T20:35:44.410584Z","shell.execute_reply.started":"2022-07-31T20:35:44.372891Z","shell.execute_reply":"2022-07-31T20:35:44.408734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Guess Clusters\n* [Yellowbrick](http://https://www.scikit-yb.org/en/latest/api/cluster/elbow.html) is used for guessing Number of clusters.","metadata":{}},{"cell_type":"code","source":"model=KMeans(random_state=42)\nvisualizer= KElbowVisualizer(model,k=(2,14))\nvisualizer.fit(train)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:35:44.413238Z","iopub.execute_input":"2022-07-31T20:35:44.413854Z","iopub.status.idle":"2022-07-31T20:37:12.690275Z","shell.execute_reply.started":"2022-07-31T20:35:44.413765Z","shell.execute_reply":"2022-07-31T20:37:12.689078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BayesianGaussianMixture\n* k-means given poor results.\n* might be clusters are not circular.\n* used probabilistic model\n* for k=6 got less score and used k=7.","metadata":{}},{"cell_type":"code","source":"model = BayesianGaussianMixture(n_components=7, covariance_type='full', random_state=1)\npredicted_clusters = model.fit_predict(train)#model will take cluster which is having maximum probability.\npredicted_proba=model.predict_proba(train)#Probabilities for each cluster.\n\n#check more things in below clusters probability section .\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:37:12.692630Z","iopub.execute_input":"2022-07-31T20:37:12.693190Z","iopub.status.idle":"2022-07-31T20:38:25.902494Z","shell.execute_reply.started":"2022-07-31T20:37:12.693140Z","shell.execute_reply":"2022-07-31T20:38:25.900755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['predicted']=predicted_clusters","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:25.904571Z","iopub.execute_input":"2022-07-31T20:38:25.905895Z","iopub.status.idle":"2022-07-31T20:38:25.916168Z","shell.execute_reply.started":"2022-07-31T20:38:25.905823Z","shell.execute_reply":"2022-07-31T20:38:25.913922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:25.919199Z","iopub.execute_input":"2022-07-31T20:38:25.920521Z","iopub.status.idle":"2022-07-31T20:38:25.956952Z","shell.execute_reply.started":"2022-07-31T20:38:25.920462Z","shell.execute_reply":"2022-07-31T20:38:25.955968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature selection!\n* The features f_07 to f_13 and f_22 to f_28 are actually contributing for clusters formation.\n* The other features are almost having mean zero, and column varaience is poor. ","metadata":{}},{"cell_type":"code","source":"#pred_columns=['predicted']\n#features=[col for  col in  train.columns if col not in pred_columns]\ncol# check in outlier section!","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:25.958243Z","iopub.execute_input":"2022-07-31T20:38:25.958737Z","iopub.status.idle":"2022-07-31T20:38:25.965525Z","shell.execute_reply.started":"2022-07-31T20:38:25.958706Z","shell.execute_reply":"2022-07-31T20:38:25.964497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig=plt.figure(figsize=(20,15))\n          \nfor i,f in enumerate(col):\n    plt.subplot(6,5,i+1)\n    sns.kdeplot(data=train,x=f,hue='predicted')\n    plt.xlabel(f)\n\nplt.tight_layout()    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:25.967411Z","iopub.execute_input":"2022-07-31T20:38:25.968231Z","iopub.status.idle":"2022-07-31T20:38:55.362030Z","shell.execute_reply.started":"2022-07-31T20:38:25.968181Z","shell.execute_reply":"2022-07-31T20:38:55.360747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Selection of features mainly depend on the density variation of data for each cluster.\n* Best columns are selected so that model  will be trained on columns having varience in the data.","metadata":{}},{"cell_type":"code","source":"best_cols=['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.363257Z","iopub.execute_input":"2022-07-31T20:38:55.363557Z","iopub.status.idle":"2022-07-31T20:38:55.369460Z","shell.execute_reply.started":"2022-07-31T20:38:55.363527Z","shell.execute_reply":"2022-07-31T20:38:55.368168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clusters probability!! \n* predicted column in the below results shows that first row belongs to cluster 5.\n* The predicted cluster is nothing but which is having maximum probability 0.978086.","metadata":{}},{"cell_type":"code","source":"proba=pd.DataFrame(predicted_proba)\neach_cluter_proba=pd.merge(train,proba,left_index=True,right_index=True)\neach_cluter_proba","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.370706Z","iopub.execute_input":"2022-07-31T20:38:55.371076Z","iopub.status.idle":"2022-07-31T20:38:55.451387Z","shell.execute_reply.started":"2022-07-31T20:38:55.371044Z","shell.execute_reply":"2022-07-31T20:38:55.450023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clustering from Classification!!\n\n> By using clustering approch resulted in score of nearly 0.55. The idea is to use the result obtained from BGM model and feeding the data to a classification algorithm. For a supervised machine learning model, labels are required and here these are nothing but results obtained from BGM model.There are rows which are in to a cluster(label) with a low probability and these are not usefull for training data.Put a optimistic thershold, lets say if we put a thershold probability > 0.85, then might be data contain number of clusters 5,4 and less of 0,1,2,3 then supervised model will be overtrain and always predicts 5,4 for maximum rows. simply like dont give imbalanced data.Putting a thershold of 0.75 is better for this notebook. still overfitting might happen, so we will use startified cross fold valiadation. Here we will used multiple classification models and take out put of aggregate of them by assiging weights. The choice of assiging weights depends on giving weak learner more weights.  \n","metadata":{}},{"cell_type":"markdown","source":"* Selecting maximum probability nothing but using output of BGM model.\n* Selecting only the rows having probability greater that 0.75.","metadata":{}},{"cell_type":"code","source":"Max_proba=proba.max(axis=1)\ntrain=train[best_cols]\ntrain['max_proba']=Max_proba\ntrain['predicted']=predicted_clusters","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.453898Z","iopub.execute_input":"2022-07-31T20:38:55.454394Z","iopub.status.idle":"2022-07-31T20:38:55.479057Z","shell.execute_reply.started":"2022-07-31T20:38:55.454345Z","shell.execute_reply":"2022-07-31T20:38:55.477927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.480484Z","iopub.execute_input":"2022-07-31T20:38:55.481415Z","iopub.status.idle":"2022-07-31T20:38:55.505664Z","shell.execute_reply.started":"2022-07-31T20:38:55.481381Z","shell.execute_reply":"2022-07-31T20:38:55.504719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_full=train.drop(columns=['max_proba','predicted'])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.507163Z","iopub.execute_input":"2022-07-31T20:38:55.507525Z","iopub.status.idle":"2022-07-31T20:38:55.522365Z","shell.execute_reply.started":"2022-07-31T20:38:55.507492Z","shell.execute_reply":"2022-07-31T20:38:55.521126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=train[train['max_proba']>0.75]#selecting rows which are having max_proba > 0.75","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.524235Z","iopub.execute_input":"2022-07-31T20:38:55.524585Z","iopub.status.idle":"2022-07-31T20:38:55.538611Z","shell.execute_reply.started":"2022-07-31T20:38:55.524554Z","shell.execute_reply":"2022-07-31T20:38:55.537439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset=data.drop(columns=['max_proba'],axis=1)\ndataset\n#In dataset there are neraly 7 lakh rows having probability > 0.75 and 3.7 lakh rows have proabability <=0.75 of being in that cluster.  ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.540215Z","iopub.execute_input":"2022-07-31T20:38:55.540970Z","iopub.status.idle":"2022-07-31T20:38:55.576622Z","shell.execute_reply.started":"2022-07-31T20:38:55.540936Z","shell.execute_reply":"2022-07-31T20:38:55.575614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DATASET FOR SUPERVISED MACHINE LEARNING\n* BayesianGMMClassifier used as algorithm.","metadata":{}},{"cell_type":"code","source":"X=dataset.drop(columns='predicted',axis=1)\ny=dataset['predicted']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.578006Z","iopub.execute_input":"2022-07-31T20:38:55.578515Z","iopub.status.idle":"2022-07-31T20:38:55.586655Z","shell.execute_reply.started":"2022-07-31T20:38:55.578483Z","shell.execute_reply":"2022-07-31T20:38:55.585166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = BayesianGMMClassifier(n_components=7, random_state=1, tol=1e-3, covariance_type='full', max_iter=400, n_init=4, init_params='kmeans')\n\nmodel.fit(X[best_cols], y)\nclusters_pred =  model.predict(X_full[best_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:38:55.588836Z","iopub.execute_input":"2022-07-31T20:38:55.590314Z","iopub.status.idle":"2022-07-31T20:43:44.930896Z","shell.execute_reply.started":"2022-07-31T20:38:55.590261Z","shell.execute_reply":"2022-07-31T20:43:44.929850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['predicted']=clusters_pred\nsubmission=df.loc[:,['id','predicted']]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:43:44.933151Z","iopub.execute_input":"2022-07-31T20:43:44.933638Z","iopub.status.idle":"2022-07-31T20:43:44.949341Z","shell.execute_reply.started":"2022-07-31T20:43:44.933590Z","shell.execute_reply":"2022-07-31T20:43:44.947896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T20:43:44.951525Z","iopub.execute_input":"2022-07-31T20:43:44.952089Z","iopub.status.idle":"2022-07-31T20:43:45.116229Z","shell.execute_reply.started":"2022-07-31T20:43:44.952039Z","shell.execute_reply":"2022-07-31T20:43:45.114849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# THANKS","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}