{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<!-- P.S. this kernel only specifies which features are to be selected for modelling part you can check my [other kernel](https://www.kaggle.com/code/ashaykatrojwar/eda-pca-bayesian-gaussian-mixture). -->","metadata":{}},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.preprocessing import RobustScaler,PowerTransformer\nfrom sklearn.mixture import BayesianGaussianMixture,GaussianMixture\nfrom sklearn.metrics import confusion_matrix, silhouette_score, calinski_harabasz_score, davies_bouldin_score,balanced_accuracy_score, roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nimport lightgbm as lgb\nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:23:59.352011Z","iopub.execute_input":"2022-07-18T17:23:59.352534Z","iopub.status.idle":"2022-07-18T17:23:59.399860Z","shell.execute_reply.started":"2022-07-18T17:23:59.352497Z","shell.execute_reply":"2022-07-18T17:23:59.398056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading Data\n*  Dropping id column.\n*  Features f_07 to f_13 are `integer features`.\n*  Other features are `continuous features`. ","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ndf=df.drop('id',axis=1)\nss=pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:51:25.779514Z","iopub.execute_input":"2022-07-18T16:51:25.780122Z","iopub.status.idle":"2022-07-18T16:51:27.535784Z","shell.execute_reply.started":"2022-07-18T16:51:25.780073Z","shell.execute_reply":"2022-07-18T16:51:27.534673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature distribution EDA\n> The important takeaways from the distributions are:\n   * All the continuous features are `normally distributed`.\n   * Data needs to Transformed by using transformer such as `Power Transformer`.","metadata":{}},{"cell_type":"code","source":"fig=plt.figure(figsize=(15,14))\nfor i, f in enumerate(df.columns):\n    # New subplot\n    plt.style.use('ggplot')\n    plt.subplot(6,5,i+1)\n    sns.histplot(x=df[f])\n    plt.title(f'Feature: {f}')\n    plt.xlabel('')\n    \nfig.suptitle('Feature distributions',  size=20)\nfig.tight_layout()  \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-17T09:38:17.945343Z","iopub.execute_input":"2022-07-17T09:38:17.945734Z","iopub.status.idle":"2022-07-17T09:38:31.655932Z","shell.execute_reply.started":"2022-07-17T09:38:17.945701Z","shell.execute_reply":"2022-07-17T09:38:31.655041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing\n>  Preprocessing data using `Power Transformer`","metadata":{}},{"cell_type":"code","source":"r=PowerTransformer()\nX=r.fit_transform(df)\ndf=pd.DataFrame(X,columns=df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:51:31.067336Z","iopub.execute_input":"2022-07-18T16:51:31.068015Z","iopub.status.idle":"2022-07-18T16:51:35.038412Z","shell.execute_reply.started":"2022-07-18T16:51:31.067956Z","shell.execute_reply":"2022-07-18T16:51:35.036749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Outliers by Inter Quartile Range Method\nIQR tells us the variation in the data set.Any value, which is beyond the range of `-1.5 IQR to 1.5 IQR` treated as outliers\n\n> * Q1 represents the `1st quartile/25th percentile` of the data.\n> * Q2 represents the `2nd quartile/median/50th percentile` of the data.\n> * Q3 represents the `3rd quartile/75th percentile` of the data.\n> * `(Q1–1.5*IQR)` represent the smallest value in the data set and `(Q3+1.5*IQR)` represnt the largest value in the data set.\n\nBy the below plot it can be observed that feature `f_26` has ~1600 outliers which is highest compared to other features but it's only `1.63%`of entire data of that feature which is not that significant. ","metadata":{}},{"cell_type":"code","source":"out=[]\ndef iqr_outliers(df,ft):\n    q1 = df[ft].quantile(0.25)\n    q3 = df[ft].quantile(0.75)\n    iqr = q3-q1\n    Lower_tail = q1 - 1.5 * iqr\n    Upper_tail = q3 + 1.5 * iqr\n    c=0\n    for i in range(len(df[ft])):\n        if df[ft][i] > Upper_tail or df[ft][i] < Lower_tail:\n            c+=1\n    return c\nod={ f:iqr_outliers(df,f) for f in df.columns }\n\n# Plotting Outliers\nplt.style.use('ggplot')\nplt.figure(figsize=(16,8))\nplt.bar(x=od.keys(),height=od.values())\nplt.xlabel(\"Features\")\nplt.ylabel(\"Outliers\")\nplt.title('Outliers')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-17T09:38:46.165298Z","iopub.execute_input":"2022-07-17T09:38:46.165960Z","iopub.status.idle":"2022-07-17T09:39:30.992466Z","shell.execute_reply.started":"2022-07-17T09:38:46.165925Z","shell.execute_reply":"2022-07-17T09:39:30.991591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODEL\nYou can check my [other kernel](https://www.kaggle.com/code/ashaykatrojwar/eda-pca-bayesian-gaussian-mixture) which has in depth analysis of how to choose `number of clusters` for our model and `Dimensionality reduction using PCA`.\n","metadata":{}},{"cell_type":"code","source":"BGM = BayesianGaussianMixture(n_components=7,covariance_type='full',random_state=1,n_init=5,tol=0.001,max_iter=400)\n# fit model and predict clusters\npreds = BGM.fit_predict(X)\npp=BGM.predict_proba(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T11:48:26.439529Z","iopub.execute_input":"2022-07-18T11:48:26.440303Z","iopub.status.idle":"2022-07-18T11:52:43.293482Z","shell.execute_reply.started":"2022-07-18T11:48:26.440252Z","shell.execute_reply":"2022-07-18T11:52:43.291616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Useful Features\n> From the plot of cluster means plotted from \"BayesianGaussianMixture\" model it is clear that features from `f_0 to f_06` and features from `f_14 to f_21` have mean zero for all clusters which is not distinct so they are of not much of use so we can drop these features. ","metadata":{}},{"cell_type":"markdown","source":"> `Useful features are` ****['f_07', 'f_08',\n>        'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26',\n>        'f_27', 'f_28']****","metadata":{}},{"cell_type":"code","source":"plt.style.use('ggplot')\nplt.figure(figsize=(15,6))\nfor i in range(BGM.means_.shape[0]):\n    plt.scatter(np.arange(X.shape[1]), BGM.means_[i])\nplt.xticks(ticks=np.arange(X.shape[1]), labels=df.columns)\nplt.xlabel('Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T17:15:56.294588Z","iopub.execute_input":"2022-07-16T17:15:56.295009Z","iopub.status.idle":"2022-07-16T17:15:56.697840Z","shell.execute_reply.started":"2022-07-16T17:15:56.294964Z","shell.execute_reply":"2022-07-16T17:15:56.696228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats=['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-18T11:53:51.030805Z","iopub.execute_input":"2022-07-18T11:53:51.031469Z","iopub.status.idle":"2022-07-18T11:53:51.037761Z","shell.execute_reply.started":"2022-07-18T11:53:51.031382Z","shell.execute_reply":"2022-07-18T11:53:51.036519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classifier\n> After creating a data from clusters predicted by Bayessian Gaussian Mixture model, a `classifier` is trained on same data but only on data points having cluster prediction probability more than `80%`.\n> Code below is inspired form [@aldparis](https://www.kaggle.com/adaubas)","metadata":{}},{"cell_type":"code","source":"df[[f'predict_proba_{i}' for i in range(7)]]=pp # creating new dataframe columns of probabilites \ndf['predict_proba']=np.max(pp,axis=1)\ndf['predict']=np.argmax(pp,axis=1)\n    \ntrn_indx=np.array([])\nfor n in range(7):\n    # using only those index whoose predicted probablity > 80\n    indx=df[(df.predict==n) & (df.predict_proba > 0.75)].index \n    trn_indx = np.concatenate((trn_indx, indx))\n    \nX = df.loc[trn_indx][feats]\ny = df.loc[trn_indx]['predict']","metadata":{"execution":{"iopub.status.busy":"2022-07-18T11:53:54.941336Z","iopub.execute_input":"2022-07-18T11:53:54.941762Z","iopub.status.idle":"2022-07-18T11:53:55.077780Z","shell.execute_reply.started":"2022-07-18T11:53:54.941727Z","shell.execute_reply":"2022-07-18T11:53:55.076653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_lgb = {'learning_rate': 0.07,'objective': 'multiclass','boosting': 'gbdt','verbosity': -1,'n_jobs': -1, 'num_classes':7} \n\nlgbm_predict_proba = 0  \nctb_predict_proba = 0\nqda_predict_proba = 0\nsvc_predict_proba = 0\n\nclassif_scores = []\n\ngkf = StratifiedKFold(10, shuffle=True, random_state = 1)\nfor fold, (trn, val) in enumerate(gkf.split(X,y)):  \n    \n    X_train, y_train = X.iloc[trn], y.iloc[trn]\n    X_valid, y_valid = X.iloc[val], y.iloc[val]\n\n    X_trn = lgb.Dataset(X_train, y_train)\n    X_val = lgb.Dataset(X_valid, y_valid)\n    \n    model = lgb.train(params = params_lgb, \n                train_set = X_trn, valid_sets =  X_val, \n                num_boost_round = 5000, \n                callbacks = [ lgb.early_stopping(stopping_rounds=100, verbose=True), lgb.log_evaluation(period=200)])  \n    \n    y_pred_proba = model.predict(X.iloc[val])\n    y_pred = np.argmax(y_pred_proba, axis=1)\n    \n    s = (balanced_accuracy_score(y.iloc[val], y_pred),\n        roc_auc_score(y.iloc[val], y_pred_proba, average=\"weighted\", multi_class=\"ovo\"))\n    print(f\"LightGBM   AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    lgbm_predict_proba += model.predict(df[feats]) / 10\n    \n    \npd.DataFrame(classif_scores, columns = [\"balanced_accuracy_score\", \"roc_auc_score\"]).mean(0)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-18T11:54:39.434172Z","iopub.execute_input":"2022-07-18T11:54:39.434633Z","iopub.status.idle":"2022-07-18T11:54:47.333465Z","shell.execute_reply.started":"2022-07-18T11:54:39.434594Z","shell.execute_reply":"2022-07-18T11:54:47.331776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix of Classifier and actual Gaussian Mixture Model","metadata":{}},{"cell_type":"code","source":"# confusion_matrix(df.predict, np.argmax(lgbm_predict_proba,axis=1))\nplt.figure(figsize=(20,10))\nsns.heatmap(confusion_matrix(df.predict, np.argmax(lgbm_predict_proba,axis=1)), annot=True)\nplt.ylabel(\"Clusters from BGM\")\nplt.xlabel('Classification by classifier')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T12:39:57.644568Z","iopub.execute_input":"2022-07-18T12:39:57.645561Z","iopub.status.idle":"2022-07-18T12:39:57.744963Z","shell.execute_reply.started":"2022-07-18T12:39:57.645422Z","shell.execute_reply":"2022-07-18T12:39:57.743170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss.Predicted=np.argmax(lgbm_predict_proba,axis=1)\nss.to_csv(\"submission.csv\",index=False)\n\nplt.figure(figsize=(15,5))\nsns.countplot(x=ss.Predicted)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T17:44:18.173582Z","iopub.execute_input":"2022-07-16T17:44:18.173950Z","iopub.status.idle":"2022-07-16T17:44:18.343555Z","shell.execute_reply.started":"2022-07-16T17:44:18.173921Z","shell.execute_reply":"2022-07-16T17:44:18.342330Z"},"trusted":true},"execution_count":null,"outputs":[]}]}