{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"id":"2ZjkUEnxvmDd","execution":{"iopub.status.busy":"2022-07-27T04:04:25.599826Z","iopub.execute_input":"2022-07-27T04:04:25.600270Z","iopub.status.idle":"2022-07-27T04:04:26.590132Z","shell.execute_reply.started":"2022-07-27T04:04:25.600177Z","shell.execute_reply":"2022-07-27T04:04:26.589361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some EDA","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\nids = data.id\ndata.drop('id', axis = 1, inplace = True)","metadata":{"id":"BR3gaS8IvrtC"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"id":"VlN0u3OWwVH9","outputId":"c0d44b15-aba8-4679-b369-3bccf91418b4"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe().T","metadata":{"id":"RtpEbO6hwG_b","outputId":"061e9d82-04f8-4b47-82e8-a444015d2fa4"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 15))\nfor idx, col in enumerate(data):\n  plt.subplot(6, 5, idx+1)\n  sns.histplot(data[col])\nplt.show()","metadata":{"id":"3DHowjKm518_","outputId":"18677da4-cb63-4bc8-bd95-54e0a84e07df"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer \nt1 = PowerTransformer()\ndata = t1.fit_transform(data)","metadata":{"id":"8fqMJSqjvx6R"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Picking number of clusters","metadata":{}},{"cell_type":"markdown","source":"> Using the  elbow method from the **inertia** and **silhouette_score** \nwe can see from the figure that the optimal number of clusters are either 6 or 7","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\ninertias = [np.inf]\nshlts = [-np.inf]\nfor i in range(2, 10+1):\n  km = KMeans(n_clusters=i, max_iter = 100, n_init = 3)\n  km.fit(data)\n  inertias.append(km.inertia_)\n  shlts.append(silhouette_score(data, km.labels_))","metadata":{"id":"D6ZI4vjUv1xs"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nplt.scatter(range(2, 12), inertias)\nplt.show()","metadata":{"id":"VEFLr8Lbxclv","outputId":"6f069dae-d0d2-418c-c2c3-e9cac2e2b284"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nplt.scatter(range(2, 12), shlts)\nplt.show()","metadata":{"id":"1O5WeFaA5M97","outputId":"261f7d48-b698-498e-d7bd-9fb7a5b8ca3c"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We are initially run a GMM with 7 clusters ","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import GaussianMixture\ngmm1 = GaussianMixture(n_components = 7, max_iter = 1000,\n                               covariance_type='full')\ngmm1.fit(data)","metadata":{"id":"Wx4WCoWh5RJ8","outputId":"72fbd50b-6ac3-475b-ff52-b3fbcb7cdf75"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = gmm1.predict(data)\ninit_predictions = pd.DataFrame(np.c_[ids, pred])\ninit_predictions.rename(columns={0: 'Id', 1: 'Predicted'}, inplace = True)","metadata":{"id":"t2-jZFkeAcYG"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15, 10))\nsns.histplot(init_predictions.Predicted)\nplt.show()","metadata":{"id":"4k6fIGbPAlFV","outputId":"7c827d78-9d5a-4aa1-aafb-443133354d21"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we can pick the most useful instances that the model predict with the heighest probability","metadata":{}},{"cell_type":"code","source":"y_pred_prop = gmm1.predict_proba(data)\ny_pred_certian = y_pred_prop[y_pred_prop > 0.9] ","metadata":{"id":"CT0MF8nuApCr"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_certian.shape[0] / y_pred_prop.shape[0]","metadata":{"id":"TTn-5sbdArKL","outputId":"33a8d8c3-521d-4c52-a78f-b069fc710691"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idxs = np.select((np.sum(y_pred_prop > 0.8, axis = 1) > 0).reshape(1, -1),\n          np.arange(98000).reshape(1, -1), default = None)\nidxs = idxs[~np.isnan(idxs.astype(np.float32))].astype(np.int64)","metadata":{"id":"bt5KISHcAt3x"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_certain = data[idxs]\ny_certain = gmm1.predict(X_certain)","metadata":{"id":"bexLzzcyAwbM"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now run onther GMM on the clean part of the data which allow us to sample a new instances from the distribution","metadata":{}},{"cell_type":"code","source":"gmm2 = GaussianMixture(n_components = 7, max_iter = 1000,\n                               covariance_type='full')\ngmm2.fit(X_certain)","metadata":{"id":"6lTCC5hKBPEG","outputId":"0857b2a7-9ef9-4594-9a2e-dbfb5adfd81d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = gmm2.predict(data)\ninit_predictions_2 = pd.DataFrame(np.c_[ids, pred])\ninit_predictions_2.rename(columns={0: 'Id', 1: 'Predicted'}, inplace = True)","metadata":{"id":"m1Ui8bayk3kE"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_new, y_new = gmm2.sample(100000)","metadata":{"id":"tgwr4eqtBlBw"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_full = np.concatenate([X_new, X_certain], axis = 0)\ny_full = np.concatenate([y_new, y_certain], axis = 0)","metadata":{"id":"o13pqnBzBrEW"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X_full, y_full, test_size = 0.1)","metadata":{"id":"v0Dtq8glBrmQ"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nlgb1 = lgb.LGBMClassifier(n_estimators=10000, max_depth = -1)\nlgb1.fit(X_train, y_train, eval_set = (X_test, y_test), early_stopping_rounds=20)","metadata":{"id":"LZIPGd-RBtqg","outputId":"e6764a68-3f12-482b-a9c3-9440fddea4d3"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = lgb1.predict(data)\nlgb_predictions = pd.DataFrame(np.c_[ids, pred])\nlgb_predictions.rename(columns={0: 'Id', 1: 'Predicted'}, inplace = True)","metadata":{"id":"rhXIqPTsBv8-"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15, 10))\nsns.histplot(lgb_predictions.Predicted)\nplt.show()","metadata":{"id":"HGzvQ2Y4B34g","outputId":"1c5a7146-02cf-4d5f-8317-bd5cf1ecfcb9"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\naccuracy_score(lgb1.predict(X_test), y_test)","metadata":{"id":"Ci6e4jqoDzhT","outputId":"3b6dd7ff-02f9-4124-c1a2-f9a44cdcff85"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(lgb1,figsize=(15, 10))\nplt.show()","metadata":{"id":"K0HcMwFoD34E","outputId":"05606df2-34be-4c64-c807-c4f0a3c97cef"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Select only the features the most important","metadata":{}},{"cell_type":"code","source":"useless_columns = [i for i in range(lgb1.feature_importances_.shape[0])\n       if lgb1.feature_importances_[i] < 7000]","metadata":{"id":"W42dI-87D7tS"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.delete(data, useless_columns, axis = 1)","metadata":{"id":"vOzPpAyZEHEG"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = np.delete(X_train, useless_columns, axis = 1)","metadata":{"id":"dwKBbxckEjcZ"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = np.delete(X_test, useless_columns, axis = 1)","metadata":{"id":"VW1c3K53EwZS"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture\nbgm = BayesianGaussianMixture(n_components=7, max_iter = 1000)\nbgm.fit(data)","metadata":{"id":"InIuaKLLETly","outputId":"7364e5ea-e330-48d3-b053-513e5007a77f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = bgm.predict(data)\nbgmm_predictions = pd.DataFrame(np.c_[ids, pred])\nbgmm_predictions.rename(columns={0: 'Id', 1: 'Predicted'}, inplace = True)\nbgmm_predictions.to_csv('sub.csv', index = False)","metadata":{"id":"wOd4xItKhYBb"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15, 10))\nsns.histplot(bgmm_predictions.Predicted)\nplt.show()","metadata":{"id":"anh-mF2shda_","outputId":"72990abf-3bf4-49bd-c04e-22470d3b8d3a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"HHf_L40wMd0G"},"execution_count":null,"outputs":[]}]}