{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T16:38:36.465698Z","iopub.execute_input":"2022-07-28T16:38:36.466216Z","iopub.status.idle":"2022-07-28T16:38:37.145314Z","shell.execute_reply.started":"2022-07-28T16:38:36.466107Z","shell.execute_reply":"2022-07-28T16:38:37.143985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-lego","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-28T16:38:37.147260Z","iopub.execute_input":"2022-07-28T16:38:37.147589Z","iopub.status.idle":"2022-07-28T16:38:51.408953Z","shell.execute_reply.started":"2022-07-28T16:38:37.147561Z","shell.execute_reply":"2022-07-28T16:38:51.407258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submis = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv', index_col='Id')\ndata = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:51.410994Z","iopub.execute_input":"2022-07-28T16:38:51.411418Z","iopub.status.idle":"2022-07-28T16:38:52.931509Z","shell.execute_reply.started":"2022-07-28T16:38:51.411382Z","shell.execute_reply":"2022-07-28T16:38:52.930168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:52.935180Z","iopub.execute_input":"2022-07-28T16:38:52.936264Z","iopub.status.idle":"2022-07-28T16:38:52.974770Z","shell.execute_reply.started":"2022-07-28T16:38:52.936210Z","shell.execute_reply":"2022-07-28T16:38:52.973636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submis.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:52.976262Z","iopub.execute_input":"2022-07-28T16:38:52.976597Z","iopub.status.idle":"2022-07-28T16:38:52.987171Z","shell.execute_reply.started":"2022-07-28T16:38:52.976569Z","shell.execute_reply":"2022-07-28T16:38:52.985819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:52.989177Z","iopub.execute_input":"2022-07-28T16:38:52.989989Z","iopub.status.idle":"2022-07-28T16:38:53.020342Z","shell.execute_reply.started":"2022-07-28T16:38:52.989953Z","shell.execute_reply":"2022-07-28T16:38:53.018825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discrete_cols = []\nfor col in data.columns:\n    if 1.*data[col].nunique()/data[col].count() < 0.05: discrete_cols.append(col)\n        \ndiscrete_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:53.021924Z","iopub.execute_input":"2022-07-28T16:38:53.022260Z","iopub.status.idle":"2022-07-28T16:38:53.149793Z","shell.execute_reply.started":"2022-07-28T16:38:53.022231Z","shell.execute_reply":"2022-07-28T16:38:53.148525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[discrete_cols].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:53.151283Z","iopub.execute_input":"2022-07-28T16:38:53.155203Z","iopub.status.idle":"2022-07-28T16:38:53.172139Z","shell.execute_reply.started":"2022-07-28T16:38:53.155151Z","shell.execute_reply":"2022-07-28T16:38:53.170998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cont_col = data.select_dtypes('float64').columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:53.173828Z","iopub.execute_input":"2022-07-28T16:38:53.174975Z","iopub.status.idle":"2022-07-28T16:38:53.185110Z","shell.execute_reply.started":"2022-07-28T16:38:53.174928Z","shell.execute_reply":"2022-07-28T16:38:53.184070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Features f_07 - f_13 are discrete and consist of values b/w 0-45.\n\nFeatures f_00 - f_06, f_19 - f_26 and f_28 consist of continuous, numeric values.","metadata":{}},{"cell_type":"code","source":"len(cont_col)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:53.189187Z","iopub.execute_input":"2022-07-28T16:38:53.189544Z","iopub.status.idle":"2022-07-28T16:38:53.198678Z","shell.execute_reply.started":"2022-07-28T16:38:53.189515Z","shell.execute_reply":"2022-07-28T16:38:53.197454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(5, 5, figsize=(16, 16))\n\nfor ax, col in zip(axs.flatten(), cont_col):\n    sns.histplot(data, x=col, ax=ax, kde=True)\n\nfor ax in axs.flat[len(cont_col):]:\n    ax.remove()\nplt.suptitle('Histograms of the float features in Training Data', y=0.93, fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:38:53.200339Z","iopub.execute_input":"2022-07-28T16:38:53.201242Z","iopub.status.idle":"2022-07-28T16:39:12.341943Z","shell.execute_reply.started":"2022-07-28T16:38:53.201201Z","shell.execute_reply":"2022-07-28T16:39:12.340729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(4, 2, figsize=(20, 20))\n\nfor ax, col in zip(axs.flatten(), discrete_cols):\n    sns.countplot(x=col, data=data, ax=ax)\n\nfig.delaxes(axs[3][1])    \n\nplt.suptitle('Histograms of the discrete features Train data', y=0.93, fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:12.343698Z","iopub.execute_input":"2022-07-28T16:39:12.344158Z","iopub.status.idle":"2022-07-28T16:39:14.568319Z","shell.execute_reply.started":"2022-07-28T16:39:12.344114Z","shell.execute_reply":"2022-07-28T16:39:14.566948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All the continuos features are normally distribute with mean 0 and std around 5. But most of the categorical features are skewed to the right. So we need to transform them for better performance during clustering.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\n\ntrans = PowerTransformer()\nX=trans.fit_transform(data)\ndata=pd.DataFrame(X,columns=data.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:14.569688Z","iopub.execute_input":"2022-07-28T16:39:14.570046Z","iopub.status.idle":"2022-07-28T16:39:18.476423Z","shell.execute_reply.started":"2022-07-28T16:39:14.570015Z","shell.execute_reply":"2022-07-28T16:39:18.474970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:18.478013Z","iopub.execute_input":"2022-07-28T16:39:18.478989Z","iopub.status.idle":"2022-07-28T16:39:18.706000Z","shell.execute_reply.started":"2022-07-28T16:39:18.478948Z","shell.execute_reply":"2022-07-28T16:39:18.705126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's first try kmeans with random number of clusters.","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:18.707169Z","iopub.execute_input":"2022-07-28T16:39:18.708150Z","iopub.status.idle":"2022-07-28T16:39:18.992867Z","shell.execute_reply.started":"2022-07-28T16:39:18.708112Z","shell.execute_reply":"2022-07-28T16:39:18.991490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = KMeans(n_clusters=7, random_state=42)\nmodel.fit(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:18.994697Z","iopub.execute_input":"2022-07-28T16:39:18.995939Z","iopub.status.idle":"2022-07-28T16:39:24.788708Z","shell.execute_reply.started":"2022-07-28T16:39:18.995899Z","shell.execute_reply":"2022-07-28T16:39:24.787857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.rcParams[\"figure.figsize\"] = (18,9)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:24.790001Z","iopub.execute_input":"2022-07-28T16:39:24.790950Z","iopub.status.idle":"2022-07-28T16:39:24.795939Z","shell.execute_reply.started":"2022-07-28T16:39:24.790918Z","shell.execute_reply":"2022-07-28T16:39:24.794592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme()\n\nfor i,means in enumerate(model.cluster_centers_):\n    plt.scatter(data.columns, means, label='Cluster '+str(i+1))\n    \nplt.legend()\nplt.xlabel('Features'),plt.ylabel('Cluster Centers')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:24.797557Z","iopub.execute_input":"2022-07-28T16:39:24.797970Z","iopub.status.idle":"2022-07-28T16:39:25.394487Z","shell.execute_reply.started":"2022-07-28T16:39:24.797936Z","shell.execute_reply":"2022-07-28T16:39:25.393332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For features `f_00-f_06` and `f_14-f_21` have cluster mean close to 0 for clusters which means all the clusters are overlapping. So importance of these features for clustering is very very low. They are acting as noise, so it will be better to just drop them.","metadata":{}},{"cell_type":"code","source":"features = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23',\n       'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\nclean_data = data[features]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:06:31.577567Z","iopub.execute_input":"2022-07-28T17:06:31.578017Z","iopub.status.idle":"2022-07-28T17:06:31.591354Z","shell.execute_reply.started":"2022-07-28T17:06:31.577981Z","shell.execute_reply":"2022-07-28T17:06:31.590171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's try to find the right number of clusters. I will be using Silohuette score and Bayesian information criterion (BIC) to find it. I have followed example from this book.\nhttps://jakevdp.github.io/PythonDataScienceHandbook/05.12-gaussian-mixtures.html","metadata":{}},{"cell_type":"markdown","source":"Before that I'm going to reduce number of feature that will make speed make these cluster more inpretable but minimizes information loss.","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture as BGM,GaussianMixture as GMM\nfrom tqdm import tqdm\n\nn_components = np.arange(1, 21)\nmodels = [GMM(n, covariance_type='full', random_state=0).fit(clean_data)\n          for n in tqdm(n_components)]\n\nplt.plot(n_components, [m.bic(clean_data) for m in models], label='BIC')\nplt.plot(n_components, [m.aic(clean_data) for m in models], label='AIC')\nplt.legend(loc='best')\nplt.xlabel('n_components');","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:39:25.409983Z","iopub.execute_input":"2022-07-28T16:39:25.410530Z","iopub.status.idle":"2022-07-28T16:48:08.193328Z","shell.execute_reply.started":"2022-07-28T16:39:25.410492Z","shell.execute_reply":"2022-07-28T16:48:08.191860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cluster=10","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:48:08.195159Z","iopub.execute_input":"2022-07-28T16:48:08.195499Z","iopub.status.idle":"2022-07-28T16:48:08.201119Z","shell.execute_reply.started":"2022-07-28T16:48:08.195470Z","shell.execute_reply":"2022-07-28T16:48:08.199823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm = BGM(n_components=cluster,\n    max_iter=300,\n    n_init=10,\n    random_state=2, verbose = 2,\n    verbose_interval=100,\n)\npreds = bgm.fit_predict(clean_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:48:08.202961Z","iopub.execute_input":"2022-07-28T16:48:08.203464Z","iopub.status.idle":"2022-07-28T17:00:57.937194Z","shell.execute_reply.started":"2022-07-28T16:48:08.203421Z","shell.execute_reply":"2022-07-28T17:00:57.935523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_data['proba'] = np.max(bgm.predict_proba(clean_data), axis=1)\nclean_data['Cluster'] = preds","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:06:37.893411Z","iopub.execute_input":"2022-07-28T17:06:37.893912Z","iopub.status.idle":"2022-07-28T17:06:38.349549Z","shell.execute_reply.started":"2022-07-28T17:06:37.893873Z","shell.execute_reply":"2022-07-28T17:06:38.346606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_data","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:07:00.157795Z","iopub.execute_input":"2022-07-28T17:07:00.158296Z","iopub.status.idle":"2022-07-28T17:07:00.188805Z","shell.execute_reply.started":"2022-07-28T17:07:00.158256Z","shell.execute_reply":"2022-07-28T17:07:00.187558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = clean_data[clean_data['proba']>0.75][features]\ny = clean_data[clean_data['proba']>0.75]['Cluster']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:07:22.935695Z","iopub.execute_input":"2022-07-28T17:07:22.936191Z","iopub.status.idle":"2022-07-28T17:07:22.957634Z","shell.execute_reply.started":"2022-07-28T17:07:22.936153Z","shell.execute_reply":"2022-07-28T17:07:22.956653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape[0]/clean_data.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:07:29.067670Z","iopub.execute_input":"2022-07-28T17:07:29.068174Z","iopub.status.idle":"2022-07-28T17:07:29.076112Z","shell.execute_reply.started":"2022-07-28T17:07:29.068132Z","shell.execute_reply":"2022-07-28T17:07:29.074849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklego.mixture import BayesianGMMClassifier\n\nbgm = BayesianGMMClassifier(\n    n_components=10,\n    random_state=42,\n    covariance_type=\"full\",\n    max_iter=500,\n    n_init=7,\n    init_params=\"kmeans\",\n)\n\nbgm.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:07:41.808406Z","iopub.execute_input":"2022-07-28T17:07:41.808965Z","iopub.status.idle":"2022-07-28T17:24:23.983472Z","shell.execute_reply.started":"2022-07-28T17:07:41.808919Z","shell.execute_reply":"2022-07-28T17:24:23.981864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\ny_pred = bgm.predict(X)\naccuracy_score(y, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:24:23.991830Z","iopub.execute_input":"2022-07-28T17:24:23.992767Z","iopub.status.idle":"2022-07-28T17:24:26.139690Z","shell.execute_reply.started":"2022-07-28T17:24:23.992699Z","shell.execute_reply":"2022-07-28T17:24:26.138571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submis[\"Predicted\"] = bgm.predict(clean_data[features])\nsubmis.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:56:40.276684Z","iopub.execute_input":"2022-07-28T17:56:40.277164Z","iopub.status.idle":"2022-07-28T17:56:43.787132Z","shell.execute_reply.started":"2022-07-28T17:56:40.277119Z","shell.execute_reply":"2022-07-28T17:56:43.785882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submis","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:56:44.093058Z","iopub.execute_input":"2022-07-28T17:56:44.094148Z","iopub.status.idle":"2022-07-28T17:56:44.106357Z","shell.execute_reply.started":"2022-07-28T17:56:44.094100Z","shell.execute_reply":"2022-07-28T17:56:44.105277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}