{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T14:30:56.369399Z","iopub.execute_input":"2022-07-12T14:30:56.369802Z","iopub.status.idle":"2022-07-12T14:30:56.379517Z","shell.execute_reply.started":"2022-07-12T14:30:56.369770Z","shell.execute_reply":"2022-07-12T14:30:56.378207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.decomposition import PCA\n#from sklearn.cluster import MiniBatchKMeans\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom scipy import stats\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:27.605202Z","iopub.execute_input":"2022-07-12T15:24:27.606034Z","iopub.status.idle":"2022-07-12T15:24:27.620673Z","shell.execute_reply.started":"2022-07-12T15:24:27.605994Z","shell.execute_reply":"2022-07-12T15:24:27.619517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:31:01.900913Z","iopub.execute_input":"2022-07-12T14:31:01.901280Z","iopub.status.idle":"2022-07-12T14:31:02.785676Z","shell.execute_reply.started":"2022-07-12T14:31:01.901250Z","shell.execute_reply":"2022-07-12T14:31:02.784488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Selection","metadata":{}},{"cell_type":"code","source":"corr_mat = data.corr()\n\ncmap = sns.diverging_palette(240, 20, as_cmap = True)\nsns.heatmap(corr_mat, vmax = 1, vmin = -1, cmap = cmap)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:31:04.820624Z","iopub.execute_input":"2022-07-12T14:31:04.821393Z","iopub.status.idle":"2022-07-12T14:31:05.799386Z","shell.execute_reply.started":"2022-07-12T14:31:04.821353Z","shell.execute_reply":"2022-07-12T14:31:05.797985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most variables don't correlate with others, so these don't provide useful information for clustering","metadata":{}},{"cell_type":"code","source":"selected_columns = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13',\\\n                    'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\n\nselected_data = data[selected_columns]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:31:22.258319Z","iopub.execute_input":"2022-07-12T14:31:22.258691Z","iopub.status.idle":"2022-07-12T14:31:22.268639Z","shell.execute_reply.started":"2022-07-12T14:31:22.258658Z","shell.execute_reply":"2022-07-12T14:31:22.267435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_mat = selected_data.corr()\n\ncmap = sns.diverging_palette(240, 20, as_cmap = True)\nsns.heatmap(corr_mat, vmax = 1, vmin = -1, cmap = cmap)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:31:27.264242Z","iopub.execute_input":"2022-07-12T14:31:27.264617Z","iopub.status.idle":"2022-07-12T14:31:27.709783Z","shell.execute_reply.started":"2022-07-12T14:31:27.264588Z","shell.execute_reply":"2022-07-12T14:31:27.708629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These are the most useful features for clustering.","metadata":{}},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"transformer = PowerTransformer()\n\nscaled_data = pd.DataFrame(transformer.fit_transform(selected_data), columns = selected_columns)\nscaled_data.boxplot(column = selected_columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:56:08.850975Z","iopub.execute_input":"2022-07-12T14:56:08.851413Z","iopub.status.idle":"2022-07-12T14:56:10.649396Z","shell.execute_reply.started":"2022-07-12T14:56:08.851379Z","shell.execute_reply":"2022-07-12T14:56:10.648071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are too many outliers. I will remove some of them based on their Z-score. ","metadata":{}},{"cell_type":"code","source":"no_outliers_data = scaled_data[(np.abs(stats.zscore(scaled_data)) < 3).all(axis=1)]\nno_outliers_data.boxplot(column = selected_columns)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K Means Clustering","metadata":{}},{"cell_type":"code","source":"model = KMeans()\n\nvisualizer = KElbowVisualizer(model, k=(2,21), timings= True)\nvisualizer.fit(scaled_data)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:08:20.777874Z","iopub.execute_input":"2022-07-12T15:08:20.778313Z","iopub.status.idle":"2022-07-12T15:10:59.747424Z","shell.execute_reply.started":"2022-07-12T15:08:20.778276Z","shell.execute_reply":"2022-07-12T15:10:59.746175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Elbow method suggest 8 clusters, however the curve is quite smooth so this might not be a good estimate.\nLet's try fitting on cleaned data ","metadata":{}},{"cell_type":"code","source":"visualizer = KElbowVisualizer(model, k=(2,21), timings= True)\nvisualizer.fit(no_outliers_data)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:56:14.488229Z","iopub.execute_input":"2022-07-12T14:56:14.488597Z","iopub.status.idle":"2022-07-12T14:56:17.088024Z","shell.execute_reply.started":"2022-07-12T14:56:14.488569Z","shell.execute_reply":"2022-07-12T14:56:17.086901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Distortion score is lower as expected (less data points). \nThe curve is still quite smooth. Visually, I believe anything from 6 to 8 could be good. Therefore I will choose 7 clusters.","metadata":{}},{"cell_type":"markdown","source":"### Fit on data with no outliers","metadata":{}},{"cell_type":"code","source":"model = KMeans(n_clusters = 7)\nmodel.fit(no_outliers_data)\n\npredictions = model.predict(scaled_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:01.661524Z","iopub.execute_input":"2022-07-12T15:24:01.662687Z","iopub.status.idle":"2022-07-12T15:24:06.501296Z","shell.execute_reply.started":"2022-07-12T15:24:01.662644Z","shell.execute_reply":"2022-07-12T15:24:06.499377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\nsub['Predicted'] = predictions\nsub.to_csv('k_means_cleaned_submissions.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:10.662256Z","iopub.execute_input":"2022-07-12T15:24:10.663104Z","iopub.status.idle":"2022-07-12T15:24:10.860583Z","shell.execute_reply.started":"2022-07-12T15:24:10.663049Z","shell.execute_reply":"2022-07-12T15:24:10.859311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fit on whole dataset","metadata":{}},{"cell_type":"code","source":"model.fit(scaled_data)\n\npredictions = model.predict(scaled_data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Predicted'] = predictions\nsub.to_csv('k_means_with_outliers_submissions.csv', index = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bayesian Gaussian Mixture","metadata":{}},{"cell_type":"markdown","source":"### Fit on cleaned data","metadata":{}},{"cell_type":"code","source":"model = BayesianGaussianMixture(n_components = 7, covariance_type = 'full')\nmodel.fit(no_outliers_data)\n\npredictions = model.predict(scaled_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:41.883079Z","iopub.execute_input":"2022-07-12T15:24:41.883457Z","iopub.status.idle":"2022-07-12T15:25:15.323293Z","shell.execute_reply.started":"2022-07-12T15:24:41.883427Z","shell.execute_reply":"2022-07-12T15:25:15.321953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Predicted'] = predictions\nsub.to_csv('BGM_cleaned_submissions.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:25:33.625817Z","iopub.execute_input":"2022-07-12T15:25:33.626271Z","iopub.status.idle":"2022-07-12T15:25:33.790402Z","shell.execute_reply.started":"2022-07-12T15:25:33.626238Z","shell.execute_reply":"2022-07-12T15:25:33.789116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fit on whole dataset","metadata":{}},{"cell_type":"code","source":"model.fit(scaled_data)\n\npredictions = model.predict(scaled_data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Predicted'] = predictions\nsub.to_csv('BGM_with_outliers_submissions.csv', index = False)","metadata":{},"execution_count":null,"outputs":[]}]}