{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Bruteforcing the optimal Clustering method\n\nSince we don't have information on the number of clusters (nor the best algorithm / approach we should investigate). \n\nWe might just have to try..\n \n#### **On this notebook we don't do anything fancy..**\n\nWe just try to find the number of clusters (also clustering algorithm / other hyperparameters) by trial and error (and LB submission).\n\n- Nothing more.\n- Nothing less. \n\nI'll keep updating this notebook with new methods / hyperparameters as I try to \"hill climb\" this notebook over the LB. \n\n**Make sure to check the previous versions so you can see what works**\n> - Use it as a reference for what you should / shoudn't try yourself. \n> - Saving you some time on failed attempts","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport matplotlib.pyplot as plt\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, PowerTransformer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T13:11:04.365265Z","iopub.execute_input":"2022-07-05T13:11:04.365726Z","iopub.status.idle":"2022-07-05T13:11:04.372668Z","shell.execute_reply.started":"2022-07-05T13:11:04.36569Z","shell.execute_reply":"2022-07-05T13:11:04.371689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:04.420329Z","iopub.execute_input":"2022-07-05T13:11:04.421297Z","iopub.status.idle":"2022-07-05T13:11:05.258772Z","shell.execute_reply.started":"2022-07-05T13:11:04.421244Z","shell.execute_reply":"2022-07-05T13:11:05.257639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:05.260811Z","iopub.execute_input":"2022-07-05T13:11:05.261159Z","iopub.status.idle":"2022-07-05T13:11:05.268355Z","shell.execute_reply.started":"2022-07-05T13:11:05.261127Z","shell.execute_reply":"2022-07-05T13:11:05.267148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:05.269725Z","iopub.execute_input":"2022-07-05T13:11:05.270059Z","iopub.status.idle":"2022-07-05T13:11:05.295357Z","shell.execute_reply.started":"2022-07-05T13:11:05.270029Z","shell.execute_reply":"2022-07-05T13:11:05.294462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(columns = \"id\")\ncols = list(df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:05.29756Z","iopub.execute_input":"2022-07-05T13:11:05.298105Z","iopub.status.idle":"2022-07-05T13:11:05.309363Z","shell.execute_reply.started":"2022-07-05T13:11:05.29807Z","shell.execute_reply":"2022-07-05T13:11:05.308445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int_cols = [i for i in df.columns if df[i].dtype == int]\nfloat_cols = [i for i in df.columns if df[i].dtype == float]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:05.310688Z","iopub.execute_input":"2022-07-05T13:11:05.311213Z","iopub.status.idle":"2022-07-05T13:11:05.32503Z","shell.execute_reply.started":"2022-07-05T13:11:05.31118Z","shell.execute_reply":"2022-07-05T13:11:05.323866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of missing values: \", df.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:05.326622Z","iopub.execute_input":"2022-07-05T13:11:05.327252Z","iopub.status.idle":"2022-07-05T13:11:05.343707Z","shell.execute_reply.started":"2022-07-05T13:11:05.327212Z","shell.execute_reply":"2022-07-05T13:11:05.342661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing\n- Here we try out different preprocessing pipelines","metadata":{}},{"cell_type":"code","source":"X_scaled = PowerTransformer().fit_transform(df)\nX_scaled = PowerTransformer().fit_transform(X_scaled)\n\nX_scaled = pd.DataFrame(X_scaled, columns = cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:05.344784Z","iopub.execute_input":"2022-07-05T13:11:05.345557Z","iopub.status.idle":"2022-07-05T13:11:13.322751Z","shell.execute_reply.started":"2022-07-05T13:11:05.345523Z","shell.execute_reply":"2022-07-05T13:11:13.321483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Clustering Algorithm\n- Here we define what clustering algorithm are we going to use","metadata":{}},{"cell_type":"code","source":"ALGORITHM = BayesianGaussianMixture","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:13.324475Z","iopub.execute_input":"2022-07-05T13:11:13.324964Z","iopub.status.idle":"2022-07-05T13:11:13.330532Z","shell.execute_reply.started":"2022-07-05T13:11:13.32492Z","shell.execute_reply":"2022-07-05T13:11:13.329324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Additional Hyperparameters\n- We now define a set of hyperparameters that we are **not** going to search values for. \n- We simply set them to be the same for all instances of our algorithm. ","metadata":{}},{"cell_type":"code","source":"additional_hyperparams = dict(\n                                n_init = 5\n                             )","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:13.332307Z","iopub.execute_input":"2022-07-05T13:11:13.333025Z","iopub.status.idle":"2022-07-05T13:11:13.342372Z","shell.execute_reply.started":"2022-07-05T13:11:13.332967Z","shell.execute_reply":"2022-07-05T13:11:13.341338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plotting the data before we start searching**","metadata":{}},{"cell_type":"code","source":"pca = PCA(random_state = 10, whiten = True)\nX_pca = pca.fit_transform(X_scaled)\nPCA_df = pd.DataFrame({\"PCA_1\" : X_pca[:,0], \"PCA_2\" : X_pca[:,1]})\nplt.figure(figsize=(14, 14))\nsns.scatterplot(data = PCA_df, x = \"PCA_1\", y = \"PCA_2\", s=3);","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:13.348488Z","iopub.execute_input":"2022-07-05T13:11:13.34974Z","iopub.status.idle":"2022-07-05T13:11:14.043817Z","shell.execute_reply.started":"2022-07-05T13:11:13.349686Z","shell.execute_reply":"2022-07-05T13:11:14.04266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"_____","metadata":{}},{"cell_type":"markdown","source":"# Brute Force 🔥\n## Searching for the optimal hyperparameters\n- We now search the range of possible values to assign to our algorithm hyperparameters\n- **Note:** The final one we use each time is names `preds_1`, this allows it to be used at the end of the notebook for submission.","metadata":{}},{"cell_type":"markdown","source":"### n_components = 2","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components = 2, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:14.045438Z","iopub.execute_input":"2022-07-05T13:11:14.045891Z","iopub.status.idle":"2022-07-05T13:11:30.170617Z","shell.execute_reply.started":"2022-07-05T13:11:14.045854Z","shell.execute_reply":"2022-07-05T13:11:30.16947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 3","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=3, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:11:30.172243Z","iopub.execute_input":"2022-07-05T13:11:30.172683Z","iopub.status.idle":"2022-07-05T13:12:03.962449Z","shell.execute_reply.started":"2022-07-05T13:11:30.172642Z","shell.execute_reply":"2022-07-05T13:12:03.961077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 4","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=4, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:12:03.964337Z","iopub.execute_input":"2022-07-05T13:12:03.965154Z","iopub.status.idle":"2022-07-05T13:12:48.879069Z","shell.execute_reply.started":"2022-07-05T13:12:03.9651Z","shell.execute_reply":"2022-07-05T13:12:48.877836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 5","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=5, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:12:48.880699Z","iopub.execute_input":"2022-07-05T13:12:48.881054Z","iopub.status.idle":"2022-07-05T13:13:48.746248Z","shell.execute_reply.started":"2022-07-05T13:12:48.881023Z","shell.execute_reply":"2022-07-05T13:13:48.744976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 6","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=6, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:13:48.747651Z","iopub.execute_input":"2022-07-05T13:13:48.748007Z","iopub.status.idle":"2022-07-05T13:14:50.949082Z","shell.execute_reply.started":"2022-07-05T13:13:48.747975Z","shell.execute_reply":"2022-07-05T13:14:50.947785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 7","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=7, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:14:50.950849Z","iopub.execute_input":"2022-07-05T13:14:50.951798Z","iopub.status.idle":"2022-07-05T13:16:04.494351Z","shell.execute_reply.started":"2022-07-05T13:14:50.95176Z","shell.execute_reply":"2022-07-05T13:16:04.493294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Assigning to `preds_1` for submission**\n\n(Move this cell wherever you want to submit different instances to the LB)","metadata":{}},{"cell_type":"code","source":"# Submission cell\npreds_1 = preds","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:16:04.495948Z","iopub.execute_input":"2022-07-05T13:16:04.497196Z","iopub.status.idle":"2022-07-05T13:16:04.501914Z","shell.execute_reply.started":"2022-07-05T13:16:04.497149Z","shell.execute_reply":"2022-07-05T13:16:04.500527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 8","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=8, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:16:04.502972Z","iopub.execute_input":"2022-07-05T13:16:04.503294Z","iopub.status.idle":"2022-07-05T13:17:25.438884Z","shell.execute_reply.started":"2022-07-05T13:16:04.503265Z","shell.execute_reply":"2022-07-05T13:17:25.437696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 9","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=9, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:17:25.440396Z","iopub.execute_input":"2022-07-05T13:17:25.441179Z","iopub.status.idle":"2022-07-05T13:19:08.755477Z","shell.execute_reply.started":"2022-07-05T13:17:25.441132Z","shell.execute_reply":"2022-07-05T13:19:08.754365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n_components = 10","metadata":{}},{"cell_type":"code","source":"gmm = ALGORITHM(n_components=10, **additional_hyperparams)\npreds = gmm.fit_predict(X_scaled)\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\ndf = pd.DataFrame({\"x\" : reduced_data[:,0], \"y\" : reduced_data[:,1], \"clusters\" : preds})\nplt.figure(figsize=(20, 10))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:19:08.757092Z","iopub.execute_input":"2022-07-05T13:19:08.757492Z","iopub.status.idle":"2022-07-05T13:21:01.654345Z","shell.execute_reply.started":"2022-07-05T13:19:08.757454Z","shell.execute_reply":"2022-07-05T13:21:01.653569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:21:01.655737Z","iopub.execute_input":"2022-07-05T13:21:01.656294Z","iopub.status.idle":"2022-07-05T13:21:01.6802Z","shell.execute_reply.started":"2022-07-05T13:21:01.656259Z","shell.execute_reply":"2022-07-05T13:21:01.6794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"Predicted\"] = preds_1\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:21:01.681514Z","iopub.execute_input":"2022-07-05T13:21:01.682116Z","iopub.status.idle":"2022-07-05T13:21:01.694664Z","shell.execute_reply.started":"2022-07-05T13:21:01.682082Z","shell.execute_reply":"2022-07-05T13:21:01.693243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:21:01.696304Z","iopub.execute_input":"2022-07-05T13:21:01.696646Z","iopub.status.idle":"2022-07-05T13:21:01.865695Z","shell.execute_reply.started":"2022-07-05T13:21:01.696616Z","shell.execute_reply":"2022-07-05T13:21:01.864532Z"},"trusted":true},"execution_count":null,"outputs":[]}]}