{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hey kagglers hope you all having good time\n\nThis is my first approach on this competition, I hope you find it usefull","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import warnings\nimport sklearn\nimport numpy as np\nimport pandas as pd\nimport sklearn.utils\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.preprocessing import *\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom sklearn.cluster import KMeans, DBSCAN\nfrom sklearn.model_selection import train_test_split\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T20:55:58.807669Z","iopub.execute_input":"2022-07-14T20:55:58.808437Z","iopub.status.idle":"2022-07-14T20:55:59.433632Z","shell.execute_reply.started":"2022-07-14T20:55:58.808364Z","shell.execute_reply":"2022-07-14T20:55:59.432387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install pycaret --ignore-installed llvmlite","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-14T20:56:06.594775Z","iopub.execute_input":"2022-07-14T20:56:06.595573Z","iopub.status.idle":"2022-07-14T20:56:06.600969Z","shell.execute_reply.started":"2022-07-14T20:56:06.595529Z","shell.execute_reply":"2022-07-14T20:56:06.599693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycaret.clustering import *","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-14T20:56:07.706393Z","iopub.execute_input":"2022-07-14T20:56:07.707304Z","iopub.status.idle":"2022-07-14T20:56:07.713029Z","shell.execute_reply.started":"2022-07-14T20:56:07.707250Z","shell.execute_reply":"2022-07-14T20:56:07.711798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:56:10.246444Z","iopub.execute_input":"2022-07-14T20:56:10.247549Z","iopub.status.idle":"2022-07-14T20:56:11.148492Z","shell.execute_reply.started":"2022-07-14T20:56:10.247502Z","shell.execute_reply":"2022-07-14T20:56:11.147051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:56:12.376484Z","iopub.execute_input":"2022-07-14T20:56:12.377639Z","iopub.status.idle":"2022-07-14T20:56:12.602369Z","shell.execute_reply.started":"2022-07-14T20:56:12.377598Z","shell.execute_reply":"2022-07-14T20:56:12.601199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:56:14.046752Z","iopub.execute_input":"2022-07-14T20:56:14.047217Z","iopub.status.idle":"2022-07-14T20:56:14.070555Z","shell.execute_reply.started":"2022-07-14T20:56:14.047181Z","shell.execute_reply":"2022-07-14T20:56:14.069458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['id'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:56:15.478692Z","iopub.execute_input":"2022-07-14T20:56:15.479849Z","iopub.status.idle":"2022-07-14T20:56:15.490852Z","shell.execute_reply.started":"2022-07-14T20:56:15.479805Z","shell.execute_reply":"2022-07-14T20:56:15.489853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"j = 1 \nplt.figure()\nint_cols = ['f_07','f_08','f_09','f_10','f_11','f_12','f_13']\nfig, ax = plt.subplots(10, 3 ,figsize=(24, 28))\n\nfor col in df.columns:\n    if col in int_cols:\n        pass\n    else:\n        plt.subplot(11, 2, j)\n        sns.histplot(df[col], kde=True,bins=40, label=col, color='green')\n        plt.xlabel(col, fontsize=9); \n        plt.rcParams['axes.facecolor'] = 'black'\n        plt.legend()\n        j += 1\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:56:40.061624Z","iopub.execute_input":"2022-07-14T20:56:40.062341Z","iopub.status.idle":"2022-07-14T20:56:58.932548Z","shell.execute_reply.started":"2022-07-14T20:56:40.062303Z","shell.execute_reply":"2022-07-14T20:56:58.931344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 1\nplt.figure();\nfig, ax = plt.subplots(7, 1 ,figsize=(22, 25))\nfor col in df.columns:\n    if col in int_cols:\n        plt.subplot(7, 1, i)\n        sns.histplot(df[col], kde=True,bins=40, label=col, color='pink')\n        plt.rcParams['axes.facecolor'] = 'black'\n        plt.xlabel(col, fontsize=9)\n        plt.legend()\n        i += 1\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:57:08.259573Z","iopub.execute_input":"2022-07-14T20:57:08.260034Z","iopub.status.idle":"2022-07-14T20:57:14.272060Z","shell.execute_reply.started":"2022-07-14T20:57:08.259995Z","shell.execute_reply":"2022-07-14T20:57:14.270464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:16.159358Z","iopub.execute_input":"2022-07-14T20:58:16.159834Z","iopub.status.idle":"2022-07-14T20:58:16.167978Z","shell.execute_reply.started":"2022-07-14T20:58:16.159802Z","shell.execute_reply":"2022-07-14T20:58:16.166988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"We need to apply Power Transformer to make out data distribution more gaussian\"\"\"\nTransformer = PowerTransformer()\ntransformed = Transformer.fit_transform(df[int_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:17.313831Z","iopub.execute_input":"2022-07-14T20:58:17.315244Z","iopub.status.idle":"2022-07-14T20:58:18.099927Z","shell.execute_reply.started":"2022-07-14T20:58:17.315197Z","shell.execute_reply":"2022-07-14T20:58:18.098591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_2 = pd.DataFrame(transformed , columns = int_cols)\n\ndf.drop(int_cols, axis=1, inplace=True)\ndf = pd.concat([df, df_2], axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:19.363759Z","iopub.execute_input":"2022-07-14T20:58:19.364668Z","iopub.status.idle":"2022-07-14T20:58:19.383967Z","shell.execute_reply.started":"2022-07-14T20:58:19.364621Z","shell.execute_reply":"2022-07-14T20:58:19.383027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:24.566894Z","iopub.execute_input":"2022-07-14T20:58:24.567362Z","iopub.status.idle":"2022-07-14T20:58:24.612380Z","shell.execute_reply.started":"2022-07-14T20:58:24.567316Z","shell.execute_reply":"2022-07-14T20:58:24.611052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization after Preprocess","metadata":{}},{"cell_type":"code","source":"i = 1\nplt.figure();\nfig, ax = plt.subplots(7, 1 ,figsize=(22, 25))\nfor col in df.columns:\n    if col in int_cols:\n        plt.subplot(7, 1, i)\n        sns.histplot(df[col], kde=True,bins=40, label=col, color='red')\n        plt.rcParams['axes.facecolor'] = 'black'\n        plt.xlabel(col, fontsize=9)\n        plt.legend()\n        i += 1\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:28.267157Z","iopub.execute_input":"2022-07-14T20:58:28.268112Z","iopub.status.idle":"2022-07-14T20:58:33.699426Z","shell.execute_reply.started":"2022-07-14T20:58:28.268074Z","shell.execute_reply":"2022-07-14T20:58:33.698503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, the data distribution became more Gaussian","metadata":{}},{"cell_type":"code","source":"\"\"\"Checking the correlation between columns\"\"\"\nfrom plotly.offline import init_notebook_mode, iplot\nimport plotly.express as px\ninit_notebook_mode(connected=True)\n\ncorr_plot = df.corr()\nplt.figure(figsize=(16,16))\nfig = px.imshow(corr_plot, text_auto=True, aspect='auto')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:39.431932Z","iopub.execute_input":"2022-07-14T20:58:39.433053Z","iopub.status.idle":"2022-07-14T20:58:39.970484Z","shell.execute_reply.started":"2022-07-14T20:58:39.432995Z","shell.execute_reply":"2022-07-14T20:58:39.969525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"Checking the abs correlation between columns\"\"\"\ncorr_plot = df.corr().abs()\nplt.figure(figsize=(16,16))\nfig = px.bar(corr_plot, )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:43.215617Z","iopub.execute_input":"2022-07-14T20:58:43.216279Z","iopub.status.idle":"2022-07-14T20:58:43.725893Z","shell.execute_reply.started":"2022-07-14T20:58:43.216244Z","shell.execute_reply":"2022-07-14T20:58:43.724674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4>As a result of correlation analysis, we can see that some features are compeletly useless because of their very low correlation\nand more important columns are : </h4>\n\n* 'f_07'\n* 'f_08'\n* 'f_09'\n* 'f_10'\n* 'f_11'\n* 'f_12'\n* 'f_13'\n* 'f_22'\n* 'f_23'\n* 'f_24'\n* 'f_25'\n* 'f_26'\n* 'f_27'\n* 'f_28'","metadata":{}},{"cell_type":"markdown","source":"Based on our EDA, we can see that we have a high dimensional dataset which means that the dataset has a large number of features.<br>The primary problem associated with high-dimensionality is, it reduces the ability to generalize beyond the examples in the training set. (Over-fitting**)\n<h4> Here come the Principal Component Analysis (PCA):<br></h4>\n    Principal Component Analysis (PCA) is an unsupervised, non-parametric statistical technique primarily used for dimensionality reduction in  machine learning. PCA reduces the dimensionality of high dimensional datasets and increases the interpretability but at the same time minimizes information loss.","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:47.417329Z","iopub.execute_input":"2022-07-14T20:58:47.418220Z","iopub.status.idle":"2022-07-14T20:58:47.422910Z","shell.execute_reply.started":"2022-07-14T20:58:47.418178Z","shell.execute_reply":"2022-07-14T20:58:47.421886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA().fit(df)\n\nplt.rcParams[\"figure.figsize\"] = (18,7)\n\nfig, ax = plt.subplots()\nxi = np.arange(0, 29, step=1)\ny = np.cumsum(pca.explained_variance_ratio_)\n\nplt.ylim(0.0,1.1)\nplt.plot(xi, y, marker='o', linestyle='-', color='black')\n\nplt.xlabel('Number of Components')\nplt.xticks(np.arange(0, 29, step=1)) \nplt.ylabel('Cumulative variance (%)')\nplt.title('The number of components needed to explain variance', fontsize=12)\nplt.rcParams['axes.facecolor'] = 'white'\n\nplt.axhline(y=0.95, color='r', linestyle='-')\nplt.text(0.5, 0.85, '95% cut-off threshold', color = 'black', fontsize=24)\n\nax.grid(axis='x')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:53.455138Z","iopub.execute_input":"2022-07-14T20:58:53.456436Z","iopub.status.idle":"2022-07-14T20:58:53.989066Z","shell.execute_reply.started":"2022-07-14T20:58:53.456375Z","shell.execute_reply":"2022-07-14T20:58:53.987797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So the best component values seems to be 25","metadata":{}},{"cell_type":"markdown","source":"# Clustering","metadata":{}},{"cell_type":"markdown","source":"<h3> K-Means</h3>\n<h4>K-Means Clustering is one of the most popular and simplest clustering methods. K is the number of all clusters, while C represents each individual cluster. Our goal is to minimize W, which is the measure of within-cluster variation.</h4> <br>\n<center><img src='https://i.imgur.com/WL1tIZ4.gif' width=\"600\" height=\"600\"></center>\n\n\n","metadata":{}},{"cell_type":"code","source":"cluster = setup(df, session_id = 1, pca=True, pca_components=25);","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:58:57.946378Z","iopub.execute_input":"2022-07-14T20:58:57.946795Z","iopub.status.idle":"2022-07-14T20:59:02.864984Z","shell.execute_reply.started":"2022-07-14T20:58:57.946764Z","shell.execute_reply":"2022-07-14T20:59:02.863507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans = create_model('kmeans');","metadata":{"execution":{"iopub.status.busy":"2022-07-14T20:59:06.576530Z","iopub.execute_input":"2022-07-14T20:59:06.576983Z","iopub.status.idle":"2022-07-14T21:02:32.985515Z","shell.execute_reply.started":"2022-07-14T20:59:06.576934Z","shell.execute_reply":"2022-07-14T21:02:32.984001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"Printing model info\"\"\"\nprint(kmeans)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:02:49.857902Z","iopub.execute_input":"2022-07-14T21:02:49.858344Z","iopub.status.idle":"2022-07-14T21:02:49.864999Z","shell.execute_reply.started":"2022-07-14T21:02:49.858310Z","shell.execute_reply":"2022-07-14T21:02:49.863980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_model(kmeans);","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:02:51.016332Z","iopub.execute_input":"2022-07-14T21:02:51.017743Z","iopub.status.idle":"2022-07-14T21:02:52.589762Z","shell.execute_reply.started":"2022-07-14T21:02:51.017700Z","shell.execute_reply":"2022-07-14T21:02:52.588894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(kmeans, plot ='elbow');","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:03:38.470036Z","iopub.execute_input":"2022-07-14T21:03:38.470486Z","iopub.status.idle":"2022-07-14T21:04:45.663860Z","shell.execute_reply.started":"2022-07-14T21:03:38.470453Z","shell.execute_reply":"2022-07-14T21:04:45.662871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Final_kmeans = create_model('kmeans', num_clusters = 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:05:12.928226Z","iopub.execute_input":"2022-07-14T21:05:12.929280Z","iopub.status.idle":"2022-07-14T21:08:37.575772Z","shell.execute_reply.started":"2022-07-14T21:05:12.929235Z","shell.execute_reply":"2022-07-14T21:08:37.574672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(kmeans, plot = 'silhouette');","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:08:58.856042Z","iopub.execute_input":"2022-07-14T21:08:58.856828Z","iopub.status.idle":"2022-07-14T21:15:43.408454Z","shell.execute_reply.started":"2022-07-14T21:08:58.856786Z","shell.execute_reply":"2022-07-14T21:15:43.407045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_model(kmeans, 'kmeans_clustering_model')","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-14T21:15:52.858677Z","iopub.execute_input":"2022-07-14T21:15:52.859185Z","iopub.status.idle":"2022-07-14T21:15:53.107815Z","shell.execute_reply.started":"2022-07-14T21:15:52.859145Z","shell.execute_reply":"2022-07-14T21:15:53.106372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = assign_model(kmeans)\nresult['Cluster'] = result['Cluster'].map(lambda x: x.lstrip('Cluster'))\nresult.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:15:54.017166Z","iopub.execute_input":"2022-07-14T21:15:54.017601Z","iopub.status.idle":"2022-07-14T21:15:54.169905Z","shell.execute_reply.started":"2022-07-14T21:15:54.017567Z","shell.execute_reply":"2022-07-14T21:15:54.168480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result['Cluster'] = result['Cluster'].astype('int16')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:15:54.986498Z","iopub.execute_input":"2022-07-14T21:15:54.986925Z","iopub.status.idle":"2022-07-14T21:15:55.006908Z","shell.execute_reply.started":"2022-07-14T21:15:54.986884Z","shell.execute_reply":"2022-07-14T21:15:55.006015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"we go back and re run the read command in order to load Ids\"\"\"\nids = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:15:55.737300Z","iopub.execute_input":"2022-07-14T21:15:55.737995Z","iopub.status.idle":"2022-07-14T21:15:56.603189Z","shell.execute_reply.started":"2022-07-14T21:15:55.737944Z","shell.execute_reply":"2022-07-14T21:15:56.601693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# new_id = pd.DataFrame(ids['id'].values , columns=['Id'])\nnew_result = pd.DataFrame(result['Cluster'].values, columns=['Predicted'])\nFinal_result = pd.concat([ids,new_result], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:15:57.878019Z","iopub.execute_input":"2022-07-14T21:15:57.878451Z","iopub.status.idle":"2022-07-14T21:15:57.892033Z","shell.execute_reply.started":"2022-07-14T21:15:57.878418Z","shell.execute_reply":"2022-07-14T21:15:57.890653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Final_result = Final_result.drop(Final_result.iloc[:,:-2], axis=1)\nFinal_result.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:16:00.589125Z","iopub.execute_input":"2022-07-14T21:16:00.589526Z","iopub.status.idle":"2022-07-14T21:16:00.622968Z","shell.execute_reply.started":"2022-07-14T21:16:00.589497Z","shell.execute_reply":"2022-07-14T21:16:00.621850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Final_result.to_csv('submission.csv', index=False)\n# The main output file is submission","metadata":{"execution":{"iopub.status.busy":"2022-07-14T21:16:01.417191Z","iopub.execute_input":"2022-07-14T21:16:01.417624Z","iopub.status.idle":"2022-07-14T21:16:01.770444Z","shell.execute_reply.started":"2022-07-14T21:16:01.417592Z","shell.execute_reply":"2022-07-14T21:16:01.769282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}