{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  **Tabular Playground Series July 2022**\nBy Tarun Jagadish","metadata":{}},{"cell_type":"markdown","source":"Importing useful libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport seaborn as sns \nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.decomposition import PCA\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.mixture import GaussianMixture\nfrom sklearn.cluster import MeanShift\nfrom sklearn.cluster import DBSCAN\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.cluster import AgglomerativeClustering\nfrom matplotlib import pyplot as plt\nimport plotly.express as px\nimport scipy.stats as stats\n%matplotlib inline\nsns.set_theme(style = \"whitegrid\")\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T05:46:48.275800Z","iopub.execute_input":"2022-07-31T05:46:48.276600Z","iopub.status.idle":"2022-07-31T05:46:50.738697Z","shell.execute_reply.started":"2022-07-31T05:46:48.276500Z","shell.execute_reply":"2022-07-31T05:46:50.737506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read the dataset","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:50.741239Z","iopub.execute_input":"2022-07-31T05:46:50.742032Z","iopub.status.idle":"2022-07-31T05:46:52.082582Z","shell.execute_reply.started":"2022-07-31T05:46:50.741989Z","shell.execute_reply":"2022-07-31T05:46:52.081517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Preview the data","metadata":{}},{"cell_type":"code","source":"df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:52.083677Z","iopub.execute_input":"2022-07-31T05:46:52.083978Z","iopub.status.idle":"2022-07-31T05:46:52.124748Z","shell.execute_reply.started":"2022-07-31T05:46:52.083952Z","shell.execute_reply":"2022-07-31T05:46:52.123738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualize the distribution of columns","metadata":{}},{"cell_type":"code","source":"df_vis=df.drop(columns='id')\nplt.figure(dpi=200, figsize=(10, 5))\nplt.boxplot(df_vis)\nplt.grid(True)\nplt.title(\"Column Distributions\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:52.126935Z","iopub.execute_input":"2022-07-31T05:46:52.127288Z","iopub.status.idle":"2022-07-31T05:46:53.226774Z","shell.execute_reply.started":"2022-07-31T05:46:52.127255Z","shell.execute_reply":"2022-07-31T05:46:53.225506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can clearly divide the columns into 4 groups:\n* f_00-f_06:normally distributed\n* f_07-f_13:skewed??\n* f_14_f_21:normally distrubuted\n* f_22-f_28:normally distributed","metadata":{}},{"cell_type":"markdown","source":"let's compare their standard deviations.","metadata":{}},{"cell_type":"code","source":"df_1=df_vis[['f_00','f_01','f_02','f_03','f_04','f_05','f_06']]\ndf_1","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.228071Z","iopub.execute_input":"2022-07-31T05:46:53.228403Z","iopub.status.idle":"2022-07-31T05:46:53.249087Z","shell.execute_reply.started":"2022-07-31T05:46:53.228373Z","shell.execute_reply":"2022-07-31T05:46:53.247887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1_stat=df_1.describe()\ndf_1_stat","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.251042Z","iopub.execute_input":"2022-07-31T05:46:53.251547Z","iopub.status.idle":"2022-07-31T05:46:53.324487Z","shell.execute_reply.started":"2022-07-31T05:46:53.251503Z","shell.execute_reply":"2022-07-31T05:46:53.323283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Standard deviation range is 0.997434-1.003061","metadata":{}},{"cell_type":"code","source":"df_2=df_vis[['f_07','f_08','f_09','f_10','f_11','f_12','f_13']]\ndf_2","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.325675Z","iopub.execute_input":"2022-07-31T05:46:53.325984Z","iopub.status.idle":"2022-07-31T05:46:53.342905Z","shell.execute_reply.started":"2022-07-31T05:46:53.325956Z","shell.execute_reply":"2022-07-31T05:46:53.341656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_2_stat=df_2.describe()\ndf_2_stat","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.344731Z","iopub.execute_input":"2022-07-31T05:46:53.345165Z","iopub.status.idle":"2022-07-31T05:46:53.402370Z","shell.execute_reply.started":"2022-07-31T05:46:53.345123Z","shell.execute_reply":"2022-07-31T05:46:53.401240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Standard deviation range is 3.691840-5.904919","metadata":{}},{"cell_type":"code","source":"df_3=df_vis[['f_14','f_15','f_16','f_17','f_18','f_19','f_20','f_21']]\ndf_3","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.403788Z","iopub.execute_input":"2022-07-31T05:46:53.404167Z","iopub.status.idle":"2022-07-31T05:46:53.425341Z","shell.execute_reply.started":"2022-07-31T05:46:53.404136Z","shell.execute_reply":"2022-07-31T05:46:53.424472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_3_stat=df_3.describe()\ndf_3_stat","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.429875Z","iopub.execute_input":"2022-07-31T05:46:53.430292Z","iopub.status.idle":"2022-07-31T05:46:53.508403Z","shell.execute_reply.started":"2022-07-31T05:46:53.430255Z","shell.execute_reply":"2022-07-31T05:46:53.507177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Standard deviation range is 0.995965-1.003277","metadata":{}},{"cell_type":"code","source":"df_4=df_vis[['f_22','f_23','f_24','f_25','f_26','f_27','f_28']]\ndf_4","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.509538Z","iopub.execute_input":"2022-07-31T05:46:53.509924Z","iopub.status.idle":"2022-07-31T05:46:53.535456Z","shell.execute_reply.started":"2022-07-31T05:46:53.509888Z","shell.execute_reply":"2022-07-31T05:46:53.534392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_4_stat=df_4.describe()\ndf_4_stat","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.536788Z","iopub.execute_input":"2022-07-31T05:46:53.537768Z","iopub.status.idle":"2022-07-31T05:46:53.604273Z","shell.execute_reply.started":"2022-07-31T05:46:53.537724Z","shell.execute_reply":"2022-07-31T05:46:53.603037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Standard deviation range is 1.305407-1.576086","metadata":{}},{"cell_type":"markdown","source":"It is understood that the second dataframe is an outlier as compared to the other three.","metadata":{}},{"cell_type":"markdown","source":"We now know that df_2 has integer values and the other dataframes have float data","metadata":{}},{"cell_type":"code","source":"df_2","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.606113Z","iopub.execute_input":"2022-07-31T05:46:53.606446Z","iopub.status.idle":"2022-07-31T05:46:53.619457Z","shell.execute_reply.started":"2022-07-31T05:46:53.606418Z","shell.execute_reply":"2022-07-31T05:46:53.618654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(dpi=200, figsize=(10, 5))\nplt.boxplot(df_2)\nplt.grid(True)\nplt.title(\"Integer Data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:53.620605Z","iopub.execute_input":"2022-07-31T05:46:53.621458Z","iopub.status.idle":"2022-07-31T05:46:54.048390Z","shell.execute_reply.started":"2022-07-31T05:46:53.621427Z","shell.execute_reply":"2022-07-31T05:46:54.047552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f7=df_2[['f_07']]\nf8=df_2[['f_08']]\nf9=df_2[['f_09']]\nf10=df_2[['f_10']]\nf11=df_2[['f_11']]\nf12=df_2[['f_12']]\nf13=df_2[['f_13']]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:54.049599Z","iopub.execute_input":"2022-07-31T05:46:54.050529Z","iopub.status.idle":"2022-07-31T05:46:54.061287Z","shell.execute_reply.started":"2022-07-31T05:46:54.050462Z","shell.execute_reply":"2022-07-31T05:46:54.060271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f7","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:54.062855Z","iopub.execute_input":"2022-07-31T05:46:54.063152Z","iopub.status.idle":"2022-07-31T05:46:54.079168Z","shell.execute_reply.started":"2022-07-31T05:46:54.063126Z","shell.execute_reply":"2022-07-31T05:46:54.078356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get rid of useless column Id","metadata":{}},{"cell_type":"code","source":"df.set_index('id', inplace = True)\nId = pd.Series(df.index, name = 'Id')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:54.080566Z","iopub.execute_input":"2022-07-31T05:46:54.081068Z","iopub.status.idle":"2022-07-31T05:46:54.107877Z","shell.execute_reply.started":"2022-07-31T05:46:54.081037Z","shell.execute_reply":"2022-07-31T05:46:54.107116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Distribution of Numeric Data","metadata":{}},{"cell_type":"code","source":"data = pd.melt(df, value_vars = df.columns)\ng = sns.FacetGrid(data, col = \"variable\", col_wrap = 4, sharex = False, sharey = False)\ng = g.map(sns.histplot, \"value\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:46:54.108978Z","iopub.execute_input":"2022-07-31T05:46:54.109509Z","iopub.status.idle":"2022-07-31T05:47:25.514772Z","shell.execute_reply.started":"2022-07-31T05:46:54.109476Z","shell.execute_reply":"2022-07-31T05:47:25.513388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Standardization and Distribution of Scaled Data","metadata":{}},{"cell_type":"code","source":"data_scaled = df.copy()\ndata_scaled[:] = PowerTransformer().fit_transform(df)\ndata_scaled[:] = RobustScaler().fit_transform(data_scaled)\n\ndf = pd.melt(data_scaled, value_vars = data_scaled.columns)\ng = sns.FacetGrid(df, col = \"variable\", col_wrap = 4, sharex = False, sharey = False)\ng = g.map(sns.histplot, \"value\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:47:25.516313Z","iopub.execute_input":"2022-07-31T05:47:25.516676Z","iopub.status.idle":"2022-07-31T05:48:01.889306Z","shell.execute_reply.started":"2022-07-31T05:47:25.516643Z","shell.execute_reply":"2022-07-31T05:48:01.888438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PCA","metadata":{}},{"cell_type":"code","source":"X = data_scaled.copy()\npca = PCA(n_components = 2)\nsample = X.sample(700)\ntransformed = pd.DataFrame(pca.fit_transform(sample))\nplt.scatter(transformed[0], transformed[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:48:01.890492Z","iopub.execute_input":"2022-07-31T05:48:01.891162Z","iopub.status.idle":"2022-07-31T05:48:02.180400Z","shell.execute_reply.started":"2022-07-31T05:48:01.891126Z","shell.execute_reply":"2022-07-31T05:48:02.179291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"KMeans Clustering","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\ncs = []\nfor i in range(1, 11):\n    kmeans = KMeans(n_clusters = i, init = 'k-means++', max_iter = 300, n_init = 10, random_state = 0)\n    kmeans.fit(X)\n    cs.append(kmeans.inertia_)\nplt.plot(range(1, 11), cs)\nplt.title('The Elbow Method')\nplt.xlabel('Number of clusters')\nplt.ylabel('CS')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:48:02.181813Z","iopub.execute_input":"2022-07-31T05:48:02.182648Z","iopub.status.idle":"2022-07-31T05:48:53.876023Z","shell.execute_reply.started":"2022-07-31T05:48:02.182616Z","shell.execute_reply":"2022-07-31T05:48:53.874755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no visible kinks, Assuming 4 as optimal number of clusters","metadata":{}},{"cell_type":"markdown","source":"Gaussian Mixture Model","metadata":{}},{"cell_type":"code","source":"gmm_model_4 = GaussianMixture(n_components = 4)\ngmm_model_4.fit(X)\n\nPredicted4 = pd.Series(gmm_model_4.predict(X), name = 'Predicted')\nsubmission4 = pd.concat([Id, Predicted4], axis = 1)\nsubmission4.to_csv('Gauss_submission_4.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:48:53.877809Z","iopub.execute_input":"2022-07-31T05:48:53.878147Z","iopub.status.idle":"2022-07-31T05:49:00.198302Z","shell.execute_reply.started":"2022-07-31T05:48:53.878116Z","shell.execute_reply":"2022-07-31T05:49:00.196891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Bayesian Gaussian Mixture","metadata":{}},{"cell_type":"code","source":"\nbgm = BayesianGaussianMixture(n_components = 2, random_state = 1, max_iter = 500, n_init = 3, verbose = 0)\nbgm.fit(X)\nPredicted = pd.Series(bgm.predict(X), name = 'Predicted')\nsubmission3 = pd.concat([Id, Predicted], axis = 1)\nsubmission3.to_csv('BGM_2.csv', index = False)\n\nbgm = BayesianGaussianMixture(n_components = 3, random_state = 1, max_iter = 500, n_init = 3, verbose = 0)\nbgm.fit(X)\nPredicted = pd.Series(bgm.predict(X), name = 'Predicted')\nsubmission3 = pd.concat([Id, Predicted], axis = 1)\nsubmission3.to_csv('BGM_3.csv', index = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm = BayesianGaussianMixture(n_components = 4, random_state = 1, max_iter = 500, n_init = 3, verbose = 0)\nbgm.fit(X)\nPredicted = pd.Series(bgm.predict(X), name = 'Predicted')\nsubmission3 = pd.concat([Id, Predicted], axis = 1)\nsubmission3.to_csv('BGM_4.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:00:36.782604Z","iopub.execute_input":"2022-07-31T06:00:36.783060Z","iopub.status.idle":"2022-07-31T06:03:02.656358Z","shell.execute_reply.started":"2022-07-31T06:00:36.783014Z","shell.execute_reply":"2022-07-31T06:03:02.655280Z"},"trusted":true},"execution_count":null,"outputs":[]}]}