{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Steps\n* Objective\n* Data Wrangling\n* Model Generation\n* Submission","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Objective\n\nObjective is to build a model to find the different data groups/clusters. \nThe data given are (simulated) manufacturing control data that can be clustered into different control states. There is no indication of how many possible control states there are. Its an Unsupervised Learning problem","metadata":{}},{"cell_type":"markdown","source":"# Data Wrangling","metadata":{}},{"cell_type":"code","source":"# Importing necessary libraries\nimport os\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.preprocessing import MaxAbsScaler,PowerTransformer,RobustScaler,StandardScaler, MinMaxScaler\n\nfrom collections import Counter\nfrom tqdm import tqdm\nprint(\"Necessary Libraries imported\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:42.687034Z","iopub.execute_input":"2022-07-14T11:22:42.687541Z","iopub.status.idle":"2022-07-14T11:22:43.656620Z","shell.execute_reply.started":"2022-07-14T11:22:42.687437Z","shell.execute_reply":"2022-07-14T11:22:43.655471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the file names\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:43.658394Z","iopub.execute_input":"2022-07-14T11:22:43.658713Z","iopub.status.idle":"2022-07-14T11:22:43.665689Z","shell.execute_reply.started":"2022-07-14T11:22:43.658687Z","shell.execute_reply":"2022-07-14T11:22:43.664695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing data\ndata = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/data.csv\",index_col=\"id\")\nsubmission = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv\")\nprint(\"Data Imported\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:43.667333Z","iopub.execute_input":"2022-07-14T11:22:43.667975Z","iopub.status.idle":"2022-07-14T11:22:45.242398Z","shell.execute_reply.started":"2022-07-14T11:22:43.667936Z","shell.execute_reply":"2022-07-14T11:22:45.241199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data top 5 rows\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:45.246739Z","iopub.execute_input":"2022-07-14T11:22:45.247077Z","iopub.status.idle":"2022-07-14T11:22:45.282191Z","shell.execute_reply.started":"2022-07-14T11:22:45.247049Z","shell.execute_reply":"2022-07-14T11:22:45.280671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data shape\ndata.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:45.283877Z","iopub.execute_input":"2022-07-14T11:22:45.284257Z","iopub.status.idle":"2022-07-14T11:22:45.292338Z","shell.execute_reply.started":"2022-07-14T11:22:45.284226Z","shell.execute_reply":"2022-07-14T11:22:45.291226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basic data info\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:45.293821Z","iopub.execute_input":"2022-07-14T11:22:45.294293Z","iopub.status.idle":"2022-07-14T11:22:45.326493Z","shell.execute_reply.started":"2022-07-14T11:22:45.294255Z","shell.execute_reply":"2022-07-14T11:22:45.325239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basic stats\ndata.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:45.327510Z","iopub.execute_input":"2022-07-14T11:22:45.327839Z","iopub.status.idle":"2022-07-14T11:22:45.528898Z","shell.execute_reply.started":"2022-07-14T11:22:45.327808Z","shell.execute_reply":"2022-07-14T11:22:45.528041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Seems no missing values here\n* Maximum value of 44 in feature f_09\n* Minimum value of -8.234305 in feature f_23\n* Need to tranform the data with scaler","metadata":{}},{"cell_type":"code","source":"# Column list of data\ncols = list(data.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:45.529993Z","iopub.execute_input":"2022-07-14T11:22:45.531014Z","iopub.status.idle":"2022-07-14T11:22:45.536034Z","shell.execute_reply.started":"2022-07-14T11:22:45.530978Z","shell.execute_reply":"2022-07-14T11:22:45.534711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for correlations\nplt.figure(figsize=(10,8))\nsns.heatmap(data.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:45.537709Z","iopub.execute_input":"2022-07-14T11:22:45.538082Z","iopub.status.idle":"2022-07-14T11:22:46.374195Z","shell.execute_reply.started":"2022-07-14T11:22:45.538051Z","shell.execute_reply":"2022-07-14T11:22:46.372936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seems no strong relationship among the features, but some features are negatively correlated also","metadata":{}},{"cell_type":"markdown","source":"Using Scalers and Transformer for data transformation\n* Scalers are linear (or more precisely affine) transformers and differ from each other in the way they estimate the parameters used to shift and scale each feature.\n* PowerTransformer provides non-linear transformations in which data is mapped to a normal distribution to stabilize variance and minimize skewness.","metadata":{}},{"cell_type":"code","source":"# Standardisation of data\n# Using MaxAbs scaler as Robust scaler used earlier\n\nAbs_scaler = MaxAbsScaler().fit(data)\ndata_standard = Abs_scaler.fit_transform(data)\n\n# Transforming data with Power Transformer\npower_transformer = PowerTransformer().fit(data_standard)\ndata_transformed = power_transformer.transform(data_standard)\ndata_transformed = pd.DataFrame(data_transformed,columns = cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:46.377778Z","iopub.execute_input":"2022-07-14T11:22:46.378215Z","iopub.status.idle":"2022-07-14T11:22:50.222339Z","shell.execute_reply.started":"2022-07-14T11:22:46.378180Z","shell.execute_reply":"2022-07-14T11:22:50.221214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting distributions of individual features\n\nplt.figure(figsize=(20,28))\nfor index,col in enumerate(cols):\n    plt.subplot(8,4,index+1)\n    sns.kdeplot(x = data[col],shade=\"fill\",color=\"red\")\n    sns.kdeplot(x = data_transformed[col],shade=\"fill\",color=\"green\")   \nplt.suptitle(\"Features distribution original and scaled\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:22:50.223975Z","iopub.execute_input":"2022-07-14T11:22:50.224495Z","iopub.status.idle":"2022-07-14T11:23:19.279579Z","shell.execute_reply.started":"2022-07-14T11:22:50.224444Z","shell.execute_reply":"2022-07-14T11:23:19.278211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The features with integer valuers are originally skewed, transformed to be Gaussian like.","metadata":{}},{"cell_type":"code","source":"# Robust scaler for comparison of scaling\n\nRobust_scaler = RobustScaler().fit(data)\ndata_standard_r = Robust_scaler.fit_transform(data)\n\n# Transforming data with Power Transformer\npower_transformer_r = PowerTransformer().fit(data_standard_r)\ndata_transformed_r = power_transformer_r.transform(data_standard_r)\ndata_transformed_r = pd.DataFrame(data_transformed_r,columns = cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:23:19.281351Z","iopub.execute_input":"2022-07-14T11:23:19.281698Z","iopub.status.idle":"2022-07-14T11:23:23.513123Z","shell.execute_reply.started":"2022-07-14T11:23:19.281668Z","shell.execute_reply":"2022-07-14T11:23:23.511983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting distributions of features with Robust scaler and MaxAbs scaler\n\nplt.figure(figsize=(20,28))\nfor index,col in enumerate(cols):\n    plt.subplot(8,4,index+1)\n    sns.kdeplot(x = data_transformed[col],shade=\"fill\",color=\"red\")   \n    sns.kdeplot(x = data_transformed_r[col]) \nplt.suptitle(\"Features distribution Robust vs MaxAbs Scalers\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:23:23.514385Z","iopub.execute_input":"2022-07-14T11:23:23.514769Z","iopub.status.idle":"2022-07-14T11:23:52.522543Z","shell.execute_reply.started":"2022-07-14T11:23:23.514728Z","shell.execute_reply":"2022-07-14T11:23:52.521389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seems there is no much difference in the distibution of data after scaling with Robust scaler and MaxAbs scaler with power transformations. I am checking with MaxAbs scaler as earlier tried Robust scaler, expecting a different result","metadata":{}},{"cell_type":"markdown","source":"# Model Generation","metadata":{}},{"cell_type":"markdown","source":"Tried KMeans & gaussian earlier, the results of KMeans were not impressive, but the results of Gaussian were promising, so focusing on Gaussian & Bayesian mixture models for clustering. Checking on AIC & BIC scores for Gaussian Models.\n\nThe Akaike information criterion (AIC) & Bayesian Information Criterion (BIC) gives us an estimation on how much is good the GMM in terms of predicting the data we actually have. The lower the values, the better is the model to actually predict the data we have.\n\nNot used Silhouette scores here, as in the earlier kernels, not able to decide the size of the clusters based on Silhouette score as the scores were near to zero.","metadata":{}},{"cell_type":"code","source":"# Gaussian Mixture model over range of n clusters\ngmm_aic = []\ngmm_bic = []\nfor k in tqdm(range(5, 20)):\n    gmm = GaussianMixture(n_components=k,random_state=1,n_init=2).fit(data_transformed)\n    labels_gmm = gmm.predict(data_transformed)\n    \n    # Appending BIC values to the list\n    gmm_bic.append(gmm.bic(data_transformed))\n    gmm_aic.append(gmm.aic(data_transformed))\n\n#Gradient of BIC values\ngmm_bic_gradient = np.gradient(gmm_bic)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:23:52.524216Z","iopub.execute_input":"2022-07-14T11:23:52.524547Z","iopub.status.idle":"2022-07-14T11:52:38.847642Z","shell.execute_reply.started":"2022-07-14T11:23:52.524518Z","shell.execute_reply":"2022-07-14T11:52:38.846637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the model scores\n\nplt.figure(figsize=(12,6))\n\nplt.subplot(121)\nplt.plot(range(5,20),gmm_bic,label='bic')\nplt.xlabel(\"Cluster Size\")\nplt.ylabel(\"values\")\nplt.legend()\nplt.title(\"GMM BIC values\")\n\nplt.subplot(122)\nplt.plot(range(5,20),gmm_aic,label='aic')\nplt.xlabel(\"Cluster Size\")\nplt.ylabel(\"values\")\nplt.legend()\nplt.title(\"GMM AIC values\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:52:38.848791Z","iopub.execute_input":"2022-07-14T11:52:38.849164Z","iopub.status.idle":"2022-07-14T11:52:39.159714Z","shell.execute_reply.started":"2022-07-14T11:52:38.849131Z","shell.execute_reply":"2022-07-14T11:52:39.158579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the chart, after 7 clusters, the bic value is smooth, Seems 7/8 may be the optimal clusters. Going with 7 as earlier also 7 clusters gave good score","metadata":{}},{"cell_type":"code","source":"# Model with Gaussian mixture\nmodel = GaussianMixture(n_components=7,random_state=1,n_init=10,covariance_type='full').fit(data_transformed)\nlabels = model.predict(data_transformed)\nCounter(labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T12:11:47.892494Z","iopub.execute_input":"2022-07-14T12:11:47.893026Z","iopub.status.idle":"2022-07-14T12:14:53.140963Z","shell.execute_reply.started":"2022-07-14T12:11:47.892986Z","shell.execute_reply":"2022-07-14T12:14:53.139579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model with Bayesian Gaussian mixture model\nmodel_b = BayesianGaussianMixture(n_components=7,random_state=1,n_init=10,covariance_type='full').fit(data_transformed)\nlabels_b = model_b.predict(data_transformed)\nCounter(labels_b)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T11:52:39.161488Z","iopub.execute_input":"2022-07-14T11:52:39.161870Z","iopub.status.idle":"2022-07-14T12:03:52.650597Z","shell.execute_reply.started":"2022-07-14T11:52:39.161838Z","shell.execute_reply":"2022-07-14T12:03:52.649468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing the clusters with PCA of Bayesian Gaussian Mixture model with normalized data\npca = PCA(2).fit_transform(data_transformed)\n\nplt.figure(figsize=(12,12))\nsns.scatterplot(x=pca[:,0],y=pca[:,1],hue=labels_b,palette='tab10')\nplt.title(\"Clusters\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T12:03:52.652040Z","iopub.execute_input":"2022-07-14T12:03:52.653152Z","iopub.status.idle":"2022-07-14T12:03:56.004468Z","shell.execute_reply.started":"2022-07-14T12:03:52.653106Z","shell.execute_reply":"2022-07-14T12:03:56.003127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Submission file\n# Considering the Bayesian Gaussian mixture model labels\nsubmission[\"Predicted\"] = labels_b\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T12:03:56.006063Z","iopub.execute_input":"2022-07-14T12:03:56.007406Z","iopub.status.idle":"2022-07-14T12:03:56.012755Z","shell.execute_reply.started":"2022-07-14T12:03:56.007342Z","shell.execute_reply":"2022-07-14T12:03:56.011454Z"},"trusted":true},"execution_count":null,"outputs":[]}]}