{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T18:17:27.195560Z","iopub.execute_input":"2022-08-01T18:17:27.196808Z","iopub.status.idle":"2022-08-01T18:17:28.401832Z","shell.execute_reply.started":"2022-08-01T18:17:27.196678Z","shell.execute_reply":"2022-08-01T18:17:28.400214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:31.484891Z","iopub.execute_input":"2022-08-01T18:17:31.485207Z","iopub.status.idle":"2022-08-01T18:17:32.467904Z","shell.execute_reply.started":"2022-08-01T18:17:31.485183Z","shell.execute_reply":"2022-08-01T18:17:32.466883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:42.255311Z","iopub.execute_input":"2022-08-01T17:12:42.256383Z","iopub.status.idle":"2022-08-01T17:12:42.297399Z","shell.execute_reply.started":"2022-08-01T17:12:42.256333Z","shell.execute_reply":"2022-08-01T17:12:42.296220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:42.300517Z","iopub.execute_input":"2022-08-01T17:12:42.301569Z","iopub.status.idle":"2022-08-01T17:12:42.337918Z","shell.execute_reply.started":"2022-08-01T17:12:42.301518Z","shell.execute_reply":"2022-08-01T17:12:42.336460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#no missing values\ndata.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:42.339644Z","iopub.execute_input":"2022-08-01T17:12:42.339997Z","iopub.status.idle":"2022-08-01T17:12:42.356927Z","shell.execute_reply.started":"2022-08-01T17:12:42.339965Z","shell.execute_reply":"2022-08-01T17:12:42.355915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#no duplicated rows\ndata.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:42.358448Z","iopub.execute_input":"2022-08-01T17:12:42.359363Z","iopub.status.idle":"2022-08-01T17:12:42.630913Z","shell.execute_reply.started":"2022-08-01T17:12:42.359328Z","shell.execute_reply":"2022-08-01T17:12:42.629647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop(\"id\",axis=1).describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:42.632869Z","iopub.execute_input":"2022-08-01T17:12:42.633684Z","iopub.status.idle":"2022-08-01T17:12:42.885124Z","shell.execute_reply.started":"2022-08-01T17:12:42.633629Z","shell.execute_reply":"2022-08-01T17:12:42.883576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,15))\nfor i in range(1,30): \n    plt.subplot(5, 6, i) \n    plt.boxplot(data.iloc[:,i]) \n    plt.title(data.columns[i])\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:42.886762Z","iopub.execute_input":"2022-08-01T17:12:42.887253Z","iopub.status.idle":"2022-08-01T17:12:46.459819Z","shell.execute_reply.started":"2022-08-01T17:12:42.887207Z","shell.execute_reply":"2022-08-01T17:12:46.458623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Every feature has outliers in boxplots.f7-8-9-10-11 features have outliers greater than upper limits \n\nInterval range of features are different from each other, som features are between 0-30, some are between -4,4. Features looks like being scaled, but some features might be left out or normalized in different ranges. Normalization is needed\n","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,15))\nfor i in range(1,30): \n    plt.subplot(5, 6, i) \n    plt.hist(data.iloc[:,i]) \n    plt.title(data.columns[i])\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:46.461102Z","iopub.execute_input":"2022-08-01T17:12:46.461539Z","iopub.status.idle":"2022-08-01T17:12:50.793890Z","shell.execute_reply.started":"2022-08-01T17:12:46.461501Z","shell.execute_reply":"2022-08-01T17:12:50.792548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"f7-8-9-10-11-12-13 features histplots gives right-skewed distributions. Other features looks like normal distributions.","metadata":{}},{"cell_type":"markdown","source":"# Feature Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:40.991601Z","iopub.execute_input":"2022-08-01T18:17:40.991986Z","iopub.status.idle":"2022-08-01T18:17:41.051895Z","shell.execute_reply.started":"2022-08-01T18:17:40.991957Z","shell.execute_reply":"2022-08-01T18:17:41.050735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#copy the original data\nX = data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:42.955893Z","iopub.execute_input":"2022-08-01T18:17:42.956243Z","iopub.status.idle":"2022-08-01T18:17:42.963098Z","shell.execute_reply.started":"2022-08-01T18:17:42.956214Z","shell.execute_reply":"2022-08-01T18:17:42.961969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop id column and assign id as index\nX.index = data[\"id\"]\nX = X.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:44.763166Z","iopub.execute_input":"2022-08-01T18:17:44.763537Z","iopub.status.idle":"2022-08-01T18:17:44.784264Z","shell.execute_reply.started":"2022-08-01T18:17:44.763508Z","shell.execute_reply":"2022-08-01T18:17:44.783263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Standartize the X\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:46.176847Z","iopub.execute_input":"2022-08-01T18:17:46.177300Z","iopub.status.idle":"2022-08-01T18:17:46.215349Z","shell.execute_reply.started":"2022-08-01T18:17:46.177265Z","shell.execute_reply":"2022-08-01T18:17:46.214436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ML Applications","metadata":{}},{"cell_type":"markdown","source":"# K-Means","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:50.993834Z","iopub.execute_input":"2022-08-01T17:12:50.994598Z","iopub.status.idle":"2022-08-01T17:12:51.307831Z","shell.execute_reply.started":"2022-08-01T17:12:50.994563Z","shell.execute_reply":"2022-08-01T17:12:51.306513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Determine number of clusters","metadata":{}},{"cell_type":"markdown","source":"## Elbow Method","metadata":{}},{"cell_type":"markdown","source":"KMeans with original Data","metadata":{}},{"cell_type":"code","source":"kmeans_per_k = [KMeans(n_clusters=k, random_state=42).fit(X)\n                for k in range(1, 10)]\ninertias = [model.inertia_ for model in kmeans_per_k]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:12:51.309288Z","iopub.execute_input":"2022-08-01T17:12:51.309635Z","iopub.status.idle":"2022-08-01T17:13:34.819526Z","shell.execute_reply.started":"2022-08-01T17:12:51.309603Z","shell.execute_reply":"2022-08-01T17:13:34.818247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 3.5))\nplt.plot(range(1, 10), inertias, \"bo-\")\nplt.xlabel(\"$k$\", fontsize=14)\nplt.ylabel(\"Inertia\", fontsize=14)\nplt.title(\"Find optimum k with elbow method\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:13:34.821852Z","iopub.execute_input":"2022-08-01T17:13:34.822387Z","iopub.status.idle":"2022-08-01T17:13:35.279913Z","shell.execute_reply.started":"2022-08-01T17:13:34.822336Z","shell.execute_reply":"2022-08-01T17:13:35.278541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Elbow method is used for determining the number of clusters in KMeans. The plot shows the inertia values for cluester values between 1-10. Optimum number of clusters is 3 or 4. Rate of inertia decrease is drops significantly these values.","metadata":{}},{"cell_type":"markdown","source":"KMeans with scaled Data","metadata":{}},{"cell_type":"code","source":"kmeans_per_k_s = [KMeans(n_clusters=k, random_state=42).fit(X_scaled)\n                for k in range(1, 10)]\ninertias_s = [model.inertia_ for model in kmeans_per_k_s]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:13:35.281369Z","iopub.execute_input":"2022-08-01T17:13:35.281745Z","iopub.status.idle":"2022-08-01T17:14:33.662032Z","shell.execute_reply.started":"2022-08-01T17:13:35.281713Z","shell.execute_reply":"2022-08-01T17:14:33.660611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 3.5))\nplt.plot(range(1, 10), inertias_s, \"bo-\")\nplt.xlabel(\"$k$\", fontsize=14)\nplt.ylabel(\"Inertia\", fontsize=14)\nplt.title(\"Find optimum k with elbow method (data is scaled)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:14:33.663800Z","iopub.execute_input":"2022-08-01T17:14:33.664325Z","iopub.status.idle":"2022-08-01T17:14:33.906181Z","shell.execute_reply.started":"2022-08-01T17:14:33.664274Z","shell.execute_reply":"2022-08-01T17:14:33.904939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I plot the inertias for scaled data above, in this graph 3 or 4 is more likely to be the number of cluster. Silhoutte Method can be used the determine the correct number of clusters.","metadata":{}},{"cell_type":"markdown","source":"#### Fit the Model","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(n_clusters=3,n_init=5,random_state=42)\nkmeans.fit(X_scaled)\nk_labels = kmeans.labels_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:22:31.506904Z","iopub.execute_input":"2022-08-01T17:22:31.507397Z","iopub.status.idle":"2022-08-01T17:22:33.163342Z","shell.execute_reply.started":"2022-08-01T17:22:31.507350Z","shell.execute_reply":"2022-08-01T17:22:33.161994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k_labels[0:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:22:33.165486Z","iopub.execute_input":"2022-08-01T17:22:33.166283Z","iopub.status.idle":"2022-08-01T17:22:33.173517Z","shell.execute_reply.started":"2022-08-01T17:22:33.166244Z","shell.execute_reply":"2022-08-01T17:22:33.172435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gaussian mixture models","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import GaussianMixture","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:22:33.175051Z","iopub.execute_input":"2022-08-01T17:22:33.176038Z","iopub.status.idle":"2022-08-01T17:22:33.194062Z","shell.execute_reply.started":"2022-08-01T17:22:33.176005Z","shell.execute_reply":"2022-08-01T17:22:33.192857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gm = GaussianMixture(n_components=3,n_init=10,random_state=42)\ngm.fit(X_scaled)\ngm_labels = gm.predict(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:22:33.195648Z","iopub.execute_input":"2022-08-01T17:22:33.196314Z","iopub.status.idle":"2022-08-01T17:23:17.171593Z","shell.execute_reply.started":"2022-08-01T17:22:33.196279Z","shell.execute_reply":"2022-08-01T17:23:17.170095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gm.converged_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:23:17.179503Z","iopub.execute_input":"2022-08-01T17:23:17.184503Z","iopub.status.idle":"2022-08-01T17:23:17.200527Z","shell.execute_reply.started":"2022-08-01T17:23:17.184417Z","shell.execute_reply":"2022-08-01T17:23:17.198888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gm.n_iter_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"The model is converged after {gm.n_iter_} iterations\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gm_labels[0:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:23:17.206838Z","iopub.execute_input":"2022-08-01T17:23:17.211737Z","iopub.status.idle":"2022-08-01T17:23:17.226615Z","shell.execute_reply.started":"2022-08-01T17:23:17.211664Z","shell.execute_reply":"2022-08-01T17:23:17.225271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bayesian Gaussian Mixture Models","metadata":{}},{"cell_type":"markdown","source":"Bayesian Gaussian Mixture Models have the advantage of not classifying the number of clusters. The model determines the number of clusters itself.","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:54.954888Z","iopub.execute_input":"2022-08-01T18:17:54.955241Z","iopub.status.idle":"2022-08-01T18:17:55.226526Z","shell.execute_reply.started":"2022-08-01T18:17:54.955217Z","shell.execute_reply":"2022-08-01T18:17:55.225706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm = BayesianGaussianMixture(n_components=10, n_init=10, random_state=42, max_iter=200)\nbgm.fit(X)\nnp.round(bgm.weights_, 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:17:56.886322Z","iopub.execute_input":"2022-08-01T18:17:56.886702Z","iopub.status.idle":"2022-08-01T18:36:37.392477Z","shell.execute_reply.started":"2022-08-01T18:17:56.886674Z","shell.execute_reply":"2022-08-01T18:36:37.391605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"There are {len(set(np.round(bgm.weights_, 2)))} clusters, but in k-means we have found 3 clusters!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:37:55.180844Z","iopub.execute_input":"2022-08-01T18:37:55.181199Z","iopub.status.idle":"2022-08-01T18:37:55.186832Z","shell.execute_reply.started":"2022-08-01T18:37:55.181171Z","shell.execute_reply":"2022-08-01T18:37:55.185434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm_labels = bgm.predict(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:38:00.669757Z","iopub.execute_input":"2022-08-01T18:38:00.670200Z","iopub.status.idle":"2022-08-01T18:38:00.976959Z","shell.execute_reply.started":"2022-08-01T18:38:00.670170Z","shell.execute_reply":"2022-08-01T18:38:00.976072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create Submission File","metadata":{}},{"cell_type":"code","source":"#add clustering labels\nX = pd.DataFrame(X,columns=data.drop(\"id\",axis=1).columns)\nX[\"Predicted\"] = bgm_labels","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:38:03.203370Z","iopub.execute_input":"2022-08-01T18:38:03.203704Z","iopub.status.idle":"2022-08-01T18:38:03.218036Z","shell.execute_reply.started":"2022-08-01T18:38:03.203678Z","shell.execute_reply":"2022-08-01T18:38:03.216520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission = pd.DataFrame({'Id': X.index, 'Predicted': X.Predicted})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:38:06.073558Z","iopub.execute_input":"2022-08-01T18:38:06.074358Z","iopub.status.idle":"2022-08-01T18:38:06.080601Z","shell.execute_reply.started":"2022-08-01T18:38:06.074322Z","shell.execute_reply":"2022-08-01T18:38:06.079319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" my_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:38:07.894526Z","iopub.execute_input":"2022-08-01T18:38:07.894854Z","iopub.status.idle":"2022-08-01T18:38:08.053003Z","shell.execute_reply.started":"2022-08-01T18:38:07.894830Z","shell.execute_reply":"2022-08-01T18:38:08.051873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:38:09.725103Z","iopub.execute_input":"2022-08-01T18:38:09.725456Z","iopub.status.idle":"2022-08-01T18:38:09.740744Z","shell.execute_reply.started":"2022-08-01T18:38:09.725431Z","shell.execute_reply":"2022-08-01T18:38:09.739435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission.Predicted.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:38:12.226663Z","iopub.execute_input":"2022-08-01T18:38:12.227584Z","iopub.status.idle":"2022-08-01T18:38:12.240148Z","shell.execute_reply.started":"2022-08-01T18:38:12.227557Z","shell.execute_reply":"2022-08-01T18:38:12.239187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References","metadata":{}},{"cell_type":"markdown","source":"* Hands-on Machine Learning with Scikit-Learn, Keras, and TensorFlow\n* https://colab.research.google.com/github/ageron/handson-ml2/blob/master/09_unsupervised_learning.ipynb#scrollTo=oCO-KVCvmqJ5","metadata":{}}]}