{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport sklearn\nfrom matplotlib import pyplot\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T15:32:20.433799Z","iopub.execute_input":"2022-07-21T15:32:20.434607Z","iopub.status.idle":"2022-07-21T15:32:20.995059Z","shell.execute_reply.started":"2022-07-21T15:32:20.434475Z","shell.execute_reply":"2022-07-21T15:32:20.994063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Goal of this notebook is to TRY the clustering Algorithm from SKLEARN. ","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')\nsubmission = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:20.997109Z","iopub.execute_input":"2022-07-21T15:32:20.998011Z","iopub.status.idle":"2022-07-21T15:32:21.978598Z","shell.execute_reply.started":"2022-07-21T15:32:20.997957Z","shell.execute_reply":"2022-07-21T15:32:21.977399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **DBSCAN**\n- It is one of the ALgorithm available in *SKLEARN*. DBSCAN is short for Density-Based Spatial Clustering of Applications with Noise. \n- it is a non-parametric DENSITY Based Algorithm which group together the data poits which are close together in space.","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import DBSCAN\n# If you use the default values it cluster it in one group so I used different parameter to have a result.\nmodel = DBSCAN(eps=9.7, min_samples=3, algorithm='ball_tree', metric='minkowski', leaf_size=90, p=2)\nprediction = model.fit_predict(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:21.979805Z","iopub.execute_input":"2022-07-21T15:32:21.980138Z","iopub.status.idle":"2022-07-21T15:32:22.856951Z","shell.execute_reply.started":"2022-07-21T15:32:21.980108Z","shell.execute_reply":"2022-07-21T15:32:22.855750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.unique(prediction)\nprint(\"Clusters:\", clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:22.859790Z","iopub.execute_input":"2022-07-21T15:32:22.860401Z","iopub.status.idle":"2022-07-21T15:32:22.871361Z","shell.execute_reply.started":"2022-07-21T15:32:22.860359Z","shell.execute_reply":"2022-07-21T15:32:22.869964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# create scatter plot for samples from each cluster\nfor cluster in clusters:\n    # get row indexes for samples with this cluster\n    row_ix = np.where(prediction == cluster)\n    # create scatter of these samples\n    pyplot.scatter(data.values[row_ix, 0],data.values[row_ix, 1])\n# show the plot\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:22.873496Z","iopub.execute_input":"2022-07-21T15:32:22.874356Z","iopub.status.idle":"2022-07-21T15:32:32.647729Z","shell.execute_reply.started":"2022-07-21T15:32:22.874302Z","shell.execute_reply":"2022-07-21T15:32:32.646145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"Predicted\"] = prediction\nsubmission.to_csv(\"submission_dbscan.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:32.649854Z","iopub.execute_input":"2022-07-21T15:32:32.650757Z","iopub.status.idle":"2022-07-21T15:32:32.827509Z","shell.execute_reply.started":"2022-07-21T15:32:32.650689Z","shell.execute_reply":"2022-07-21T15:32:32.826108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-Mean Clustering\n- Create a clusters iteratively where each data point belongs to a cluster with their nearest Mean","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nmodel = KMeans()\nprediction = model.fit_predict(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:32.829898Z","iopub.execute_input":"2022-07-21T15:32:32.830460Z","iopub.status.idle":"2022-07-21T15:32:36.835076Z","shell.execute_reply.started":"2022-07-21T15:32:32.830401Z","shell.execute_reply":"2022-07-21T15:32:36.833953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.unique(prediction)\nprint(\"Clusters:\", clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:36.836449Z","iopub.execute_input":"2022-07-21T15:32:36.836799Z","iopub.status.idle":"2022-07-21T15:32:36.845295Z","shell.execute_reply.started":"2022-07-21T15:32:36.836768Z","shell.execute_reply":"2022-07-21T15:32:36.844352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create scatter plot for samples from each cluster\nfor cluster in clusters:\n    # get row indexes for samples with this cluster\n    row_ix = np.where(prediction == cluster)\n    # create scatter of these samples\n    pyplot.scatter(data.values[row_ix, 0],data.values[row_ix, 1])\n# show the plot\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:36.846695Z","iopub.execute_input":"2022-07-21T15:32:36.847528Z","iopub.status.idle":"2022-07-21T15:32:37.361212Z","shell.execute_reply.started":"2022-07-21T15:32:36.847486Z","shell.execute_reply":"2022-07-21T15:32:37.360076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"Predicted\"] = prediction\nsubmission.to_csv(\"submission_kmean.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:37.363080Z","iopub.execute_input":"2022-07-21T15:32:37.363465Z","iopub.status.idle":"2022-07-21T15:32:37.525650Z","shell.execute_reply.started":"2022-07-21T15:32:37.363429Z","shell.execute_reply":"2022-07-21T15:32:37.524323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mini-Batch K-Means\n\n- Mini-Batch K-Means is a modified version of k-means that makes updates to the cluster centroids using mini-batches of samples rather than the entire dataset, which can make it faster for large datasets, and perhaps more robust to statistical noise.([MachineLearningMastery](https://machinelearningmastery.com/clustering-algorithms-with-python/))","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import MiniBatchKMeans\nmodel = MiniBatchKMeans(n_clusters=50)\nprediction = model.fit_predict(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:37.527151Z","iopub.execute_input":"2022-07-21T15:32:37.527644Z","iopub.status.idle":"2022-07-21T15:32:38.201808Z","shell.execute_reply.started":"2022-07-21T15:32:37.527607Z","shell.execute_reply":"2022-07-21T15:32:38.200320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.unique(prediction)\nprint(\"Clusters:\", clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:38.206724Z","iopub.execute_input":"2022-07-21T15:32:38.210953Z","iopub.status.idle":"2022-07-21T15:32:38.222651Z","shell.execute_reply.started":"2022-07-21T15:32:38.210895Z","shell.execute_reply":"2022-07-21T15:32:38.221477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create scatter plot for samples from each cluster\nfor cluster in clusters:\n    # get row indexes for samples with this cluster\n    row_ix = np.where(prediction == cluster)\n    # create scatter of these samples\n    pyplot.scatter(data.values[row_ix, 0],data.values[row_ix, 1])\n# show the plot\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:38.228564Z","iopub.execute_input":"2022-07-21T15:32:38.228950Z","iopub.status.idle":"2022-07-21T15:32:39.627266Z","shell.execute_reply.started":"2022-07-21T15:32:38.228917Z","shell.execute_reply":"2022-07-21T15:32:39.626022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"Predicted\"] = prediction\nsubmission.to_csv(\"submissionMiniKmeans.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:39.628781Z","iopub.execute_input":"2022-07-21T15:32:39.629154Z","iopub.status.idle":"2022-07-21T15:32:39.805708Z","shell.execute_reply.started":"2022-07-21T15:32:39.629118Z","shell.execute_reply":"2022-07-21T15:32:39.804044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gaussian Mixture\n- Gaussian mixture models are a probabilistic model for representing normally distributed subpopulations within an overall population. Mixture models in general don't require knowing which subpopulation a data point belongs to, allowing the model to learn the subpopulations automatically. Since subpopulation assignment is not known, this constitutes a form of unsupervised learning.","metadata":{}},{"cell_type":"code","source":"# After version:13\nfrom sklearn.preprocessing import StandardScaler\ndel data['id']\n\nstand= StandardScaler()\nFit_Transform = stand.fit_transform(data.values)\ndata = pd.DataFrame(Fit_Transform, columns=data.columns)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:39.808006Z","iopub.execute_input":"2022-07-21T15:32:39.808472Z","iopub.status.idle":"2022-07-21T15:32:39.895545Z","shell.execute_reply.started":"2022-07-21T15:32:39.808426Z","shell.execute_reply":"2022-07-21T15:32:39.893891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.mixture import GaussianMixture\nmodel = GaussianMixture(n_components=7)\nprediction = model.fit_predict(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:39.897600Z","iopub.execute_input":"2022-07-21T15:32:39.898537Z","iopub.status.idle":"2022-07-21T15:32:55.397461Z","shell.execute_reply.started":"2022-07-21T15:32:39.898475Z","shell.execute_reply":"2022-07-21T15:32:55.395845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.unique(prediction)\nprint(\"Clusters:\", clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:55.399275Z","iopub.execute_input":"2022-07-21T15:32:55.400146Z","iopub.status.idle":"2022-07-21T15:32:55.412295Z","shell.execute_reply.started":"2022-07-21T15:32:55.400089Z","shell.execute_reply":"2022-07-21T15:32:55.411030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create scatter plot for samples from each cluster\nfor cluster in clusters:\n    # get row indexes for samples with this cluster\n    row_ix = np.where(prediction == cluster)\n    # create scatter of these samples\n    pyplot.scatter(data.values[row_ix, 0],data.values[row_ix, 1])\n# show the plot\npyplot.show()\n\nsubmission[\"Predicted\"] = prediction\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:55.414223Z","iopub.execute_input":"2022-07-21T15:32:55.414997Z","iopub.status.idle":"2022-07-21T15:32:56.006043Z","shell.execute_reply.started":"2022-07-21T15:32:55.414947Z","shell.execute_reply":"2022-07-21T15:32:56.004990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bayesian Gaussian Mixture\n- Bayesian Gaussian mixture models constitutes a form of unsupervised learning","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture\nmodel = BayesianGaussianMixture(n_components=7)#weight_concentration_prior_type='dirichlet_distribution'\nprediction = model.fit_predict(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:32:56.007400Z","iopub.execute_input":"2022-07-21T15:32:56.007974Z","iopub.status.idle":"2022-07-21T15:34:04.813299Z","shell.execute_reply.started":"2022-07-21T15:32:56.007937Z","shell.execute_reply":"2022-07-21T15:34:04.811810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.unique(prediction)\nprint(\"Clusters:\", clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:34:04.815636Z","iopub.execute_input":"2022-07-21T15:34:04.816623Z","iopub.status.idle":"2022-07-21T15:34:04.830511Z","shell.execute_reply.started":"2022-07-21T15:34:04.816563Z","shell.execute_reply":"2022-07-21T15:34:04.829132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create scatter plot for samples from each cluster\nfor cluster in clusters:\n    # get row indexes for samples with this cluster\n    row_ix = np.where(prediction == cluster)\n    # create scatter of these samples\n    pyplot.scatter(data.values[row_ix, 0],data.values[row_ix, 1])\n# show the plot\npyplot.show()\n\nsubmission[\"Predicted\"] = prediction\nsubmission.to_csv(\"submissionBGM.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:34:04.833047Z","iopub.execute_input":"2022-07-21T15:34:04.833919Z","iopub.status.idle":"2022-07-21T15:34:05.421232Z","shell.execute_reply.started":"2022-07-21T15:34:04.833859Z","shell.execute_reply":"2022-07-21T15:34:05.419623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"NOW we RUN most of the Algorithm on the raw data. Let scale and Normalize our data and we will see that it will increase the score.","metadata":{}}]}