{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nprint('Bismillahir Rohmanir Rohiym!')\n\n# data visualisation and manipulation\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport datatable as dt\nimport gc\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T01:29:51.43568Z","iopub.execute_input":"2022-07-06T01:29:51.436064Z","iopub.status.idle":"2022-07-06T01:29:52.595452Z","shell.execute_reply.started":"2022-07-06T01:29:51.435972Z","shell.execute_reply":"2022-07-06T01:29:52.594637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import clear_output\n!pip install pycaret --user\nclear_output()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-06T01:29:52.596921Z","iopub.execute_input":"2022-07-06T01:29:52.597229Z","iopub.status.idle":"2022-07-06T01:30:28.894534Z","shell.execute_reply.started":"2022-07-06T01:29:52.597197Z","shell.execute_reply":"2022-07-06T01:30:28.893222Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pycaret clustering library\n\nfrom pycaret.clustering import *","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:30:28.89635Z","iopub.execute_input":"2022-07-06T01:30:28.896598Z","iopub.status.idle":"2022-07-06T01:30:33.249047Z","shell.execute_reply.started":"2022-07-06T01:30:28.896572Z","shell.execute_reply":"2022-07-06T01:30:33.247895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data\n\ndata = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\nprint(data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:30:33.25138Z","iopub.execute_input":"2022-07-06T01:30:33.251804Z","iopub.status.idle":"2022-07-06T01:30:34.491116Z","shell.execute_reply.started":"2022-07-06T01:30:33.251755Z","shell.execute_reply":"2022-07-06T01:30:34.490196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data = data.sample(frac=0.15)\n#print(data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:30:34.49221Z","iopub.execute_input":"2022-07-06T01:30:34.492498Z","iopub.status.idle":"2022-07-06T01:30:34.496172Z","shell.execute_reply.started":"2022-07-06T01:30:34.492468Z","shell.execute_reply":"2022-07-06T01:30:34.495272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# you can change settings here\n\nsetup(data, # show the data name\n      normalize = True, \n      normalize_method = 'robust', #normalization method\n      transformation=True, #When set to True, applies power transform to make data more Gaussian-like\n      ignore_features = ['id'], # ignore some not useful features\n      session_id = 123,\n      silent=True,) #profile=True","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-06T01:30:34.497201Z","iopub.execute_input":"2022-07-06T01:30:34.497432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# models which pycaret library provides\n\nmodels()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# choose from above difference models\n\nkmeans = create_model('kmeans',num_clusters = 6)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T07:43:32.443627Z","iopub.execute_input":"2022-07-03T07:43:32.445591Z","iopub.status.idle":"2022-07-03T07:46:30.006075Z","shell.execute_reply.started":"2022-07-03T07:43:32.4455Z","shell.execute_reply":"2022-07-03T07:46:30.005368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(kmeans)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:20:28.340155Z","iopub.execute_input":"2022-07-02T05:20:28.340464Z","iopub.status.idle":"2022-07-02T05:20:28.344982Z","shell.execute_reply.started":"2022-07-02T05:20:28.340406Z","shell.execute_reply":"2022-07-02T05:20:28.344396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# assign clustering results\n\nkmean_results = assign_model(kmeans)\nkmean_results.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:20:29.77595Z","iopub.execute_input":"2022-07-02T05:20:29.77668Z","iopub.status.idle":"2022-07-02T05:20:29.810664Z","shell.execute_reply.started":"2022-07-02T05:20:29.776628Z","shell.execute_reply":"2022-07-02T05:20:29.809842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot clustering model\n\nplot_model(kmeans)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:20:30.995481Z","iopub.execute_input":"2022-07-02T05:20:30.995783Z","iopub.status.idle":"2022-07-02T05:20:32.482082Z","shell.execute_reply.started":"2022-07-02T05:20:30.99575Z","shell.execute_reply":"2022-07-02T05:20:32.481397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# ELBOW PLOT to DETECT BEST number of clusters\n\nplot_model(kmeans, plot = 'elbow')","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:20:40.231929Z","iopub.execute_input":"2022-07-02T05:20:40.232401Z","iopub.status.idle":"2022-07-02T05:20:59.284363Z","shell.execute_reply.started":"2022-07-02T05:20:40.23235Z","shell.execute_reply":"2022-07-02T05:20:59.283487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# TSNE  - PLOT\n\nplot_model(kmeans,plot='tsne')","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:25:15.875439Z","iopub.execute_input":"2022-07-02T05:25:15.875724Z","iopub.status.idle":"2022-07-02T05:26:07.767969Z","shell.execute_reply.started":"2022-07-02T05:25:15.87569Z","shell.execute_reply":"2022-07-02T05:26:07.767195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this is silhouette plot\n\nplot_model(kmeans, plot = 'silhouette')","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:25:01.242562Z","iopub.execute_input":"2022-07-02T05:25:01.242879Z","iopub.status.idle":"2022-07-02T05:25:05.06191Z","shell.execute_reply.started":"2022-07-02T05:25:01.24285Z","shell.execute_reply":"2022-07-02T05:25:05.060919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot distribution to see size of clusters\n\nplot_model(kmeans, plot = 'distribution') ","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:25:10.2076Z","iopub.execute_input":"2022-07-02T05:25:10.208023Z","iopub.status.idle":"2022-07-02T05:25:10.826977Z","shell.execute_reply.started":"2022-07-02T05:25:10.207992Z","shell.execute_reply":"2022-07-02T05:25:10.825834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract cluster numbers from Cluster column\n\nkmean_results['Predicted'] = kmean_results['Cluster'].str.extract('(\\d+)', expand=False)\nkmean_results.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:23:43.468911Z","iopub.execute_input":"2022-07-02T05:23:43.469197Z","iopub.status.idle":"2022-07-02T05:23:43.510794Z","shell.execute_reply.started":"2022-07-02T05:23:43.469167Z","shell.execute_reply":"2022-07-02T05:23:43.509871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submit clustering results\n\nsample_sub = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\n\nsample_sub['Predicted'] = kmean_results['Predicted']\nsample_sub.to_csv('submission.csv',index=0)\nsample_sub","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:26:07.77181Z","iopub.execute_input":"2022-07-02T05:26:07.77394Z","iopub.status.idle":"2022-07-02T05:26:07.995194Z","shell.execute_reply.started":"2022-07-02T05:26:07.773892Z","shell.execute_reply":"2022-07-02T05:26:07.994488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Source for Further Learning:\nhttps://github.com/pycaret/pycaret/blob/master/tutorials/Clustering%20Tutorial%20Level%20Beginner%20-%20CLU101.ipynb","metadata":{"execution":{"iopub.status.busy":"2022-07-02T05:20:08.686839Z","iopub.status.idle":"2022-07-02T05:20:08.68747Z","shell.execute_reply.started":"2022-07-02T05:20:08.68716Z","shell.execute_reply":"2022-07-02T05:20:08.687189Z"}}},{"cell_type":"markdown","source":"### Thank You for Reading!","metadata":{}}]}