{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T01:52:23.056226Z","iopub.execute_input":"2022-07-13T01:52:23.057291Z","iopub.status.idle":"2022-07-13T01:52:23.066700Z","shell.execute_reply.started":"2022-07-13T01:52:23.057227Z","shell.execute_reply":"2022-07-13T01:52:23.065525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Import data\nimport pandas as pd\ndf=pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/data.csv\")\nss=pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T01:52:31.044019Z","iopub.execute_input":"2022-07-13T01:52:31.044398Z","iopub.status.idle":"2022-07-13T01:52:31.858125Z","shell.execute_reply.started":"2022-07-13T01:52:31.044369Z","shell.execute_reply":"2022-07-13T01:52:31.856895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check for null value\ndf.isnull().sum()\n# df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T01:53:01.209178Z","iopub.execute_input":"2022-07-13T01:53:01.210248Z","iopub.status.idle":"2022-07-13T01:53:01.227688Z","shell.execute_reply.started":"2022-07-13T01:53:01.210205Z","shell.execute_reply":"2022-07-13T01:53:01.226700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T01:53:58.153118Z","iopub.execute_input":"2022-07-13T01:53:58.154005Z","iopub.status.idle":"2022-07-13T01:53:58.262140Z","shell.execute_reply.started":"2022-07-13T01:53:58.153970Z","shell.execute_reply":"2022-07-13T01:53:58.261113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T02:00:41.248193Z","iopub.execute_input":"2022-07-13T02:00:41.248599Z","iopub.status.idle":"2022-07-13T02:00:41.545068Z","shell.execute_reply.started":"2022-07-13T02:00:41.248567Z","shell.execute_reply":"2022-07-13T02:00:41.543881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import plotting function\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n#change the figure size of the heatmap by inch\nsns.set(rc={'figure.figsize':(24,20)})\n#fmtstr, optional - String formatting code to use when adding annotations.\n#annot - write the data value in each cell. If an array-like with the same shape\nsns.heatmap(df.corr(),annot=True,fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:18:48.000928Z","iopub.execute_input":"2022-07-12T09:18:48.001496Z","iopub.status.idle":"2022-07-12T09:18:52.051933Z","shell.execute_reply.started":"2022-07-12T09:18:48.001449Z","shell.execute_reply":"2022-07-12T09:18:52.050444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#enumerate(iterable, start=0)\nenumerate(list(df.columns),1)\ndf.sample(1000)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:41:01.038195Z","iopub.execute_input":"2022-07-12T09:41:01.038740Z","iopub.status.idle":"2022-07-12T09:41:01.095859Z","shell.execute_reply.started":"2022-07-12T09:41:01.038694Z","shell.execute_reply":"2022-07-12T09:41:01.093861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(15,15)})\n#to see our randome sample's distribution\nfor i, column in enumerate(list(df.columns), 1):\n#     create empty canvas\n    plt.subplot(5,6,i)\n#     df.samle() = returns a list with a randomly selection of a specified number of items from a sequnce\n# kde = kernel density line\n    p=sns.histplot(x=column,data=df.sample(1000),stat='count',kde=True,color='green')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:42:14.192397Z","iopub.execute_input":"2022-07-12T09:42:14.192813Z","iopub.status.idle":"2022-07-12T09:42:19.940496Z","shell.execute_reply.started":"2022-07-12T09:42:14.192779Z","shell.execute_reply":"2022-07-12T09:42:19.939400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Python program to illustrate\n# enumerate function\nl1 = [\"eat\", \"sleep\", \"repeat\"]\ns1 = \"geek\"\n  \n# creating enumerate objects\nobj1 = enumerate(l1)\nobj2 = enumerate(s1)\n  \nprint (\"Return type:\", type(obj1))\nprint (list(enumerate(l1)))\n  \n# changing start index to 2 from 0\nprint (list(enumerate(s1, 2)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:37:24.445245Z","iopub.execute_input":"2022-07-12T09:37:24.445633Z","iopub.status.idle":"2022-07-12T09:37:24.453359Z","shell.execute_reply.started":"2022-07-12T09:37:24.445601Z","shell.execute_reply":"2022-07-12T09:37:24.451956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nr=RobustScaler()\ndf_r=r.fit_transform(df)\ndf_r\ndf_r=pd.DataFrame(df_r, columns = df.columns )\ndf_r\n# df_r.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:57:03.464269Z","iopub.execute_input":"2022-07-12T13:57:03.464780Z","iopub.status.idle":"2022-07-12T13:57:03.714058Z","shell.execute_reply.started":"2022-07-12T13:57:03.464739Z","shell.execute_reply":"2022-07-12T13:57:03.712447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets try k-Means\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.cluster import KMeans\nfrom numpy import unique\nfrom numpy import where\nfrom matplotlib import pyplot\nfrom sklearn.datasets import make_classification\n\n#randomstate: guarantee the same sandom number output\nElbow_M = KElbowVisualizer(KMeans(random_state=23), k=(4,12))\nElbow_M.fit(df_r)\nElbow_M.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:00:29.978380Z","iopub.execute_input":"2022-07-12T14:00:29.979448Z","iopub.status.idle":"2022-07-12T14:01:28.803465Z","shell.execute_reply.started":"2022-07-12T14:00:29.979391Z","shell.execute_reply":"2022-07-12T14:01:28.802282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define the model\nkmeans_model = KMeans(n_clusters=7)\n\n# assign each data point to a cluster\nkmeans_result = kmeans_model.fit_predict(df_r)\n\n# get all of the unique clusters\nkmeans_clusters = unique(kmeans_result)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:19:36.553871Z","iopub.execute_input":"2022-07-12T14:19:36.554294Z","iopub.status.idle":"2022-07-12T14:19:43.564491Z","shell.execute_reply.started":"2022-07-12T14:19:36.554261Z","shell.execute_reply":"2022-07-12T14:19:43.563383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(kmeans_result)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:20:18.608906Z","iopub.execute_input":"2022-07-12T14:20:18.609745Z","iopub.status.idle":"2022-07-12T14:20:18.618485Z","shell.execute_reply.started":"2022-07-12T14:20:18.609705Z","shell.execute_reply":"2022-07-12T14:20:18.617002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Predicted']=kmeans_result","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:21:20.019189Z","iopub.execute_input":"2022-07-12T14:21:20.019712Z","iopub.status.idle":"2022-07-12T14:21:20.027638Z","shell.execute_reply.started":"2022-07-12T14:21:20.019670Z","shell.execute_reply":"2022-07-12T14:21:20.026095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[:,['id','Predicted']].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T14:21:29.334403Z","iopub.execute_input":"2022-07-12T14:21:29.335219Z","iopub.status.idle":"2022-07-12T14:21:29.351783Z","shell.execute_reply.started":"2022-07-12T14:21:29.335176Z","shell.execute_reply":"2022-07-12T14:21:29.350390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[:,['id','Predicted']].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:25:33.711876Z","iopub.execute_input":"2022-07-12T10:25:33.712265Z","iopub.status.idle":"2022-07-12T10:25:33.724696Z","shell.execute_reply.started":"2022-07-12T10:25:33.712232Z","shell.execute_reply":"2022-07-12T10:25:33.723463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub=df.loc[:,['id','Predicted']]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:25:35.802473Z","iopub.execute_input":"2022-07-12T10:25:35.802866Z","iopub.status.idle":"2022-07-12T10:25:35.810879Z","shell.execute_reply.started":"2022-07-12T10:25:35.802835Z","shell.execute_reply":"2022-07-12T10:25:35.809615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:25:56.583046Z","iopub.execute_input":"2022-07-12T10:25:56.583479Z","iopub.status.idle":"2022-07-12T10:25:56.590757Z","shell.execute_reply.started":"2022-07-12T10:25:56.583430Z","shell.execute_reply":"2022-07-12T10:25:56.589614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('sub_july.csv',index=False)","metadata":{},"execution_count":null,"outputs":[]}]}