{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing libraries for the required dataset","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:10.909132Z","iopub.execute_input":"2022-07-10T23:17:10.909545Z","iopub.status.idle":"2022-07-10T23:17:10.915636Z","shell.execute_reply.started":"2022-07-10T23:17:10.909510Z","shell.execute_reply":"2022-07-10T23:17:10.914465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reading dataset","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\nsubmission=pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:10.917609Z","iopub.execute_input":"2022-07-10T23:17:10.918296Z","iopub.status.idle":"2022-07-10T23:17:12.067351Z","shell.execute_reply.started":"2022-07-10T23:17:10.918263Z","shell.execute_reply":"2022-07-10T23:17:12.066275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function is used to display top 5 rows of the dataset.\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.069200Z","iopub.execute_input":"2022-07-10T23:17:12.069755Z","iopub.status.idle":"2022-07-10T23:17:12.097365Z","shell.execute_reply.started":"2022-07-10T23:17:12.069720Z","shell.execute_reply":"2022-07-10T23:17:12.095804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function is used to display Dtype, Non-Null Values and count of the dataset.\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.099439Z","iopub.execute_input":"2022-07-10T23:17:12.099996Z","iopub.status.idle":"2022-07-10T23:17:12.125976Z","shell.execute_reply.started":"2022-07-10T23:17:12.099954Z","shell.execute_reply":"2022-07-10T23:17:12.124749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# It displays the shape of the dataset\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.128845Z","iopub.execute_input":"2022-07-10T23:17:12.129230Z","iopub.status.idle":"2022-07-10T23:17:12.138541Z","shell.execute_reply.started":"2022-07-10T23:17:12.129197Z","shell.execute_reply":"2022-07-10T23:17:12.137239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# It displays whether null values are there or not in dataset.\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.140880Z","iopub.execute_input":"2022-07-10T23:17:12.141988Z","iopub.status.idle":"2022-07-10T23:17:12.162302Z","shell.execute_reply.started":"2022-07-10T23:17:12.141932Z","shell.execute_reply":"2022-07-10T23:17:12.160767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# It describe the statistical summary dataset\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.163862Z","iopub.execute_input":"2022-07-10T23:17:12.164497Z","iopub.status.idle":"2022-07-10T23:17:12.385787Z","shell.execute_reply.started":"2022-07-10T23:17:12.164459Z","shell.execute_reply":"2022-07-10T23:17:12.384513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Id column is dropped from the dataset\ndf.drop(['id'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.387376Z","iopub.execute_input":"2022-07-10T23:17:12.388018Z","iopub.status.idle":"2022-07-10T23:17:12.400908Z","shell.execute_reply.started":"2022-07-10T23:17:12.387983Z","shell.execute_reply":"2022-07-10T23:17:12.399302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.403026Z","iopub.execute_input":"2022-07-10T23:17:12.403661Z","iopub.status.idle":"2022-07-10T23:17:12.428451Z","shell.execute_reply.started":"2022-07-10T23:17:12.403628Z","shell.execute_reply":"2022-07-10T23:17:12.427251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.429758Z","iopub.execute_input":"2022-07-10T23:17:12.430070Z","iopub.status.idle":"2022-07-10T23:17:12.437243Z","shell.execute_reply.started":"2022-07-10T23:17:12.430041Z","shell.execute_reply":"2022-07-10T23:17:12.436163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function is used to describe histogram of the float feature.\n\nfeature=[f for f in df.columns if df[f].dtype=='float64']\n\nfig,axs=plt.subplots(4,4,figsize=(12,12))\nfor f,ax in zip(feature,axs.ravel()):\n    ax.hist(df[f],density=True,bins=100)\n    ax.set_title(f'Train{f},std={df[f].std():.1f}')\nplt.suptitle('Histograms of the float features',y=0.98,fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:12.440541Z","iopub.execute_input":"2022-07-10T23:17:12.440941Z","iopub.status.idle":"2022-07-10T23:17:16.694833Z","shell.execute_reply.started":"2022-07-10T23:17:12.440903Z","shell.execute_reply":"2022-07-10T23:17:16.693544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It displays correlation among all the columns of the dataset.","metadata":{}},{"cell_type":"code","source":"\ndf.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:16.696629Z","iopub.execute_input":"2022-07-10T23:17:16.697037Z","iopub.status.idle":"2022-07-10T23:17:16.987088Z","shell.execute_reply.started":"2022-07-10T23:17:16.697004Z","shell.execute_reply":"2022-07-10T23:17:16.985957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=df.corr()\nplt.subplots(figsize=(7,5))\nsns.heatmap(corr,vmax=0.9,cmap='viridis',square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:16.988775Z","iopub.execute_input":"2022-07-10T23:17:16.989121Z","iopub.status.idle":"2022-07-10T23:17:17.710167Z","shell.execute_reply.started":"2022-07-10T23:17:16.989090Z","shell.execute_reply":"2022-07-10T23:17:17.708582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-Means Clustering method has been used for Elbow method. Where X- axis is used to represnt number of clusters and Y-axis is total intertia ","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.cluster import KMeans\ninertias = []\nfor k in range(1,15):\n    km = KMeans(n_clusters=k)\n    km.fit(df.iloc[:10000,:])\n    inertias.append(km.inertia_)\n\n# Plot inertias\nplt.figure(figsize=(8,4))\nplt.plot(range(1,15), inertias, 'bx-')\nplt.xlabel('Number of clusters')\nplt.ylabel('Total inertia')\nplt.title('Elbow method')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:17.712020Z","iopub.execute_input":"2022-07-10T23:17:17.713307Z","iopub.status.idle":"2022-07-10T23:17:47.837922Z","shell.execute_reply.started":"2022-07-10T23:17:17.713259Z","shell.execute_reply":"2022-07-10T23:17:47.836661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dendogram is created to study the hierarchical clusters before deciding the number of clusters appropriate to the dataset. The distance at which two clusters combine is referred to as the dendrogram distance.","metadata":{}},{"cell_type":"code","source":"X=df.iloc[:10000,:]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:47.839844Z","iopub.execute_input":"2022-07-10T23:17:47.842070Z","iopub.status.idle":"2022-07-10T23:17:47.848929Z","shell.execute_reply.started":"2022-07-10T23:17:47.842016Z","shell.execute_reply":"2022-07-10T23:17:47.847391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.cluster.hierarchy import dendrogram, linkage\nZ=linkage(X,'ward')\ndendrogram(Z)\nax = plt.gca()\nbounds = ax.get_xbound()\nax.plot(bounds, [525, 525], '--', c='k')\nax.plot(bounds, [325, 325], '--', c='k')\nax.text(bounds[1], 525, ' two clusters', va='center', fontdict={'size': 15})\nax.text(bounds[1], 325, ' three clusters', va='center', fontdict={'size': 15})\nplt.xlabel(\"Clusters\")\nplt.ylabel(\"Inertia\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:17:47.851186Z","iopub.execute_input":"2022-07-10T23:17:47.852006Z","iopub.status.idle":"2022-07-10T23:21:09.876872Z","shell.execute_reply.started":"2022-07-10T23:17:47.851959Z","shell.execute_reply":"2022-07-10T23:21:09.875192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.mixture import GaussianMixture\nmodel = GaussianMixture(n_components=7)\npreds = model.fit_predict(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:21:09.878581Z","iopub.execute_input":"2022-07-10T23:21:09.879099Z","iopub.status.idle":"2022-07-10T23:21:21.459052Z","shell.execute_reply.started":"2022-07-10T23:21:09.879064Z","shell.execute_reply":"2022-07-10T23:21:21.457470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['Predicted']=preds\nsubmission.to_csv('submission.csv',index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:21:21.467198Z","iopub.execute_input":"2022-07-10T23:21:21.471336Z","iopub.status.idle":"2022-07-10T23:21:21.665643Z","shell.execute_reply.started":"2022-07-10T23:21:21.471244Z","shell.execute_reply":"2022-07-10T23:21:21.663735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}