{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T13:29:27.334671Z","iopub.execute_input":"2022-07-31T13:29:27.335344Z","iopub.status.idle":"2022-07-31T13:29:27.362670Z","shell.execute_reply.started":"2022-07-31T13:29:27.335244Z","shell.execute_reply":"2022-07-31T13:29:27.361772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 200)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:27.364629Z","iopub.execute_input":"2022-07-31T13:29:27.365069Z","iopub.status.idle":"2022-07-31T13:29:27.370010Z","shell.execute_reply.started":"2022-07-31T13:29:27.365033Z","shell.execute_reply":"2022-07-31T13:29:27.368994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:27.384824Z","iopub.execute_input":"2022-07-31T13:29:27.386728Z","iopub.status.idle":"2022-07-31T13:29:28.355195Z","shell.execute_reply.started":"2022-07-31T13:29:27.386689Z","shell.execute_reply":"2022-07-31T13:29:28.354287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"f_07 to f_13 are integers. They can be categories","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.357164Z","iopub.execute_input":"2022-07-31T13:29:28.357551Z","iopub.status.idle":"2022-07-31T13:29:28.387257Z","shell.execute_reply.started":"2022-07-31T13:29:28.357515Z","shell.execute_reply":"2022-07-31T13:29:28.386249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.388994Z","iopub.execute_input":"2022-07-31T13:29:28.389342Z","iopub.status.idle":"2022-07-31T13:29:28.396606Z","shell.execute_reply.started":"2022-07-31T13:29:28.389308Z","shell.execute_reply":"2022-07-31T13:29:28.395526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.399651Z","iopub.execute_input":"2022-07-31T13:29:28.400026Z","iopub.status.idle":"2022-07-31T13:29:28.432604Z","shell.execute_reply.started":"2022-07-31T13:29:28.399992Z","shell.execute_reply":"2022-07-31T13:29:28.431613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No Missing values :)","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.434161Z","iopub.execute_input":"2022-07-31T13:29:28.434518Z","iopub.status.idle":"2022-07-31T13:29:28.622216Z","shell.execute_reply.started":"2022-07-31T13:29:28.434484Z","shell.execute_reply":"2022-07-31T13:29:28.621211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.623625Z","iopub.execute_input":"2022-07-31T13:29:28.624688Z","iopub.status.idle":"2022-07-31T13:29:28.631636Z","shell.execute_reply.started":"2022-07-31T13:29:28.624649Z","shell.execute_reply":"2022-07-31T13:29:28.630513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.633031Z","iopub.execute_input":"2022-07-31T13:29:28.633962Z","iopub.status.idle":"2022-07-31T13:29:28.644300Z","shell.execute_reply.started":"2022-07-31T13:29:28.633927Z","shell.execute_reply":"2022-07-31T13:29:28.643458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfig = plt.figure(figsize = (20,20))\nfig.subplots(5,6)\nfor i,col in enumerate(tqdm(df.columns[1:])) :\n    plt.subplot(5,6,i+1)\n    sns.histplot(df[col], kde = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:28.647212Z","iopub.execute_input":"2022-07-31T13:29:28.647479Z","iopub.status.idle":"2022-07-31T13:29:54.664071Z","shell.execute_reply.started":"2022-07-31T13:29:28.647456Z","shell.execute_reply":"2022-07-31T13:29:54.663178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,20))\nfig.subplots(5,6)\nfor i,col in enumerate(tqdm(df.columns[1:])) :\n    plt.subplot(5,6,i+1)\n    sns.boxplot(x = df[col])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:54.665111Z","iopub.execute_input":"2022-07-31T13:29:54.665483Z","iopub.status.idle":"2022-07-31T13:29:57.285679Z","shell.execute_reply.started":"2022-07-31T13:29:54.665440Z","shell.execute_reply":"2022-07-31T13:29:57.284740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,20))\nsns.heatmap(df.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:57.293431Z","iopub.execute_input":"2022-07-31T13:29:57.295677Z","iopub.status.idle":"2022-07-31T13:29:58.132214Z","shell.execute_reply.started":"2022-07-31T13:29:57.295636Z","shell.execute_reply":"2022-07-31T13:29:58.131269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.loc[:, 'f_07':'f_13'].corr(), annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:58.133553Z","iopub.execute_input":"2022-07-31T13:29:58.134629Z","iopub.status.idle":"2022-07-31T13:29:58.538421Z","shell.execute_reply.started":"2022-07-31T13:29:58.134587Z","shell.execute_reply":"2022-07-31T13:29:58.537374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.loc[:, 'f_22':'f_28'].corr(), annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:58.539690Z","iopub.execute_input":"2022-07-31T13:29:58.540057Z","iopub.status.idle":"2022-07-31T13:29:58.943201Z","shell.execute_reply.started":"2022-07-31T13:29:58.540020Z","shell.execute_reply":"2022-07-31T13:29:58.942293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Outlier analysis and treatment using transformation and limiting values","metadata":{}},{"cell_type":"code","source":"int_col = []\nfor col in df.columns[1:]:\n    if df[col].dtype == 'int64':\n        int_col.append(col)\nint_col        ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:58.944785Z","iopub.execute_input":"2022-07-31T13:29:58.945124Z","iopub.status.idle":"2022-07-31T13:29:58.953987Z","shell.execute_reply.started":"2022-07-31T13:29:58.945088Z","shell.execute_reply":"2022-07-31T13:29:58.953015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\npt = PowerTransformer()\ndf_pt = pt.fit_transform(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:29:58.955448Z","iopub.execute_input":"2022-07-31T13:29:58.956103Z","iopub.status.idle":"2022-07-31T13:30:02.306774Z","shell.execute_reply.started":"2022-07-31T13:29:58.956066Z","shell.execute_reply":"2022-07-31T13:30:02.305800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pt","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:02.308457Z","iopub.execute_input":"2022-07-31T13:30:02.308857Z","iopub.status.idle":"2022-07-31T13:30:02.316629Z","shell.execute_reply.started":"2022-07-31T13:30:02.308817Z","shell.execute_reply":"2022-07-31T13:30:02.315539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (10,10))\nfig.subplots(4,2)\nfor i,col in enumerate(tqdm(int_col)) :\n    plt.subplot(4,2,i+1)\n    sns.boxplot(x = df[col])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:02.318648Z","iopub.execute_input":"2022-07-31T13:30:02.319557Z","iopub.status.idle":"2022-07-31T13:30:03.141273Z","shell.execute_reply.started":"2022-07-31T13:30:02.319520Z","shell.execute_reply":"2022-07-31T13:30:03.140302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Used Power transformation to make distribution gaussian type, outliers are removed since outliers are bad for kmeans clustering","metadata":{}},{"cell_type":"markdown","source":"### Standard Scaling helps improve clustering","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nsc_df = scaler.fit_transform(df_pt)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.142844Z","iopub.execute_input":"2022-07-31T13:30:03.143205Z","iopub.status.idle":"2022-07-31T13:30:03.179010Z","shell.execute_reply.started":"2022-07-31T13:30:03.143169Z","shell.execute_reply":"2022-07-31T13:30:03.178014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.180592Z","iopub.execute_input":"2022-07-31T13:30:03.180964Z","iopub.status.idle":"2022-07-31T13:30:03.209364Z","shell.execute_reply.started":"2022-07-31T13:30:03.180925Z","shell.execute_reply":"2022-07-31T13:30:03.208417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.210649Z","iopub.execute_input":"2022-07-31T13:30:03.211442Z","iopub.status.idle":"2022-07-31T13:30:03.218037Z","shell.execute_reply.started":"2022-07-31T13:30:03.211404Z","shell.execute_reply":"2022-07-31T13:30:03.216814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.220009Z","iopub.execute_input":"2022-07-31T13:30:03.221245Z","iopub.status.idle":"2022-07-31T13:30:03.435832Z","shell.execute_reply.started":"2022-07-31T13:30:03.221207Z","shell.execute_reply":"2022-07-31T13:30:03.434886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Find optimal K","metadata":{}},{"cell_type":"code","source":"# inertia_sc = []\n# sil_sc = []\n# for k in range(4,10):\n#     km = KMeans(n_clusters = k, random_state=42)\n#     km.fit(sc_df)\n#     inertia_sc.append(km.inertia_)\n#     sil_sc.append(silhouette_score(sc_df, km.labels_))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.437036Z","iopub.execute_input":"2022-07-31T13:30:03.437396Z","iopub.status.idle":"2022-07-31T13:30:03.441943Z","shell.execute_reply.started":"2022-07-31T13:30:03.437341Z","shell.execute_reply":"2022-07-31T13:30:03.441011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(inertia_sc)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.443245Z","iopub.execute_input":"2022-07-31T13:30:03.444191Z","iopub.status.idle":"2022-07-31T13:30:03.451760Z","shell.execute_reply.started":"2022-07-31T13:30:03.444154Z","shell.execute_reply":"2022-07-31T13:30:03.450834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Elbow Curve","metadata":{}},{"cell_type":"code","source":"# sns.lineplot(x = range(4, 10), y = inertia_sc)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.453118Z","iopub.execute_input":"2022-07-31T13:30:03.453522Z","iopub.status.idle":"2022-07-31T13:30:03.465097Z","shell.execute_reply.started":"2022-07-31T13:30:03.453454Z","shell.execute_reply":"2022-07-31T13:30:03.464044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Slope of the curve is decreasing. Optimum K value seems at the point where reduction in SSD is highest. \nPossibly K values : 4, 5, 6, 7","metadata":{}},{"cell_type":"markdown","source":"### Silhoutte Score Curve","metadata":{}},{"cell_type":"markdown","source":"Silhouette Score = (b-a)/max(a,b)\n\nwhere\n\na= average intra-cluster distance i.e the average distance between each point within a cluster.\n\nb= average inter-cluster distance i.e the average distance between all clusters.","metadata":{}},{"cell_type":"code","source":"# sns.lineplot(x = range(4, 10), y = sil_sc)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.466999Z","iopub.execute_input":"2022-07-31T13:30:03.467404Z","iopub.status.idle":"2022-07-31T13:30:03.479368Z","shell.execute_reply.started":"2022-07-31T13:30:03.467352Z","shell.execute_reply":"2022-07-31T13:30:03.478379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Gives the goodness of clusters.\n\n(bad) -1 < score < 1 (good)\n\nOptimum value seems at 4, 5, 7, 8 clusters\n","metadata":{}},{"cell_type":"code","source":"from yellowbrick.cluster import KElbowVisualizer\nkm = KMeans(random_state=42)\n# k is range of number of clusters.\nvisualizer = KElbowVisualizer(km, k=(4,10), timings=True)\nvisualizer.fit(sc_df)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:03.480826Z","iopub.execute_input":"2022-07-31T13:30:03.481487Z","iopub.status.idle":"2022-07-31T13:30:49.740195Z","shell.execute_reply.started":"2022-07-31T13:30:03.481452Z","shell.execute_reply":"2022-07-31T13:30:49.739272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cluster number K = 7","metadata":{}},{"cell_type":"code","source":"km = KMeans(n_clusters = 7, random_state=42)\ny_pred = km.fit_predict(sc_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:49.741525Z","iopub.execute_input":"2022-07-31T13:30:49.741949Z","iopub.status.idle":"2022-07-31T13:30:57.936887Z","shell.execute_reply.started":"2022-07-31T13:30:49.741912Z","shell.execute_reply":"2022-07-31T13:30:57.935890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gaussian Mixture Model ","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import GaussianMixture, BayesianGaussianMixture\ngmm = BayesianGaussianMixture(n_components=6, covariance_type='full', random_state=1)\npreds = gmm.fit_predict(sc_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:30:57.938215Z","iopub.execute_input":"2022-07-31T13:30:57.938853Z","iopub.status.idle":"2022-07-31T13:31:50.930158Z","shell.execute_reply.started":"2022-07-31T13:30:57.938813Z","shell.execute_reply":"2022-07-31T13:31:50.928867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"Predicted\"] = preds\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:31:50.940912Z","iopub.execute_input":"2022-07-31T13:31:50.943758Z","iopub.status.idle":"2022-07-31T13:31:51.117303Z","shell.execute_reply.started":"2022-07-31T13:31:50.943706Z","shell.execute_reply":"2022-07-31T13:31:51.116289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:31:51.118535Z","iopub.execute_input":"2022-07-31T13:31:51.118877Z","iopub.status.idle":"2022-07-31T13:31:51.130994Z","shell.execute_reply.started":"2022-07-31T13:31:51.118833Z","shell.execute_reply":"2022-07-31T13:31:51.130080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}