{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nfrom sklearn.preprocessing import RobustScaler,PowerTransformer, StandardScaler, MaxAbsScaler\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.decomposition import PCA\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T01:08:47.898446Z","iopub.execute_input":"2022-07-19T01:08:47.898790Z","iopub.status.idle":"2022-07-19T01:08:47.906952Z","shell.execute_reply.started":"2022-07-19T01:08:47.898761Z","shell.execute_reply":"2022-07-19T01:08:47.905625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv',index_col = 'id')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:08:55.583457Z","iopub.execute_input":"2022-07-19T01:08:55.583823Z","iopub.status.idle":"2022-07-19T01:08:56.449624Z","shell.execute_reply.started":"2022-07-19T01:08:55.583795Z","shell.execute_reply":"2022-07-19T01:08:56.448576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categ_col = [col for col in data.columns if data[col].dtypes == 'int64']\ncateg_col","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:09:06.958007Z","iopub.execute_input":"2022-07-19T01:09:06.958321Z","iopub.status.idle":"2022-07-19T01:09:06.966901Z","shell.execute_reply.started":"2022-07-19T01:09:06.958298Z","shell.execute_reply":"2022-07-19T01:09:06.965508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data['f_07'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T00:26:47.325889Z","iopub.execute_input":"2022-07-19T00:26:47.326259Z","iopub.status.idle":"2022-07-19T00:26:47.887783Z","shell.execute_reply.started":"2022-07-19T00:26:47.326229Z","shell.execute_reply":"2022-07-19T00:26:47.886427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (30,20))\nplt.suptitle(\"categorical columns distributions\",fontsize = 18)\nfor i,col in enumerate(categ_col,start = 1):\n    plt.subplot(3,3,i)\n    sns.histplot(data = data,x = col)\n    plt.title(f\"{col} distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:09:18.312188Z","iopub.execute_input":"2022-07-19T01:09:18.312497Z","iopub.status.idle":"2022-07-19T01:09:21.164604Z","shell.execute_reply.started":"2022-07-19T01:09:18.312474Z","shell.execute_reply":"2022-07-19T01:09:21.163231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\n# ls1 = ['f_07', 'f_08', 'f_09', 'f_10', 'f_12', 'f_13']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:09:44.047716Z","iopub.execute_input":"2022-07-19T01:09:44.048104Z","iopub.status.idle":"2022-07-19T01:09:44.053387Z","shell.execute_reply.started":"2022-07-19T01:09:44.048078Z","shell.execute_reply":"2022-07-19T01:09:44.052347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_ls = data[ls]\npower_transformer = PowerTransformer().fit(data_ls)\ndata_new = power_transformer.transform(data_ls)\nX_scaled = pd.DataFrame(data_new, columns=ls)\nX_scaled\n# rb_scaler=RobustScaler()\n# X=rb_scaler.fit_transform(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:10:36.731009Z","iopub.execute_input":"2022-07-19T01:10:36.731403Z","iopub.status.idle":"2022-07-19T01:10:39.709453Z","shell.execute_reply.started":"2022-07-19T01:10:36.731372Z","shell.execute_reply":"2022-07-19T01:10:39.708064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (30,20))\nplt.suptitle(\"categorical columns distributions\",fontsize = 18)\nfor i,col in enumerate(categ_col,start = 1):\n    plt.subplot(3,3,i)\n    sns.histplot(data = X_scaled,x = col)\n    plt.title(f\"{col} distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:11:47.823859Z","iopub.execute_input":"2022-07-19T01:11:47.824189Z","iopub.status.idle":"2022-07-19T01:11:50.374342Z","shell.execute_reply.started":"2022-07-19T01:11:47.824164Z","shell.execute_reply":"2022-07-19T01:11:50.373221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting outlier\nout=[]\ndef iqr_outliers(df,ft):\n    q1 = df[ft].quantile(0.25)\n    q3 = df[ft].quantile(0.75)\n    iqr = q3-q1\n    Lower_tail = q1 - 1.5 * iqr\n    Upper_tail = q3 + 1.5 * iqr\n    c=0\n    for i in range(len(df[ft])):\n        if df[ft][i] > Upper_tail or df[ft][i] < Lower_tail:\n            c+=1\n    return c\nod={ f:iqr_outliers(X_scaled,f) for f in X_scaled[ls].columns }\n\n# Plotting Outliers\nplt.style.use('ggplot')\nplt.figure(figsize=(16,8))\nplt.bar(x=od.keys(),height=od.values())\nplt.xlabel(\"Features\")\nplt.ylabel(\"Outliers\")\nplt.title('Outliers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:12:07.345509Z","iopub.execute_input":"2022-07-19T01:12:07.345878Z","iopub.status.idle":"2022-07-19T01:12:24.249577Z","shell.execute_reply.started":"2022-07-19T01:12:07.345848Z","shell.execute_reply":"2022-07-19T01:12:24.248008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls1 = ['f_08','f_22','f_23','f_24','f_25','f_26','f_27','f_28']\nfor i in X_scaled[ls]:\n    tenth_perc = np.percentile(X_scaled[i],10)\n    ninty_perc = np.percentile(X_scaled[i],90)\n    print(f'{i}, {tenth_perc} and {ninty_perc}')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:15:09.501502Z","iopub.execute_input":"2022-07-19T01:15:09.501818Z","iopub.status.idle":"2022-07-19T01:15:09.546167Z","shell.execute_reply.started":"2022-07-19T01:15:09.501792Z","shell.execute_reply":"2022-07-19T01:15:09.544499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#above plots looks good\n#lets cap values which are highr or lower than threshold\nfor col in X_scaled[ls]:\n    tenth_perc = np.percentile(X_scaled[col],10)\n    ninty_perc = np.percentile(X_scaled[col],90)\n    b = np.where(X_scaled[col]>ninty_perc, ninty_perc, X_scaled[col])\n    X_scaled[col] = np.where(b<tenth_perc, tenth_perc, b)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:18:40.244059Z","iopub.execute_input":"2022-07-19T01:18:40.244382Z","iopub.status.idle":"2022-07-19T01:18:40.300360Z","shell.execute_reply.started":"2022-07-19T01:18:40.244357Z","shell.execute_reply":"2022-07-19T01:18:40.299113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting to see outlier are removded\nod={ f:iqr_outliers(X_scaled,f) for f in X_scaled[ls].columns }\n\n# Plotting Outliers\nplt.style.use('ggplot')\nplt.figure(figsize=(16,8))\nplt.bar(x=od.keys(),height=od.values())\nplt.xlabel(\"Features\")\nplt.ylabel(\"Outliers\")\nplt.title('Outliers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:19:27.000986Z","iopub.execute_input":"2022-07-19T01:19:27.001365Z","iopub.status.idle":"2022-07-19T01:19:51.248998Z","shell.execute_reply.started":"2022-07-19T01:19:27.001335Z","shell.execute_reply":"2022-07-19T01:19:51.248015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rb_scaler=RobustScaler()\nX=rb_scaler.fit_transform(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:21:07.483431Z","iopub.execute_input":"2022-07-19T01:21:07.483784Z","iopub.status.idle":"2022-07-19T01:21:07.545606Z","shell.execute_reply.started":"2022-07-19T01:21:07.483755Z","shell.execute_reply":"2022-07-19T01:21:07.544526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BGM = BayesianGaussianMixture(n_components=7,covariance_type='full',random_state=1,n_init=5,tol=0.01)\n# # fit model and predict clusters\n# preds = BGM.fit_predict(X)\n# pp=BGM.predict_proba(X)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #get feature importance details\n# plt.style.use('ggplot')\n# plt.figure(figsize=(15,6))\n# for i in range(BGM.means_.shape[0]):\n#     plt.scatter(np.arange(X.shape[1]), BGM.means_[i])\n# plt.xticks(ticks=np.arange(X.shape[1]), labels1=data.columns)\n# plt.xlabel('Features')\n# plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ls = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:58:54.775435Z","iopub.execute_input":"2022-07-17T17:58:54.775821Z","iopub.status.idle":"2022-07-17T17:58:54.781487Z","shell.execute_reply.started":"2022-07-17T17:58:54.775789Z","shell.execute_reply":"2022-07-17T17:58:54.780413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.cluster import KMeans\n\n# print('Elbow Method to determine the number of clusters to be formed:')\n# Elbow_M = KElbowVisualizer(KMeans(random_state=23), k=(4,12))\n# Elbow_M.fit(X)\n# Elbow_M.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:22:13.331200Z","iopub.execute_input":"2022-07-19T01:22:13.331577Z","iopub.status.idle":"2022-07-19T01:23:02.946814Z","shell.execute_reply.started":"2022-07-19T01:22:13.331547Z","shell.execute_reply":"2022-07-19T01:23:02.945705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#new block\n# num_data1 = data[num_data].drop(['f_23','f_26','f_27'],axis = 1)\ngmm1 = GaussianMixture(n_components = 6,covariance_type='full',random_state = 10)\ngmm1.fit(X)\n \nlabels1 = gmm1.predict(X)\nframe1 = pd.DataFrame(X)\nframe1['cluster'] = labels1\n# sns.scatterplot(data = frame1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:43:54.928823Z","iopub.execute_input":"2022-07-19T01:43:54.929143Z","iopub.status.idle":"2022-07-19T01:44:03.957798Z","shell.execute_reply.started":"2022-07-19T01:43:54.929120Z","shell.execute_reply":"2022-07-19T01:44:03.957017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(gmm1.n_iter_)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:25:31.782615Z","iopub.execute_input":"2022-07-19T01:25:31.782939Z","iopub.status.idle":"2022-07-19T01:25:31.787865Z","shell.execute_reply.started":"2022-07-19T01:25:31.782906Z","shell.execute_reply":"2022-07-19T01:25:31.786806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = labels1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:44:15.076938Z","iopub.execute_input":"2022-07-19T01:44:15.077285Z","iopub.status.idle":"2022-07-19T01:44:15.258975Z","shell.execute_reply.started":"2022-07-19T01:44:15.077261Z","shell.execute_reply":"2022-07-19T01:44:15.257814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = labels1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:38:42.157719Z","iopub.execute_input":"2022-07-17T17:38:42.158112Z","iopub.status.idle":"2022-07-17T17:38:42.388635Z","shell.execute_reply.started":"2022-07-17T17:38:42.158081Z","shell.execute_reply":"2022-07-17T17:38:42.387564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\nsample['Predicted'] = labels1\nsample.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:44:23.792020Z","iopub.execute_input":"2022-07-19T01:44:23.792345Z","iopub.status.idle":"2022-07-19T01:44:23.810498Z","shell.execute_reply.started":"2022-07-19T01:44:23.792321Z","shell.execute_reply":"2022-07-19T01:44:23.809697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.to_csv('submission_gmm_11.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T01:44:32.023879Z","iopub.execute_input":"2022-07-19T01:44:32.024677Z","iopub.status.idle":"2022-07-19T01:44:32.124610Z","shell.execute_reply.started":"2022-07-19T01:44:32.024647Z","shell.execute_reply":"2022-07-19T01:44:32.123593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}