{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T19:25:17.342702Z","iopub.execute_input":"2022-07-15T19:25:17.343657Z","iopub.status.idle":"2022-07-15T19:25:17.386471Z","shell.execute_reply.started":"2022-07-15T19:25:17.343509Z","shell.execute_reply":"2022-07-15T19:25:17.385223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.cluster import SpectralClustering \nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:17.389117Z","iopub.execute_input":"2022-07-15T19:25:17.390056Z","iopub.status.idle":"2022-07-15T19:25:19.105772Z","shell.execute_reply.started":"2022-07-15T19:25:17.389995Z","shell.execute_reply":"2022-07-15T19:25:19.104391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in input data\ndata = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/data.csv\", index_col='id')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:19.108423Z","iopub.execute_input":"2022-07-15T19:25:19.108816Z","iopub.status.idle":"2022-07-15T19:25:21.007141Z","shell.execute_reply.started":"2022-07-15T19:25:19.108780Z","shell.execute_reply":"2022-07-15T19:25:21.005875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:21.009782Z","iopub.execute_input":"2022-07-15T19:25:21.010717Z","iopub.status.idle":"2022-07-15T19:25:21.044656Z","shell.execute_reply.started":"2022-07-15T19:25:21.010659Z","shell.execute_reply":"2022-07-15T19:25:21.043356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_col = ['f_07','f_08','f_09','f_10','f_11','f_12','f_13',\n          'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\ndata_cat = data[imp_col]","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:21.047375Z","iopub.execute_input":"2022-07-15T19:25:21.048378Z","iopub.status.idle":"2022-07-15T19:25:21.058896Z","shell.execute_reply.started":"2022-07-15T19:25:21.048333Z","shell.execute_reply":"2022-07-15T19:25:21.057762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cat.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:21.060525Z","iopub.execute_input":"2022-07-15T19:25:21.060942Z","iopub.status.idle":"2022-07-15T19:25:21.092199Z","shell.execute_reply.started":"2022-07-15T19:25:21.060897Z","shell.execute_reply":"2022-07-15T19:25:21.091213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n\nx = data_cat.values\nmin_max_scaler = preprocessing.MinMaxScaler()\nx_scaled = min_max_scaler.fit_transform(x)\n\nrb_scalar = preprocessing.PowerTransformer()\nx_scaled = rb_scalar.fit_transform(x_scaled)\ndata_scaled = pd.DataFrame(x_scaled, columns=data_cat.columns)\n#data_scaled = data_num","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:21.093806Z","iopub.execute_input":"2022-07-15T19:25:21.094648Z","iopub.status.idle":"2022-07-15T19:25:22.858423Z","shell.execute_reply.started":"2022-07-15T19:25:21.094603Z","shell.execute_reply":"2022-07-15T19:25:22.856977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#bics = {}\n#aics = {}\n#for n_cluster in range(2, 21):\n#    print(n_cluster)\n#    gm = GaussianMixture(n_components = n_cluster, n_init=10)\n#    gm.fit(data_scaled)\n#    bic = gm.bic(data_scaled)\n#    aic = gm.aic(data_scaled)\n#    \n#    bics[n_cluster] = bic\n#    aics[n_cluster] = aic","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:22.860355Z","iopub.execute_input":"2022-07-15T19:25:22.860865Z","iopub.status.idle":"2022-07-15T19:25:22.867214Z","shell.execute_reply.started":"2022-07-15T19:25:22.860791Z","shell.execute_reply":"2022-07-15T19:25:22.865885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.plot(bics.keys(), bics.values(), 'g-')\n#plt.plot(aics.keys(), aics.values(), 'r--')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:25:22.869391Z","iopub.execute_input":"2022-07-15T19:25:22.870386Z","iopub.status.idle":"2022-07-15T19:25:22.892439Z","shell.execute_reply.started":"2022-07-15T19:25:22.870328Z","shell.execute_reply":"2022-07-15T19:25:22.890935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clustering with 7 cluster\n#gm = GaussianMixture(n_components = 7, covariance_type='full')\n#data_clst_predict = gm.fit_predict(data_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T16:57:04.779471Z","iopub.execute_input":"2022-07-14T16:57:04.779841Z","iopub.status.idle":"2022-07-14T16:57:04.789589Z","shell.execute_reply.started":"2022-07-14T16:57:04.779809Z","shell.execute_reply":"2022-07-14T16:57:04.788510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_clst = pd.DataFrame()\npredictions_prob = pd.DataFrame()\n\nbgm = BayesianGaussianMixture(n_components=7,\n                              covariance_type='full', \n                              random_state=0, \n                              n_init=3, \n                              tol = 0.01)\nbgm.fit(data_scaled)\npredictions_clst['pred_0'] = bgm.predict(data_scaled)\npredictions_prob = bgm.predict_proba(data_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:14:52.474535Z","iopub.execute_input":"2022-07-15T21:14:52.475771Z","iopub.status.idle":"2022-07-15T21:15:03.515354Z","shell.execute_reply.started":"2022-07-15T21:14:52.475710Z","shell.execute_reply":"2022-07-15T21:15:03.513924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_prob = pd.DataFrame(predictions_prob)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:03.523179Z","iopub.execute_input":"2022-07-15T21:15:03.526793Z","iopub.status.idle":"2022-07-15T21:15:03.536961Z","shell.execute_reply.started":"2022-07-15T21:15:03.526720Z","shell.execute_reply":"2022-07-15T21:15:03.534905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a= {1:2, 3:4, 2:4}\nlen(a.keys())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:03.539665Z","iopub.execute_input":"2022-07-15T21:15:03.540719Z","iopub.status.idle":"2022-07-15T21:15:03.555663Z","shell.execute_reply.started":"2022-07-15T21:15:03.540662Z","shell.execute_reply":"2022-07-15T21:15:03.554215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clustering with 7 cluster\n# full already done: score __ \n\npredictions = pd.DataFrame()\nskip_seed = []\n\nfor rs in tqdm(range(1, 50)):\n    bgm = BayesianGaussianMixture(n_components=7,\n                                  covariance_type='full', \n                                  random_state=rs, \n                                  n_init=3, \n                                  tol = 0.01)\n    bgm.fit(data_scaled)\n    new_col = f\"pred_{rs}\"\n    predictions_clst[new_col] = bgm.predict(data_scaled)\n    \n    index_comb = predictions_clst[['pred_0', new_col]].value_counts().head(7).index.to_list()\n    index_comb = dict(index_comb)\n    index_comb = dict((v, k) for k, v in index_comb.items())\n    \n    if (len(index_comb) == 7):\n    \n        predictions_clst[new_col] = predictions_clst[new_col].map(index_comb)\n    \n        _prob = pd.DataFrame(bgm.predict_proba(data_scaled))\n        _prob.rename(columns = index_comb, inplace = True)\n    \n        predictions_prob = predictions_prob + _prob\n    else:\n        skip_seed.append(rs)\n        print(rs)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:03.562999Z","iopub.execute_input":"2022-07-15T21:15:03.565597Z","iopub.status.idle":"2022-07-15T21:15:25.958311Z","shell.execute_reply.started":"2022-07-15T21:15:03.565529Z","shell.execute_reply":"2022-07-15T21:15:25.956585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_prob.to_csv(f\"Summed_prob_BGMM_C7_MMS-PT-Scaled.csv\", index=False)\npredictions_clst.to_csv(f\"predict_BGMM_C7_MMS-PT-Scaled.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:25.961703Z","iopub.execute_input":"2022-07-15T21:15:25.964160Z","iopub.status.idle":"2022-07-15T21:15:26.117307Z","shell.execute_reply.started":"2022-07-15T21:15:25.964087Z","shell.execute_reply":"2022-07-15T21:15:26.115928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_prob","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:18:55.909396Z","iopub.execute_input":"2022-07-15T21:18:55.909898Z","iopub.status.idle":"2022-07-15T21:18:55.931602Z","shell.execute_reply.started":"2022-07-15T21:18:55.909840Z","shell.execute_reply":"2022-07-15T21:18:55.930305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.argmax(predictions_prob.to_numpy(), axis=1)\nss = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv\")\nss['Predicted'] = pred\nss.to_csv(f\"submit_BGMM_C7_MMS-PT-Scaled_SummedProba.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:26.119045Z","iopub.execute_input":"2022-07-15T21:15:26.119496Z","iopub.status.idle":"2022-07-15T21:15:26.172587Z","shell.execute_reply.started":"2022-07-15T21:15:26.119455Z","shell.execute_reply":"2022-07-15T21:15:26.170996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## KModes","metadata":{}},{"cell_type":"code","source":"#kmode = KModes(n_clusters=7, n_init = 10, verbose=1, random_state=42)\n#kmode_conse_pred = kmode.fit_predict(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:26.183298Z","iopub.status.idle":"2022-07-15T21:15:26.184162Z","shell.execute_reply.started":"2022-07-15T21:15:26.183926Z","shell.execute_reply":"2022-07-15T21:15:26.183958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submit_df = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv\")\n#submit_df['Predicted'] = kmode_conse_pred\n#submit_df.to_csv(f\"sample_submission_BGMM_C7_MMS-PT-Scaled_MULTIseed_kmode.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T21:15:26.185929Z","iopub.status.idle":"2022-07-15T21:15:26.186318Z","shell.execute_reply.started":"2022-07-15T21:15:26.186129Z","shell.execute_reply":"2022-07-15T21:15:26.186147Z"},"trusted":true},"execution_count":null,"outputs":[]}]}