{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T20:29:38.277253Z","iopub.execute_input":"2022-07-17T20:29:38.277683Z","iopub.status.idle":"2022-07-17T20:29:38.285653Z","shell.execute_reply.started":"2022-07-17T20:29:38.277649Z","shell.execute_reply":"2022-07-17T20:29:38.284393Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Introduction\nThank you for coming my notebook.First Let me explain I do in this notebook.   \nSeveral notebook have shown that certain　columns　and GBMM are usefull in this competition.   \nIn this notebook, I am examining methods using BGMM with the major voting and the probability averaging.   \nAnd I also use UMAP to visualize if the data with high probability is really well clustered.   \nFinally, I try to train data with high probability to predict data with low probability by LightGBM.\n\nReference:   \n[AMBROSM's notebook](https://www.kaggle.com/code/ambrosm/tpsjul22-gaussian-mixture-cluster-analysis)   \n[AKIO ONODERA' notebook](https://www.kaggle.com/code/akioonodera/tps-jul2022-bgmm)\n[RICOPUE' notebook](https://www.kaggle.com/code/ricopue/tps-jul22-clusters-and-lgb)\n\n\n日本語記載   \nまず初めにこのノートブックで行っていることを説明します。   \nいくつかのノートブックにより今回のコンペティションに有用なカラムやBGMMが有効であることが示されています。     \nこのノートブックでは、私はBGMMを使った、多数決とprobabilityの平均を用いた方法を検討しています。     \nまたprobabilityの高いデータが本当にクラスタリングがうまくいっているかどうかをUMAPを用いて、可視化しています。  \n最後に、probabilityの高いデータを使って、LightGBMで学習させ、probabilityの低いデータを予測することを検討しています。","metadata":{}},{"cell_type":"markdown","source":"# Install","metadata":{}},{"cell_type":"code","source":"!pip install japanize-matplotlib","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T20:29:42.060517Z","iopub.execute_input":"2022-07-17T20:29:42.061210Z","iopub.status.idle":"2022-07-17T20:29:56.655381Z","shell.execute_reply.started":"2022-07-17T20:29:42.061169Z","shell.execute_reply":"2022-07-17T20:29:56.654302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pyclustering","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T20:29:56.658360Z","iopub.execute_input":"2022-07-17T20:29:56.658750Z","iopub.status.idle":"2022-07-17T20:30:10.807881Z","shell.execute_reply.started":"2022-07-17T20:29:56.658709Z","shell.execute_reply":"2022-07-17T20:30:10.806721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install umap-learn","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T20:30:10.809758Z","iopub.execute_input":"2022-07-17T20:30:10.810297Z","iopub.status.idle":"2022-07-17T20:30:20.713359Z","shell.execute_reply.started":"2022-07-17T20:30:10.810253Z","shell.execute_reply":"2022-07-17T20:30:20.712137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Library import","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-17T20:30:30.906528Z","iopub.execute_input":"2022-07-17T20:30:30.907114Z","iopub.status.idle":"2022-07-17T20:30:30.913293Z","shell.execute_reply.started":"2022-07-17T20:30:30.907079Z","shell.execute_reply":"2022-07-17T20:30:30.912317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loading","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')\nsub = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T20:30:31.694251Z","iopub.execute_input":"2022-07-17T20:30:31.694612Z","iopub.status.idle":"2022-07-17T20:30:32.249197Z","shell.execute_reply.started":"2022-07-17T20:30:31.694581Z","shell.execute_reply":"2022-07-17T20:30:32.248268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_col = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22','f_23','f_24','f_25','f_26','f_27','f_28' ]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T20:30:32.370384Z","iopub.execute_input":"2022-07-17T20:30:32.371062Z","iopub.status.idle":"2022-07-17T20:30:32.377442Z","shell.execute_reply.started":"2022-07-17T20:30:32.371028Z","shell.execute_reply":"2022-07-17T20:30:32.376545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.preprocessing import PowerTransformer\n#std_scaled_df = pd.DataFrame(StandardScaler().fit_transform(df.drop('id')), columns=df.columns)\n#minmax_scaled_df = pd.DataFrame(MinMaxScaler().fit_transform(df.drop('id')), columns=df.columns)\n#robust_scaled_df = pd.DataFrame(RobustScaler().fit_transform(df.drop('id')), columns=df.columns)\npt_scaled_df = pd.DataFrame(PowerTransformer().fit_transform(df.drop('id', axis=1)), columns=df.drop('id', axis=1).columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T20:30:34.815791Z","iopub.execute_input":"2022-07-17T20:30:34.816174Z","iopub.status.idle":"2022-07-17T20:30:38.147564Z","shell.execute_reply.started":"2022-07-17T20:30:34.816142Z","shell.execute_reply":"2022-07-17T20:30:38.146488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BGMM(Bayesian GMM)","metadata":{}},{"cell_type":"markdown","source":"Since even a one shift in random_state can significantly change the clustering results.      \nTherefore, I change rondom_state 10 times and used the major voting and the probability averaging.\n\nrandom_stateが1ずれるだけでも、クラスタリングの結果は大きく変わる可能性があるため、   \nrondom_stateを１０回変えて、多数決及びprobaの平均値で求める方法を行なった。","metadata":{}},{"cell_type":"markdown","source":"# BGMM with the major voting\n多数決を使ったBGMM","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture","metadata":{"execution":{"iopub.status.busy":"2022-07-17T20:30:40.273892Z","iopub.execute_input":"2022-07-17T20:30:40.274300Z","iopub.status.idle":"2022-07-17T20:30:40.279051Z","shell.execute_reply.started":"2022-07-17T20:30:40.274265Z","shell.execute_reply":"2022-07-17T20:30:40.277848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = sub.copy()\nfor i in range(10):\n    bgmm = BayesianGaussianMixture(n_components=7, covariance_type='full', max_iter=100, random_state=i, n_init = 5)\n    bgmm.fit(pt_scaled_df[use_col])\n    predict = bgmm.predict(pt_scaled_df[use_col])\n    submission[f'Predicted_{i}'] = predict","metadata":{"execution":{"iopub.status.busy":"2022-07-17T20:30:41.896148Z","iopub.execute_input":"2022-07-17T20:30:41.896608Z","iopub.status.idle":"2022-07-17T20:57:03.404827Z","shell.execute_reply.started":"2022-07-17T20:30:41.896572Z","shell.execute_reply":"2022-07-17T20:57:03.403497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the assigned Cluster_No. is not necessarily the same, it is necessary to adjust to the same Cluster_No.      \nIn each Cluster with rondom_state=0, the ClusterNo. of each random_state with the highest number is checked.   \n\n必ずしも割り振られたCluster_No.が同一とは限らないため、同じcluster_No.に調整する必要がある。   \nrondom_state=0の各Clusterにおいて、一番数の多い各条件のClusterNo.を確認している。","metadata":{}},{"cell_type":"code","source":"for i in range(1,10):\n    for c in range(7):\n        temp = submission[submission['Predicted_0'] == c]\n        temp_index = temp[f'Predicted_{i}'].value_counts().index[0]\n        temp_value = temp[f'Predicted_{i}'].value_counts()[temp_index]\n        ratio = temp_value/len(temp)\n        print(f'(ClusterNo.{temp_index} of Predicted_{i})/(ClusterNo.{c} of Predicted_0) : {ratio:.2f}')\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:06:40.633812Z","iopub.execute_input":"2022-07-17T21:06:40.634151Z","iopub.status.idle":"2022-07-17T21:06:40.773125Z","shell.execute_reply.started":"2022-07-17T21:06:40.634122Z","shell.execute_reply":"2022-07-17T21:06:40.772080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Match the Cluster_No. of each random_state to the Cluster_No. of random_state=0.      \n   \n各random_stateのCluster_No.をrandom_state=0のCluster_Noに合わせる。","metadata":{}},{"cell_type":"code","source":"convert_df = pd.DataFrame(np.random.random([7, 9]), columns=['convert_1', 'convert_2', 'convert_3', 'convert_4', 'convert_5', 'convert_6', 'convert_7', 'convert_8', 'convert_9'])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:05:29.573010Z","iopub.execute_input":"2022-07-17T21:05:29.573362Z","iopub.status.idle":"2022-07-17T21:05:29.579326Z","shell.execute_reply.started":"2022-07-17T21:05:29.573330Z","shell.execute_reply":"2022-07-17T21:05:29.578079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,10):\n    convert_list=[]\n    for c in range(7):\n        temp = submission[submission['Predicted_0'] == c]\n        temp_index = temp[f'Predicted_{i}'].value_counts().index[0]\n        temp_value = temp[f'Predicted_{i}'].value_counts()[temp_index]\n        convert_list.append(temp_index)\n    convert_df[f'convert_{i}'] = convert_list","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:06:07.300205Z","iopub.execute_input":"2022-07-17T21:06:07.300585Z","iopub.status.idle":"2022-07-17T21:06:07.431438Z","shell.execute_reply.started":"2022-07-17T21:06:07.300554Z","shell.execute_reply":"2022-07-17T21:06:07.430429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert_df","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:06:10.171033Z","iopub.execute_input":"2022-07-17T21:06:10.171380Z","iopub.status.idle":"2022-07-17T21:06:10.190190Z","shell.execute_reply.started":"2022-07-17T21:06:10.171349Z","shell.execute_reply":"2022-07-17T21:06:10.189203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_df = submission.copy()\npredict_df2 = submission.copy()\nfor i in range(1,10):\n    for j in range(7):\n        convert_no = convert_df[f'convert_{i}'][j]\n        index_no = predict_df[predict_df[f'Predicted_{i}']==convert_no].index.tolist()\n        predict_df2[f'Predicted_{i}'].iloc[index_no] = j","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:07:01.322826Z","iopub.execute_input":"2022-07-17T21:07:01.323573Z","iopub.status.idle":"2022-07-17T21:07:01.786793Z","shell.execute_reply.started":"2022-07-17T21:07:01.323522Z","shell.execute_reply":"2022-07-17T21:07:01.785703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_df3 = predict_df2.drop(['Id', 'Predicted'], axis=1)\npredict_mode = predict_df3.mode(axis=1)\nsubmission = sub.copy()\nsubmission['Predicted'] = predict_mode[0]\nsubmission.to_csv(\"/kaggle/working/submission_bgmm_majority_decision.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:07:05.391572Z","iopub.execute_input":"2022-07-17T21:07:05.392513Z","iopub.status.idle":"2022-07-17T21:07:43.542750Z","shell.execute_reply.started":"2022-07-17T21:07:05.392452Z","shell.execute_reply":"2022-07-17T21:07:43.541790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BGMM with the probability averaging\n\n次にprobabilityの平均を求める方法","metadata":{}},{"cell_type":"code","source":"submission = sub.copy()\ni=0#randomstate\nbgmm = BayesianGaussianMixture(n_components=7, covariance_type='full', max_iter=100, random_state=i, n_init = 5)\nbgmm.fit(pt_scaled_df[use_col])\npredict = bgmm.predict_proba(pt_scaled_df[use_col])\ntemp = pd.DataFrame(predict, columns=[f'{i}_No0',f'{i}_No1', f'{i}_No2', f'{i}_No3', f'{i}_No4', f'{i}_No5', f'{i}_No6'])\nsubmission = pd.concat([submission, temp], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:07:43.544662Z","iopub.execute_input":"2022-07-17T21:07:43.545020Z","iopub.status.idle":"2022-07-17T21:10:41.364512Z","shell.execute_reply.started":"2022-07-17T21:07:43.544983Z","shell.execute_reply":"2022-07-17T21:10:41.363332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1, 10):#randomstate\n    bgmm = BayesianGaussianMixture(n_components=7, covariance_type='full', max_iter=100, random_state=i, n_init = 5)\n    bgmm.fit(pt_scaled_df[use_col])\n    predict = bgmm.predict_proba(pt_scaled_df[use_col])\n    con0 = convert_df[convert_df[f'convert_{i}'] == 0].index[0]\n    con1 = convert_df[convert_df[f'convert_{i}'] == 1].index[0]\n    con2 = convert_df[convert_df[f'convert_{i}'] == 2].index[0]\n    con3 = convert_df[convert_df[f'convert_{i}'] == 3].index[0]\n    con4 = convert_df[convert_df[f'convert_{i}'] == 4].index[0]\n    con5 = convert_df[convert_df[f'convert_{i}'] == 5].index[0]\n    con6 = convert_df[convert_df[f'convert_{i}'] == 6].index[0]\n    temp = pd.DataFrame(predict, columns=[f'{i}_No{con0}',f'{i}_No{con1}', f'{i}_No{con2}', f'{i}_No{con3}', f'{i}_No{con4}', f'{i}_No{con5}', f'{i}_No{con6}'])\n    submission = pd.concat([submission, temp], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:10:41.371033Z","iopub.execute_input":"2022-07-17T21:10:41.373804Z","iopub.status.idle":"2022-07-17T21:34:46.985371Z","shell.execute_reply.started":"2022-07-17T21:10:41.373755Z","shell.execute_reply":"2022-07-17T21:34:46.984030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"No0_col = [c for c in submission.columns if 'No0' in c]\nNo1_col = [c for c in submission.columns if 'No1' in c]\nNo2_col = [c for c in submission.columns if 'No2' in c]\nNo3_col = [c for c in submission.columns if 'No3' in c]\nNo4_col = [c for c in submission.columns if 'No4' in c]\nNo5_col = [c for c in submission.columns if 'No5' in c]\nNo6_col = [c for c in submission.columns if 'No6' in c]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:46.987033Z","iopub.execute_input":"2022-07-17T21:34:46.987366Z","iopub.status.idle":"2022-07-17T21:34:47.001758Z","shell.execute_reply.started":"2022-07-17T21:34:46.987328Z","shell.execute_reply":"2022-07-17T21:34:47.000038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"No_list = [No0_col, No1_col, No2_col, No3_col, No4_col, No5_col, No6_col]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:47.005589Z","iopub.execute_input":"2022-07-17T21:34:47.006019Z","iopub.status.idle":"2022-07-17T21:34:47.012569Z","shell.execute_reply.started":"2022-07-17T21:34:47.005985Z","shell.execute_reply":"2022-07-17T21:34:47.011174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_mean = sub.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:47.013908Z","iopub.execute_input":"2022-07-17T21:34:47.014334Z","iopub.status.idle":"2022-07-17T21:34:47.023505Z","shell.execute_reply.started":"2022-07-17T21:34:47.014297Z","shell.execute_reply":"2022-07-17T21:34:47.022501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, l in enumerate(No_list):\n    submission_mean[f'No{i}_mean'] = submission[l].mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:47.024503Z","iopub.execute_input":"2022-07-17T21:34:47.025202Z","iopub.status.idle":"2022-07-17T21:34:47.109846Z","shell.execute_reply.started":"2022-07-17T21:34:47.025162Z","shell.execute_reply":"2022-07-17T21:34:47.108854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_mean","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:47.112296Z","iopub.execute_input":"2022-07-17T21:34:47.113055Z","iopub.status.idle":"2022-07-17T21:34:47.138434Z","shell.execute_reply.started":"2022-07-17T21:34:47.113016Z","shell.execute_reply":"2022-07-17T21:34:47.137613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_list = [c for c in submission_mean.columns if 'mean' in c]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:47.140547Z","iopub.execute_input":"2022-07-17T21:34:47.141112Z","iopub.status.idle":"2022-07-17T21:34:47.146032Z","shell.execute_reply.started":"2022-07-17T21:34:47.141075Z","shell.execute_reply":"2022-07-17T21:34:47.145077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#columnの中で最も大きいcolumnを選択する。\nsubmission_mean['Predicted'] = submission_mean[mean_list].idxmax(axis=1).str[2]\nsubmission2 = submission_mean[['Id', 'Predicted']]\nsubmission2.to_csv(\"/kaggle/working/submission_bgmm_mean_proba.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:40:13.567323Z","iopub.execute_input":"2022-07-17T22:40:13.568028Z","iopub.status.idle":"2022-07-17T22:40:13.924191Z","shell.execute_reply.started":"2022-07-17T22:40:13.567988Z","shell.execute_reply":"2022-07-17T22:40:13.923213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The method using the probability averaging had a slightly better LB score than the majority voting method.","metadata":{}},{"cell_type":"markdown","source":"# Visualization by UMAP","metadata":{}},{"cell_type":"markdown","source":"confirm the number of each clustering   \n   \nクラスタリングされた数の確認","metadata":{}},{"cell_type":"code","source":"submission_mean['proba'] = submission_mean[mean_list].max(axis=1)\nplt.bar(submission_mean['Predicted'].value_counts().sort_index().index, submission_mean['Predicted'].value_counts().sort_index())\nplt.xlabel('cluster_No')\nplt.ylabel('counts')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:34:47.549121Z","iopub.execute_input":"2022-07-17T21:34:47.549741Z","iopub.status.idle":"2022-07-17T21:34:47.785634Z","shell.execute_reply.started":"2022-07-17T21:34:47.549700Z","shell.execute_reply":"2022-07-17T21:34:47.784541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I check how much the data ratio changes depending on the threshold of probability.\n\n閾値によってどれぐらいデータ割合が変わるかを確認した","metadata":{}},{"cell_type":"code","source":"ratio_list=[]\nfor i in range(7):\n    for threshold in range(1,10):\n        temp = submission_mean[submission_mean['Predicted']==f'{i}']\n        ratio = len(temp[temp['proba'] > threshold/10])/len(temp)\n        ratio_list.append([i, threshold/10, ratio])\nratio_list_df = pd.DataFrame(ratio_list, columns=['cluster_no', 'threshold', 'ratio'])\nfig = plt.figure(figsize=(18,8))\nfig.suptitle('threshold of probability - the data ratio', fontsize =16)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i in range(7):\n    plt.subplot(2, 4, i+1)  \n    temp = ratio_list_df[ratio_list_df['cluster_no']==i]\n    plt.plot(temp['threshold'], temp['ratio'])\n    plt.title(f'cluster_No{i}')\n    plt.xlabel('threshold of probability')\n    plt.ylabel('ratio of data')\n    plt.ylim(0,1.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:25:48.054964Z","iopub.execute_input":"2022-07-17T22:25:48.055324Z","iopub.status.idle":"2022-07-17T22:25:49.736384Z","shell.execute_reply.started":"2022-07-17T22:25:48.055294Z","shell.execute_reply":"2022-07-17T22:25:49.735398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Are they really clustering well with a high threshold?","metadata":{}},{"cell_type":"code","source":"pt_scaled_df_umap = pt_scaled_df.copy()\npt_scaled_df_umap[['Predicted', 'proba']] = submission_mean[['Predicted', 'proba']]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T21:48:00.024909Z","iopub.execute_input":"2022-07-17T21:48:00.025264Z","iopub.status.idle":"2022-07-17T21:48:00.045178Z","shell.execute_reply.started":"2022-07-17T21:48:00.025226Z","shell.execute_reply":"2022-07-17T21:48:00.044146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for threshold in range(4,10,1):\n    import umap\n    from scipy.sparse.csgraph import connected_components\n    umap = umap.UMAP(n_components=2, random_state=0)\n    temp = pt_scaled_df_umap[pt_scaled_df_umap['proba']> threshold/10]\n    X_umap = umap.fit_transform(temp.drop(['Predicted', 'proba'], axis=1))\n    temp[['X_umap_0', 'X_umap_1']] = X_umap\n    plt.scatter(temp['X_umap_0'], temp['X_umap_1'], c=temp.Predicted.astype(int), s=1)\n    plt.title(f'proba_threshold = {threshold/10}')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:29:42.632941Z","iopub.execute_input":"2022-07-17T22:29:42.633294Z","iopub.status.idle":"2022-07-17T22:37:24.071433Z","shell.execute_reply.started":"2022-07-17T22:29:42.633262Z","shell.execute_reply":"2022-07-17T22:37:24.070426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It certainly appears to cluster well with high threshold data.","metadata":{}},{"cell_type":"markdown","source":"# lightGBM modeling","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import train_test_split\nlightgbm_df = df.copy()\nlightgbm_df[['proba', 'Predicted']] = submission_mean[['proba', 'Predicted']]\nlightgbm_df['Predicted'] = lightgbm_df['Predicted'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:12:02.488862Z","iopub.execute_input":"2022-07-17T22:12:02.489279Z","iopub.status.idle":"2022-07-17T22:12:04.516192Z","shell.execute_reply.started":"2022-07-17T22:12:02.489241Z","shell.execute_reply":"2022-07-17T22:12:04.515206Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.8\ntemp_train = lightgbm_df[lightgbm_df['proba']>threshold]\ntemp_test = lightgbm_df[lightgbm_df['proba']<=threshold]\nresult_df = temp_test[['id']]\ndel temp_test['id']\ndel temp_train['id']\n\ny = temp_train['Predicted']\nX = temp_train.drop(['Predicted', 'proba'], axis = 1)\n\ntrain_X, test_X, train_y, test_y = train_test_split(X, y, test_size=0.3, random_state=100)\ntrain_set = lgb.Dataset(train_X, train_y)\nvalid_set = lgb.Dataset(test_X, test_y)\n\nparams = {\n    \"objective\" : \"multiclass\",\n    \"metric\" : \"multi_logloss\",\n    \"num_class\" : 7\n}\nresult_data = {}\nmodel = lgb.train(\n    params = params,\n    train_set = train_set,\n    valid_sets = [train_set, valid_set],\n    num_boost_round = 1000,\n    early_stopping_rounds = 5,\n    verbose_eval = 50,\n    evals_result = result_data\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:16:26.198014Z","iopub.execute_input":"2022-07-17T22:16:26.198363Z","iopub.status.idle":"2022-07-17T22:16:47.005000Z","shell.execute_reply.started":"2022-07-17T22:16:26.198332Z","shell.execute_reply":"2022-07-17T22:16:47.004110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import precision_score, recall_score\npred_test = model.predict(test_X)\npred_test = pred_test.argmax(axis = 1)\nacc = accuracy_score(test_y.astype(int), pred_test)\n\n\nprecision = precision_score(test_y.astype(int), pred_test, average = \"micro\")\nrecall = recall_score(test_y.astype(int), pred_test, average = \"micro\")\nprint(f'Accuracy score:{acc}, Precision:{precision}, Recall:{recall}')\n\ndisplay(confusion_matrix(test_y.astype(int), pred_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:37:24.073328Z","iopub.execute_input":"2022-07-17T22:37:24.074317Z","iopub.status.idle":"2022-07-17T22:37:26.377124Z","shell.execute_reply.started":"2022-07-17T22:37:24.074276Z","shell.execute_reply":"2022-07-17T22:37:26.375884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_temp_test = model.predict(temp_test.drop(['Predicted', 'proba'], axis=1))\npred_temp_test = pred_temp_test.argmax(axis = 1)\nresult_df['Predicted'] = pred_temp_test\nresult_df = result_df.rename(columns={'id':'Id'})\nsubmission3 = submission2.copy()\nsubmission3.update(result_df)\nsubmission3['Predicted'] = submission3['Predicted'].astype(int)\nsubmission3['Id'] = submission3['Id'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:22:48.392965Z","iopub.execute_input":"2022-07-17T22:22:48.393373Z","iopub.status.idle":"2022-07-17T22:22:53.541278Z","shell.execute_reply.started":"2022-07-17T22:22:48.393338Z","shell.execute_reply":"2022-07-17T22:22:53.540298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission3.to_csv(\"/kaggle/working/submission_bgmm_lightgbm.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T22:23:01.241865Z","iopub.execute_input":"2022-07-17T22:23:01.242215Z","iopub.status.idle":"2022-07-17T22:23:01.388176Z","shell.execute_reply.started":"2022-07-17T22:23:01.242185Z","shell.execute_reply":"2022-07-17T22:23:01.387252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Next\nOptimization of thresholds used for LightGBM may result in higher scores","metadata":{}}]}