{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\ntrain=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nsubmission=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:57.263954Z","iopub.execute_input":"2024-02-19T10:10:57.264374Z","iopub.status.idle":"2024-02-19T10:10:57.945046Z","shell.execute_reply.started":"2024-02-19T10:10:57.264341Z","shell.execute_reply":"2024-02-19T10:10:57.943768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['eeg_id'] != 1628180742].head()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:57.947118Z","iopub.execute_input":"2024-02-19T10:10:57.947558Z","iopub.status.idle":"2024-02-19T10:10:57.976687Z","shell.execute_reply.started":"2024-02-19T10:10:57.947523Z","shell.execute_reply":"2024-02-19T10:10:57.975481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:57.978013Z","iopub.execute_input":"2024-02-19T10:10:57.978444Z","iopub.status.idle":"2024-02-19T10:10:58.014400Z","shell.execute_reply.started":"2024-02-19T10:10:57.978410Z","shell.execute_reply":"2024-02-19T10:10:58.012858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.016709Z","iopub.execute_input":"2024-02-19T10:10:58.017362Z","iopub.status.idle":"2024-02-19T10:10:58.033724Z","shell.execute_reply.started":"2024-02-19T10:10:58.017331Z","shell.execute_reply":"2024-02-19T10:10:58.032481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.035241Z","iopub.execute_input":"2024-02-19T10:10:58.035869Z","iopub.status.idle":"2024-02-19T10:10:58.136165Z","shell.execute_reply.started":"2024-02-19T10:10:58.035830Z","shell.execute_reply":"2024-02-19T10:10:58.134991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.137664Z","iopub.execute_input":"2024-02-19T10:10:58.137986Z","iopub.status.idle":"2024-02-19T10:10:58.147604Z","shell.execute_reply.started":"2024-02-19T10:10:58.137957Z","shell.execute_reply":"2024-02-19T10:10:58.146324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.149177Z","iopub.execute_input":"2024-02-19T10:10:58.149718Z","iopub.status.idle":"2024-02-19T10:10:58.161867Z","shell.execute_reply.started":"2024-02-19T10:10:58.149679Z","shell.execute_reply":"2024-02-19T10:10:58.160402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.163302Z","iopub.execute_input":"2024-02-19T10:10:58.164059Z","iopub.status.idle":"2024-02-19T10:10:58.180803Z","shell.execute_reply.started":"2024-02-19T10:10:58.164016Z","shell.execute_reply":"2024-02-19T10:10:58.179496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'Data shape: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['Data type'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns={'index': 'Features'})\n    summary['Number of Nan'] = df.isnull().sum().values\n    summary['Number of Unique values'] = df.nunique().values\n    summary['First value'] = df.loc[0].values\n    summary['Second value'] = df.loc[1].values\n    \n    return summary\n   \nsummary = resumetable(train)\nsummary","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:12:56.601862Z","iopub.execute_input":"2024-02-19T10:12:56.602377Z","iopub.status.idle":"2024-02-19T10:12:56.654272Z","shell.execute_reply.started":"2024-02-19T10:12:56.602332Z","shell.execute_reply":"2024-02-19T10:12:56.652837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_copy = train.copy()\nt_copy =t_copy.sort_values(['eeg_id'])\nt_copy","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.232463Z","iopub.execute_input":"2024-02-19T10:10:58.233540Z","iopub.status.idle":"2024-02-19T10:10:58.278187Z","shell.execute_reply.started":"2024-02-19T10:10:58.233506Z","shell.execute_reply":"2024-02-19T10:10:58.277136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print([t_copy['eeg_id'].min(),t_copy['eeg_id'].max()])\nprint([t_copy['spectrogram_id'].min(),t_copy['spectrogram_id'].max()])\nprint([t_copy['patient_id'].min(),t_copy['patient_id'].max()])","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.279527Z","iopub.execute_input":"2024-02-19T10:10:58.279899Z","iopub.status.idle":"2024-02-19T10:10:58.288564Z","shell.execute_reply.started":"2024-02-19T10:10:58.279868Z","shell.execute_reply":"2024-02-19T10:10:58.287477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats\nimport seaborn as sns\nimport matplotlib.pyplot as plt #시각화 툴\nstats.pearsonr(train['eeg_id'],train['spectrogram_id'] )","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:58.290144Z","iopub.execute_input":"2024-02-19T10:10:58.290965Z","iopub.status.idle":"2024-02-19T10:10:58.999368Z","shell.execute_reply.started":"2024-02-19T10:10:58.290915Z","shell.execute_reply":"2024-02-19T10:10:58.997646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.pearsonr(t_copy['eeg_id'],t_copy['spectrogram_id'] )","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:59.005059Z","iopub.execute_input":"2024-02-19T10:10:59.005757Z","iopub.status.idle":"2024-02-19T10:10:59.026431Z","shell.execute_reply.started":"2024-02-19T10:10:59.005709Z","shell.execute_reply":"2024-02-19T10:10:59.025222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x =t_copy['eeg_id'], y=t_copy['spectrogram_id'])","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:59.028439Z","iopub.execute_input":"2024-02-19T10:10:59.029025Z","iopub.status.idle":"2024-02-19T10:10:59.395886Z","shell.execute_reply.started":"2024-02-19T10:10:59.028980Z","shell.execute_reply":"2024-02-19T10:10:59.394917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x =t_copy['eeg_id'], y=t_copy['spectrogram_id'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:59.397206Z","iopub.execute_input":"2024-02-19T10:10:59.397547Z","iopub.status.idle":"2024-02-19T10:10:59.877954Z","shell.execute_reply.started":"2024-02-19T10:10:59.397519Z","shell.execute_reply":"2024-02-19T10:10:59.876939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(t_copy[['eeg_id','spectrogram_id','patient_id']])\nplt.show","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:10:59.879484Z","iopub.execute_input":"2024-02-19T10:10:59.880361Z","iopub.status.idle":"2024-02-19T10:11:05.245730Z","shell.execute_reply.started":"2024-02-19T10:10:59.880302Z","shell.execute_reply":"2024-02-19T10:11:05.244665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_copy = t_copy.drop_duplicates('eeg_id')\nt_copy","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:05.247639Z","iopub.execute_input":"2024-02-19T10:11:05.248063Z","iopub.status.idle":"2024-02-19T10:11:05.275855Z","shell.execute_reply.started":"2024-02-19T10:11:05.248028Z","shell.execute_reply":"2024-02-19T10:11:05.274586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x=t_copy['eeg_id'])","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:05.277307Z","iopub.execute_input":"2024-02-19T10:11:05.277702Z","iopub.status.idle":"2024-02-19T10:11:05.671317Z","shell.execute_reply.started":"2024-02-19T10:11:05.277669Z","shell.execute_reply":"2024-02-19T10:11:05.670162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(t_copy['eeg_id'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:05.672854Z","iopub.execute_input":"2024-02-19T10:11:05.673571Z","iopub.status.idle":"2024-02-19T10:11:05.934160Z","shell.execute_reply.started":"2024-02-19T10:11:05.673539Z","shell.execute_reply":"2024-02-19T10:11:05.932857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(ncols=3, figsize=(15,5))\n\nsns.boxplot(t_copy['eeg_id'],ax=ax[0])\nsns.boxplot(t_copy['spectrogram_id'],ax=ax[1])\nsns.boxplot(t_copy['patient_id'],ax=ax[2])","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:05.935656Z","iopub.execute_input":"2024-02-19T10:11:05.936050Z","iopub.status.idle":"2024-02-19T10:11:06.327354Z","shell.execute_reply.started":"2024-02-19T10:11:05.936020Z","shell.execute_reply":"2024-02-19T10:11:06.325994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(t_copy['eeg_id'])","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:06.329019Z","iopub.execute_input":"2024-02-19T10:11:06.329401Z","iopub.status.idle":"2024-02-19T10:11:06.680843Z","shell.execute_reply.started":"2024-02-19T10:11:06.329355Z","shell.execute_reply":"2024-02-19T10:11:06.679690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_copy2 = train.copy()\nt_copy2","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:06.682473Z","iopub.execute_input":"2024-02-19T10:11:06.683074Z","iopub.status.idle":"2024-02-19T10:11:06.713888Z","shell.execute_reply.started":"2024-02-19T10:11:06.683019Z","shell.execute_reply":"2024-02-19T10:11:06.712487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_drop = t_copy2.drop_duplicates(['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote' ])\nt_drop","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:06.718205Z","iopub.execute_input":"2024-02-19T10:11:06.718610Z","iopub.status.idle":"2024-02-19T10:11:06.747492Z","shell.execute_reply.started":"2024-02-19T10:11:06.718580Z","shell.execute_reply":"2024-02-19T10:11:06.746523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import missingno as msno\nt_copy2 = train.copy()\nmsno.bar(df=t_copy2.iloc[:, :], figsize=(13, 6))","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:06.748756Z","iopub.execute_input":"2024-02-19T10:11:06.749059Z","iopub.status.idle":"2024-02-19T10:11:07.995276Z","shell.execute_reply.started":"2024-02-19T10:11:06.749034Z","shell.execute_reply":"2024-02-19T10:11:07.994145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:07.996738Z","iopub.execute_input":"2024-02-19T10:11:07.997147Z","iopub.status.idle":"2024-02-19T10:11:08.004201Z","shell.execute_reply.started":"2024-02-19T10:11:07.997111Z","shell.execute_reply":"2024-02-19T10:11:08.003152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    '''도형 객체를 순회하며 막대 그래프 상단에 타깃값 비율 표시'''\n    for patch in ax.patches:\n        height = patch.get_height()     # 도형 높이(데이터 개수)\n        width = patch.get_width()       # 도형 너비\n        left_coord = patch.get_x()      # 도형 왼쪽 테두리의 x축 위치\n        percent = height/total_size*100 # 타깃값 비율\n        \n        # (x, y) 좌표에 텍스트 입력\n        ax.text(left_coord + width/2.0,     # x축 위치\n                height + total_size*0.001,  # y축 위치\n                '{:1.3f}%'.format(percent), # 입력 텍스트\n                ha='center')                # 가운데 정렬\n    \nmpl.rc('font', size=15)\nplt.figure(figsize=(7, 6))\nax = sns.countplot(x='expert_consensus', data=train)\nwrite_percent(ax, len(train)) # 비율 표시\nax.set_title('Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:08.005884Z","iopub.execute_input":"2024-02-19T10:11:08.006292Z","iopub.status.idle":"2024-02-19T10:11:08.347050Z","shell.execute_reply.started":"2024-02-19T10:11:08.006255Z","shell.execute_reply":"2024-02-19T10:11:08.345939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'Data shape: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['Data type'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns={'index': 'Features'})\n    summary['Number of Nan'] = df.isnull().sum().values\n    summary['Number of Unique values'] = df.nunique().values\n    summary['First value'] = df.loc[0].values\n    summary['Second value'] = df.loc[1].values\n    \n    return summary\n   \nsummary = resumetable(train)\nsummary","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:08.348456Z","iopub.execute_input":"2024-02-19T10:11:08.348813Z","iopub.status.idle":"2024-02-19T10:11:08.403132Z","shell.execute_reply.started":"2024-02-19T10:11:08.348782Z","shell.execute_reply":"2024-02-19T10:11:08.402028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec\n\ndef plot_target_ratio_by_features(df, features, num_rows, num_cols, \n                                  size=(12, 18)):\n    mpl.rc('font', size=9) \n    plt.figure(figsize=size)                     # 전체 Figure 크기 설정\n    grid = gridspec.GridSpec(num_rows, num_cols) # 서브플롯 배치\n    plt.subplots_adjust(wspace=0.3, hspace=0.3)  # 서브플롯 좌우/상하 여백 설정\n\n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        # ax축에 고윳값별 타깃값 1 비율을 막대 그래프로 그리기\n        sns.barplot(x=feature, y='expert_consensus', data=df, palette='Set2', ax=ax)\n\nt_copy3 = train.drop('expert_consensus',axis=1)\nbin_features = t_copy3.columns\n# 이진 피처 고윳값별 타깃값 1 비율을 막대 그래프로 그리기\nplot_target_ratio_by_features(train, bin_features, 6, 3) # 6행 3열 배치","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:08.409747Z","iopub.execute_input":"2024-02-19T10:11:08.410132Z","iopub.status.idle":"2024-02-19T10:11:27.534366Z","shell.execute_reply.started":"2024-02-19T10:11:08.410105Z","shell.execute_reply":"2024-02-19T10:11:27.533122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\ncont_corr = t_copy3[bin_features].corr()     # corr with every feature except target\nsns.heatmap(cont_corr, annot=True, cmap='OrRd'); # 히트맵 그리기\nplt.savefig('상관계수1.jpg', format='jpeg')","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:27.535818Z","iopub.execute_input":"2024-02-19T10:11:27.536142Z","iopub.status.idle":"2024-02-19T10:11:28.775877Z","shell.execute_reply.started":"2024-02-19T10:11:27.536116Z","shell.execute_reply":"2024-02-19T10:11:28.774490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature = ['Seizure', 'GPD', 'LRDA', 'Other', 'GRDA', 'LPD']\n\nt_copy4 = train.copy()\n\n# Create a dictionary to map feature names to their index numbers\nfeature_index = {feat: idx for idx, feat in enumerate(feature)}\n\n# Apply the mapping to create a new column 'transform'\nt_copy4['transform'] = t_copy4['expert_consensus'].map(feature_index)\n\nt_copy4\n","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:28.777631Z","iopub.execute_input":"2024-02-19T10:11:28.778660Z","iopub.status.idle":"2024-02-19T10:11:28.811902Z","shell.execute_reply.started":"2024-02-19T10:11:28.778623Z","shell.execute_reply":"2024-02-19T10:11:28.810722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nt_copy4 = t_copy4.drop('expert_consensus',axis=1)\ncol = t_copy4.columns\ncont_corr = t_copy4[col].corr()     # corr with target(transform)\nsns.heatmap(cont_corr, annot=True, cmap='OrRd'); # 히트맵 그리기\nplt.savefig('상관계수2.jpg', format='jpeg')","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:28.813108Z","iopub.execute_input":"2024-02-19T10:11:28.813830Z","iopub.status.idle":"2024-02-19T10:11:30.175005Z","shell.execute_reply.started":"2024-02-19T10:11:28.813799Z","shell.execute_reply":"2024-02-19T10:11:30.174202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\ncol = t_copy4.columns\ncont_corr = t_copy4[col].corr(method='spearman')     # 연속형 피처 간 상관관계 \nsns.heatmap(cont_corr, annot=True, cmap='OrRd'); # 히트맵 그리기\nplt.savefig('상관계수2.jpg', format='jpeg')","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:30.176020Z","iopub.execute_input":"2024-02-19T10:11:30.176978Z","iopub.status.idle":"2024-02-19T10:11:31.874333Z","shell.execute_reply.started":"2024-02-19T10:11:30.176937Z","shell.execute_reply":"2024-02-19T10:11:31.873527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터 세트 형상: {df.shape}') \n    summary = pd.DataFrame(df.dtypes, columns=['데이터 타입']) #데이터 타입별로 서머리\n    summary = summary.reset_index() #그렇게 구한 서버리를 index 리셋\n    summary = summary.rename(columns={'index': '피처'}) #피처를 \n    summary['결측값 개수'] = df.isnull().sum().values #결측값 개수 열 추가\n    summary['고윳값 개수'] = df.nunique().values #고윳값 개수 열 추가\n    summary['첫 번째 값'] = df.loc[0].values #첫째값\n    summary['두 번째 값'] = df.loc[1].values #둘째값\n    \n    return summary\n   \nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:31.875622Z","iopub.execute_input":"2024-02-19T10:11:31.876269Z","iopub.status.idle":"2024-02-19T10:11:31.929281Z","shell.execute_reply.started":"2024-02-19T10:11:31.876240Z","shell.execute_reply":"2024-02-19T10:11:31.928255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터 세트 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['데이터 타입'])\n    \n    summary['결측값 개수'] = df.isnull().sum().values #결측값 개수 열 추가\n    summary['고윳값 개수'] = df.nunique().values\n    summary['데이터 종류'] = None\n    for col in df.columns: #데이터 종류 추가\n        if 'id' in col:\n            summary.loc[col, '데이터 종류'] = 'id형'\n        elif df[col].dtype == object:\n            summary.loc[col, '데이터 종류'] = '명목형'\n        elif df[col].dtype == float:\n            summary.loc[col, '데이터 종류'] = '연속형'\n        elif 'vote' in col:\n            summary.loc[col, '데이터 종류'] = '투표형'\n    summary['첫 번째 값'] = df.loc[0].values #첫째값\n    summary['두 번째 값'] = df.loc[1].values #둘째값\n    return summary\n\n\nsummary=resumetable(train)\nsummary","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:31.930778Z","iopub.execute_input":"2024-02-19T10:11:31.931198Z","iopub.status.idle":"2024-02-19T10:11:31.988371Z","shell.execute_reply.started":"2024-02-19T10:11:31.931170Z","shell.execute_reply":"2024-02-19T10:11:31.987460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_features = summary[summary['데이터 종류'] == 'id형'].index\n\ndef plot_target_ratio_by_features(df, features, num_rows, num_cols, \n                                  size=(12, 18)):\n    mpl.rc('font', size=9) \n    plt.figure(figsize=size)                     # 전체 Figure 크기 설정\n    grid = gridspec.GridSpec(num_rows, num_cols) # 서브플롯 배치\n    plt.subplots_adjust(wspace=0.3, hspace=0.3)  # 서브플롯 좌우/상하 여백 설정\n\n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        # ax축에 고윳값별 타깃값 1 비율을 막대 그래프로 그리기\n        sns.barplot(x=feature, y='expert_consensus', data=df, palette='Set2', ax=ax)\n\nt_copy5= train.copy()\nprint(id_features)\nfor idx, id_features in enumerate(id_features):\n    # 값을 5개 구간으로 나누기\n    t_copy5[id_features] = pd.cut(t_copy5[id_features], 5)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:31.989740Z","iopub.execute_input":"2024-02-19T10:11:31.990041Z","iopub.status.idle":"2024-02-19T10:11:32.034822Z","shell.execute_reply.started":"2024-02-19T10:11:31.990016Z","shell.execute_reply":"2024-02-19T10:11:32.033290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\ntarget_votes = Counter(list(train['expert_consensus']))\ntarget_votes = {f\"{k.lower()}_vote\":v for k,v in target_votes.items()}\ntotal_votes = sum([v for _,v in target_votes.items()])\nmean_vote_ratio = {k:(target_votes[k]/total_votes) for k,target in target_votes.items()}\nprint(target_votes)\nprint(total_votes)\nprint(mean_vote_ratio)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T10:11:32.036511Z","iopub.execute_input":"2024-02-19T10:11:32.036889Z","iopub.status.idle":"2024-02-19T10:11:32.060969Z","shell.execute_reply.started":"2024-02-19T10:11:32.036859Z","shell.execute_reply":"2024-02-19T10:11:32.059815Z"},"trusted":true},"execution_count":null,"outputs":[]}]}