{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Melanoma Detection:\nGenerally in any medical image diagnosis Machine Learning problems, the number of positive labelled data will be less compared to negative labelled data since the number of people suffering from the disease will be less compared to number of people tested.It is no different in our current dataset.\n\nThe number of images corresponding to benign tumours is 98% which leads to huge Class Imbalance Problem.\n\nThere are various techniques for handling Class Imbalance.The one used is this kernel is ***UnderSampling***.\nUnderSampling in simple terms can be thought of as reducing the number of data points corresponding to the class which has significantly more data points in a class imbalance scenario\n\n![](http://)\n\n\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Installing Necessary Packages\n\n!pip install efficientnet\n!pip install sweetviz","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport albumentations\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn import metrics\nfrom sklearn import model_selection\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport plotly.graph_objects as go\nimport cv2\nimport efficientnet.tfkeras as efn \nimport tensorflow.keras.layers as L\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.callbacks import (ModelCheckpoint, LearningRateScheduler,\n                                        EarlyStopping, ReduceLROnPlateau, CSVLogger)\nimport math\nimport sweetviz as sv\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import KFold","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create a dataframe out of train csv file\ndf = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")\ndf_test = pd.read_csv(\"../input/siim-isic-melanoma-classification/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels=df['diagnosis'].value_counts().index[1:]\nvalues=df['diagnosis'].value_counts().values[1:]\nfig = go.Figure(data=[go.Pie(labels=labels, values=values, textinfo='label+percent',\n                             insidetextorientation='radial'\n                            )])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels=df['anatom_site_general_challenge'].value_counts().index\nvalues=df['anatom_site_general_challenge'].value_counts().values\nfig = go.Figure(data=[go.Pie(labels=labels, values=values, textinfo='label+percent',\n                             insidetextorientation='radial'\n                            )])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 欠損値処理、testにないcolの消去","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"diagnosisの扱いについてはまた今度考える","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df=df.drop(['benign_malignant','diagnosis'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.isnull().sum()/len(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.isnull().sum()/len(df_test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 欠損値代入","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df['age_approx'] = df['age_approx'].fillna(df['age_approx'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 特徴量","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp = df.groupby('patient_id').size()\ndf['exam_num'] = df['patient_id'].map(tmp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp = df_test.groupby('patient_id').size()\ndf_test['exam_num'] = df_test['patient_id'].map(tmp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['exam_freq'] = df['age_approx']/df['exam_num']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test['exam_freq'] = df_test['age_approx']/df_test['exam_num']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp = df.groupby('patient_id')['age_approx'].std()\ndf['exam_std'] = df['patient_id'].map(tmp).fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp = df_test.groupby('patient_id')['age_approx'].std()\ndf_test['exam_std'] = df_test['patient_id'].map(tmp).fillna(0)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### encording","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df['image_name'] = df['image_name']+'.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['target'] = df['target'].astype('int')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def to_sex_encord(word):\n    if word=='male':\n        return pd.Series([1,0])\n    if word=='female':\n        return pd.Series([0,1])\n    else:\n        return pd.Series([0.5,0.5])\n\ndf[['male','female']]=df['sex'].apply(to_sex_encord)\ndf_test[['male','female']]=df_test['sex'].apply(to_sex_encord)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['anatom_site_general_challenge2'] = df['anatom_site_general_challenge'].replace( {'head/neck':'other','palms/soles':'other', 'oral/genital':'other',np.nan:'other'})\ndf_test['anatom_site_general_challenge2'] = df_test['anatom_site_general_challenge'].replace( {'head/neck':'other','palms/soles':'other', 'oral/genital':'other',np.nan:'other'})\n\ndf = pd.get_dummies(df,columns=['anatom_site_general_challenge2'])\ndf_test = pd.get_dummies(df_test,columns=['anatom_site_general_challenge2'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.mean()[['age_approx','exam_num']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 正規化\ndef to_normalize( df,cols ):\n    df[cols] = ( df[cols] - df.mean()[cols] )/df.std()[cols]\n    return df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## UnderSampling","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"On looking into the dataset it can be noted that for the same person[Patient ID] and for the same region of the body[anatom_site_general_challenge] , there are multiple Images. Only one image per person per anatomy region only is used, the rest all are dropped for benign cases.The malignant datapoints are not touched.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"malignant →　584\n\nbenign →　32542\n\nbenign(一致例消去) → 6271","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def undersampling(df,n_models=1):\n    \n    if n_models!=1:\n        print('undersamplingはデータフレーム１つしかない')\n        \n    #一致例消去benignをn_splitsで分割してそれぞれにmalignant例を加えたdfのlistを返す\n    #これだけはdfsは１つしかないことに注意\n    dfs = []\n    \n    df_malignant = df[df['target'] == 1]\n    df_benign = df[df['target'] == 0].drop_duplicates(subset=['patient_id','anatom_site_general_challenge'], keep = \"first\")\n    \n    df_concat = pd.concat([df_malignant, df_benign]).reset_index(drop = True)\n    df_concat = df_concat.sample(frac=1).reset_index(drop=True)\n    \n    dfs.append(df_concat)\n    \n    return dfs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def undersampling2(df,n_models=6):\n    \n    #benignをn_splitsで分割してそれぞれにmalignant例を加えたdfのlistを返す\n    dfs = []\n    \n    df_malignant = df[df['target'] == 1]\n    df_benign = df[df['target'] == 0]\n    \n    kf = KFold(n_splits=n_models,shuffle=True)\n    for _,index in kf.split(df_benign):\n        df_benign2 = df_benign.iloc[index]\n        dfs.append( pd.concat([df_malignant, df_benign2]).sample(frac=1).reset_index(drop = True) )\n    \n    return dfs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def undersampling3(df,n_models=6):\n    \n    #benignをn_splitsで分割してそれぞれにmalignant例を加えたdfのlistを返す\n    #できるだけpatient_id+siteがかぶらないように\n    \n    dfs = []\n    \n    df_malignant = df[df['target'] == 1]\n    df_benign = df[df['target'] == 0]\n    df_benign['id'] = df_benign['patient_id']+(df_benign['anatom_site_general_challenge'].replace({np.nan:'NAN'}))\n    kf = StratifiedKFold(n_models)\n    df_benign = df_benign.sample(frac=1).reset_index(drop=True)\n\n    for _, (_, index) in enumerate(kf.split(X=df_benign, y=df_benign.id.values)):\n        df_benign2 = df_benign.iloc[index].drop('id',axis=1)\n        dfs.append( pd.concat([df_malignant, df_benign2]).sample(frac=1).reset_index(drop = True) )\n    \n    return dfs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"a = undersampling3(df)\nfor i in range(len(a)):\n    print(a[i].shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def undersampling4(df,n_models=6):\n    \n    #benignをn_splitsで分割してそれぞれにmalignant例を加えたdfのlistを返す\n    #できるだけpatient_id+siteがかぶらないように、そしてかぶりは消す\n    \n    dfs = []\n    \n    df_malignant = df[df['target'] == 1]\n    df_benign = df[df['target'] == 0]\n    df_benign['id'] = df_benign['patient_id']+(df_benign['anatom_site_general_challenge'].replace({np.nan:'NAN'}))\n    kf = StratifiedKFold(n_models)\n    df_benign = df_benign.sample(frac=1).reset_index(drop=True)\n\n    for _, (_, index) in enumerate(kf.split(X=df_benign, y=df_benign.id.values)):\n        df_benign2 = df_benign.iloc[index].drop_duplicates(subset='id', keep = \"first\").drop('id',axis=1)\n        dfs.append( pd.concat([df_malignant, df_benign2]).sample(frac=1).reset_index(drop = True) )\n    \n    return dfs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"a = undersampling4(df)\nfor i in range(len(a)):\n    print(a[i].shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def undersampling5(df ,normalized_cols ,n_models=6):\n    \n    #benignをn_splitsで分割してそれぞれにmalignant例を加えたdfのlistを返す\n    #できるだけpatient_id+siteがかぶらないように、そしてかぶりは消す\n    \n    dfs = []\n    \n    df_malignant = df[df['target'] == 1]\n    df_benign = df[df['target'] == 0]\n    df_benign['id'] = df_benign['patient_id']+(df_benign['anatom_site_general_challenge'].replace({np.nan:'NAN'}))\n    kf = StratifiedKFold(n_models)\n    df_benign = df_benign.sample(frac=1).reset_index(drop=True)\n\n    for _, (_, index) in enumerate(kf.split(X=df_benign, y=df_benign.id.values)):\n        df_benign2 = df_benign.iloc[index].drop_duplicates(subset='id', keep = \"first\").drop('id',axis=1)\n        dfs.append( to_normalize( pd.concat([df_malignant, df_benign2]) , normalized_cols ).sample(frac=1).reset_index(drop = True) )\n    \n    return dfs","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **生成部**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"data_col = ['age_approx','exam_num',\n       'exam_freq', 'exam_std', \n       'anatom_site_general_challenge2_lower extremity',\n       'anatom_site_general_challenge2_other',\n       'anatom_site_general_challenge2_torso',\n       'anatom_site_general_challenge2_upper extremity', 'male', 'female']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_models=1\n\ndfs = undersampling(df,n_models)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"\n# foldの作成","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Stratified KFold samples\ndef my_KFold( df , n_splits=5):\n\n    df = df.sample(frac=1).reset_index(drop=True)\n    \n    kf = StratifiedKFold(n_splits)\n    for fold, (_, val_ind) in enumerate(kf.split(X=df, y=df.target.values)):\n        df.loc[val_ind, 'fold'] = fold\n\n    df = df.sample(frac=1).reset_index(drop=True) # shuffling \n    \n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(n_models):\n    df_model = my_KFold( dfs[i] )\n    df_model.to_csv('df'+str(i)+'.csv' , index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_model.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test2 = df_test.copy()\ndf_test2['image_name'] = df_test2['image_name']+'.png'\ndf_test2['index'] =  np.arange(df_test2.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test2.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test2.to_csv('df_test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submit = pd.DataFrame({'image_name':df_test['image_name']})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submit.to_csv('df_submit.csv')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}