{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check patient_id column is valid feature\n\ntrain_df['p_id']=train_df['patient_id'].str[3:].astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['int_sex'] = 0\nfemale_ind=train_df[train_df['sex'] == 'female'].index\ntrain_df.loc[female_ind,'int_sex'] = 1\ntrain_df['int_benign_malignant'] = 0\nmalignant_ind = train_df[train_df['benign_malignant']=='malignant'].index\ntrain_df.loc[malignant_ind,'int_benign_malignant'] = 1\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[['int_benign_malignant','target']].corr()\ntrain_df.drop(['int_benign_malignant','benign_malignant'],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_df['patient_id'].str[:3].unique())\ntrain_df.drop('patient_id',axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_df[['p_id','target']].corr())\ntrain_df.drop('p_id',axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[train_df['sex'].isnull()]['age_approx'].unique()  # NULL sex row == NULL age row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[train_df['sex'].isnull()]['target'].unique()  # All benign row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(np.round(train_df['target'].value_counts()[1] / train_df.shape[0] * 100,2),'%  malignant')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby('sex')['target'].mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.hist(train_df['sex'].tolist())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.kdeplot(train_df['age_approx'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['anatom_site_general_challenge'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby('anatom_site_general_challenge')['target'].mean()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# UnderSampling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"target_rate = .1\nidx_0 = train_df[train_df.target==0].index\nidx_1 = train_df[train_df.target==1].index\n\nsampling_rate = (((1-target_rate)*len(train_df.loc[idx_1]))/(len(train_df.loc[idx_0])*target_rate))\nunder_sample_len = int(sampling_rate*len(train_df.loc[idx_0]))\nprint(sampling_rate)\nprint(under_sample_len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.utils import shuffle\nundersampled_idx = shuffle(idx_0,random_state=801, n_samples=under_sample_len)\nlen(undersampled_idx)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_idx = list(undersampled_idx)+list(idx_1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df = train_df.loc[all_idx].reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df['anatom_site_general_challenge'].fillna('NULL',inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.drop(undersampled_train_df[undersampled_train_df['sex'].isnull()].index,axis=0,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.drop(undersampled_train_df[undersampled_train_df['age_approx'].isnull()].index,axis=0,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.isnull().sum().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.drop(['sex','index'],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"undersampled_train_df.drop('diagnosis',axis=1,inplace=True)\noh_us_train_df=pd.get_dummies(undersampled_train_df,columns=['anatom_site_general_challenge'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nscaled_age=scaler.fit_transform(oh_us_train_df['age_approx'].values.reshape(-1,1))\noh_us_train_df['age'] = scaled_age\noh_us_train_df.drop('age_approx',axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_test, y_train, y_test = train_test_split(oh_us_train_df.drop('target',axis=1),oh_us_train_df['target'],\n                                                   test_size=0.05,stratify = oh_us_train_df['target'])\nprint(x_train.shape,y_train.shape,x_test.shape,y_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import GridSearchCV\nxgb = XGBClassifier(n_estimators=400,learning_rate=.1,max_depth=3,gpu_id=0)\nparams = {'max_depth':[2,3,5,10],'min_child_weight':[1,3,7],'colsample_bytree':[0.5,0.75],'n_estimators':[100,200,400]}\ngridcv = GridSearchCV(xgb,param_grid=params)\ngridcv.fit(x_train.drop('image_name',axis=1),y_train,eval_metric='auc',eval_set=[(x_train.drop('image_name',axis=1),y_train),\n                                                                                (x_test.drop('image_name',axis=1),y_test)])\nprint('best param',gridcv.best_params_)\npreds=gridcv.predict(x_test.drop('image_name',axis=1))\nprint(roc_auc_score(preds,y_test.tolist()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(24,24))\nsns.heatmap(oh_us_train_df.corr(),annot=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for x,y in zip(xgb.feature_importances_,x_train.columns[1:]):\n    print('# {} feature_importance : {:.4f}'.format(y,x))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Classification","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.init as init\ntr_img_path = '../input/siim-isic-melanoma-classification/jpeg/train'\nte_img_path = '../input/siim-isic-melanoma-classification/jpeg/test'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.image as img\n\nimg.imread(tr_img_path+'/ISIC_0015719.jpg').shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torchvision.models as models\nmnasnet = models.mnasnet1_0()\n\n    \n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}