{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom glob import glob\nfrom tqdm import tqdm\nimport seaborn as sns\nsns.set(style = 'dark')\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Loading image files","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_files_dir = glob('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/*')\ntest_files_dir = glob('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/*')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Loading Metadata","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Imputing missing values & Feature Engineering","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Sex","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Imputing mising values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['sex'].fillna('unkown',inplace = True) # missing value","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Label encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nenc = LabelEncoder()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['sex_enc'] = enc.fit_transform(train_df.sex.astype('str'))\ntest_df['sex_enc'] = enc.transform(test_df.sex.astype('str'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12,6))\nsns.countplot(x = 'sex', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's check the encoding columns","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Anatom_site_general_challenge","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Imputing missing values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.anatom_site_general_challenge = test_df.anatom_site_general_challenge.fillna('unknown')\ntrain_df.anatom_site_general_challenge = train_df.anatom_site_general_challenge.fillna('unknown')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Label encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['anatom_enc']= enc.fit_transform(train_df.anatom_site_general_challenge.astype('str'))\ntest_df['anatom_enc']= enc.transform(test_df.anatom_site_general_challenge.astype('str'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Age","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Imputing missing values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['age_approx'] = train_df['age_approx'].fillna(train_df['age_approx'].mode().values[0])\ntest_df['age_approx']  = test_df['age_approx'].fillna(test_df['age_approx'].mode().values[0]) # Test data doesn't have any NaN in age_approx","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,6))\nsns.countplot(x = 'age_approx', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Images Per Patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['n_images'] = train_df.patient_id.map(train_df.groupby(['patient_id']).image_name.count())\ntest_df['n_images'] = test_df.patient_id.map(test_df.groupby(['patient_id']).image_name.count())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image Size ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images = train_df['image_name'].values\ntrain_sizes = np.zeros(train_images.shape[0])\nfor i, img_path in enumerate(tqdm(train_images)):\n    train_sizes[i] = os.path.getsize(os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg'))\n    \ntrain_df['image_size'] = train_sizes\n\n\ntest_images = test_df['image_name'].values\ntest_sizes = np.zeros(test_images.shape[0])\nfor i, img_path in enumerate(tqdm(test_images)):\n    test_sizes[i] = os.path.getsize(os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg'))\n    \ntest_df['image_size'] = test_sizes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Scaling Image Size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, MinMaxScaler\nscale = MinMaxScaler()\ntrain_df['image_size_scaled'] = scale.fit_transform(train_df['image_size'].values.reshape(-1, 1))\ntest_df['image_size_scaled'] = scale.transform(test_df['image_size'].values.reshape(-1, 1))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Min-Max age of Patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['age_id_min']  = train_df['patient_id'].map(train_df.groupby(['patient_id']).age_approx.min())\ntrain_df['age_id_max']  = train_df['patient_id'].map(train_df.groupby(['patient_id']).age_approx.max())\n\ntest_df['age_id_min']  = test_df['patient_id'].map(test_df.groupby(['patient_id']).age_approx.min())\ntest_df['age_id_max']  = test_df['patient_id'].map(test_df.groupby(['patient_id']).age_approx.max())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Training the model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"features = [\n            'age_approx',\n            'age_id_min',\n            'age_id_max',\n            'sex_enc',\n            'anatom_enc',\n            'n_images',\n            'image_size_scaled',\n           ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train_df[features]\ny = train_df['target']\n\nX_test = test_df[features]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load libraries for training\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Training Xgboost with Stratified K-Fold Cross Validation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model = XGBRegressor(base_score=0.5, booster=None, colsample_bylevel=1,\n             colsample_bynode=1, colsample_bytree=0.8, gamma=1, gpu_id=-1,\n             importance_type='gain', interaction_constraints=None,\n             learning_rate=0.002, max_delta_step=0, max_depth=10,\n             min_child_weight=1, missing=None, monotone_constraints=None,\n             n_estimators=700, n_jobs=-1, nthread=-1, num_parallel_tree=1,\n             objective='binary:logistic', random_state=0, reg_alpha=0,\n             reg_lambda=1, scale_pos_weight=1, silent=True, subsample=0.8,\n             tree_method=None, validate_parameters=False, verbosity=None)\n\nkfold = StratifiedKFold(n_splits=5, random_state=1001, shuffle=True)\ncv_results = cross_val_score(model, X, y, cv=kfold, scoring='roc_auc', verbose = 3)\ncv_results.mean()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Training on entire data for making predictions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(X,y)\npred_xgb = model.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Feature Importance","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_important = model.get_booster().get_score(importance_type='weight')\nkeys = list(feature_important.keys())\nvalues = list(feature_important.values())\n\ndata = pd.DataFrame(data=values, index=keys, columns=[\"score\"]).sort_values(by = \"score\", ascending=False)\nplt.figure(figsize= (12,10))\nsns.barplot(x = data.score , y = data.index, orient = 'h', palette = 'Blues_r')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creating submission file","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.DataFrame({'image_name':test_df.image_name.values,\n                    'target':pred_xgb})\nsub.to_csv('submission.csv',index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}