{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Melanoma Prediction Using Tabular Data\nIn this solely based on feature engineering and using Machine learning Model to detect Skin Cancer. No Deep Learning Model has been used here.\n\n**Special Thank to [oliver](https://www.kaggle.com/ogrellier) for sharing some key features in the data**","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom glob import glob\nfrom tqdm import tqdm\nimport seaborn as sns\nsns.set(style = 'dark')\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_files_dir = glob('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/*')\ntest_files_dir = glob('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/*')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Label Encoding","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Sex","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['sex'].fillna('unkown',inplace = True) # missing value","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nenc = LabelEncoder()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['sex_enc'] = enc.fit_transform(train_df.sex.astype('str'))\ntest_df['sex_enc'] = enc.transform(test_df.sex.astype('str'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12,6))\nsns.countplot(x = 'sex', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Anatom_site_general_challenge","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.anatom_site_general_challenge = test_df.anatom_site_general_challenge.fillna('unknown')\ntrain_df.anatom_site_general_challenge = train_df.anatom_site_general_challenge.fillna('unknown')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['anatom_enc']= enc.fit_transform(train_df.anatom_site_general_challenge.astype('str'))\ntest_df['anatom_enc']= enc.transform(test_df.anatom_site_general_challenge.astype('str'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Age","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['age_approx'] = train_df['age_approx'].fillna(train_df['age_approx'].mode().values[0])\ntest_df['age_approx']  = test_df['age_approx'].fillna(test_df['age_approx'].mode().values[0]) # Test data doesn't have any NaN in age_approx","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['age_enc']= enc.fit_transform(train_df['age_approx'].astype('str'))\ntest_df['age_enc']= enc.transform(test_df['age_approx'].astype('str'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,6))\nsns.countplot(x = 'age_approx', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Images Per Patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['n_images'] = train_df.patient_id.map(train_df.groupby(['patient_id']).image_name.count())\ntest_df['n_images'] = test_df.patient_id.map(test_df.groupby(['patient_id']).image_name.count())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Categorize Number of Images Per Patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import KBinsDiscretizer\ncategorize = KBinsDiscretizer(n_bins = 10, encode = 'ordinal', strategy = 'uniform')\ntrain_df['n_images_enc'] = categorize.fit_transform(train_df['n_images'].values.reshape(-1, 1)).astype(int).squeeze()\ntest_df['n_images_enc'] = categorize.transform(test_df['n_images'].values.reshape(-1, 1)).astype(int).squeeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12,6))\nsns.countplot(x = 'n_images_enc', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Size ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images = train_df['image_name'].values\ntrain_sizes = np.zeros(train_images.shape[0])\nfor i, img_path in enumerate(tqdm(train_images)):\n    train_sizes[i] = os.path.getsize(os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg'))\n    \ntrain_df['image_size'] = train_sizes\n\n\ntest_images = test_df['image_name'].values\ntest_sizes = np.zeros(test_images.shape[0])\nfor i, img_path in enumerate(tqdm(test_images)):\n    test_sizes[i] = os.path.getsize(os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg'))\n    \ntest_df['image_size'] = test_sizes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Scaling Image Size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, MinMaxScaler\nscale = MinMaxScaler()\ntrain_df['image_size_scaled'] = scale.fit_transform(train_df['image_size'].values.reshape(-1, 1))\ntest_df['image_size_scaled'] = scale.transform(test_df['image_size'].values.reshape(-1, 1))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Categorize Image Size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import KBinsDiscretizer\ncategorize = KBinsDiscretizer(n_bins = 10, encode = 'ordinal', strategy = 'uniform')\ntrain_df['image_size_enc'] = categorize.fit_transform(train_df.image_size_scaled.values.reshape(-1, 1)).astype(int).squeeze()\ntest_df['image_size_enc'] = categorize.transform(test_df.image_size_scaled.values.reshape(-1, 1)).astype(int).squeeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12,6))\nsns.countplot(x = 'image_size_enc', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Mean Color(Used previously saved data)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_mean_color = pd.read_csv('/kaggle/input/mean-color-isic2020/train_color.csv')\ntest_mean_color = pd.read_csv('/kaggle/input/mean-color-isic2020/test_color.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['mean_color'] = train_mean_color.values\ntest_df['mean_color'] = test_mean_color.values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Categorize Mean Color","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import KBinsDiscretizer\ncategorize = KBinsDiscretizer(n_bins = 10, encode = 'ordinal', strategy = 'uniform')\ntrain_df['mean_color_enc'] = categorize.fit_transform(train_df['mean_color'].values.reshape(-1, 1)).astype(int).squeeze()\ntest_df['mean_color_enc'] = categorize.transform(test_df['mean_color'].values.reshape(-1, 1)).astype(int).squeeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12,6))\nsns.countplot(x = 'mean_color_enc', hue = 'target', data = train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Min-Max age of Patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['age_id_min']  = train_df['patient_id'].map(train_df.groupby(['patient_id']).age_approx.min())\ntrain_df['age_id_max']  = train_df['patient_id'].map(train_df.groupby(['patient_id']).age_approx.max())\n\ntest_df['age_id_min']  = test_df['patient_id'].map(test_df.groupby(['patient_id']).age_approx.min())\ntest_df['age_id_max']  = test_df['patient_id'].map(test_df.groupby(['patient_id']).age_approx.max())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Mean Encoding","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Plotting Barplot with number","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_bar_plot(df, figsize = (12,6)):\n \n    import seaborn as sns\n    import matplotlib.pyplot as plt\n    import numpy as np\n    \n    def show_values_on_bars(axs, h_v=\"v\", space=0.4, v_space = 0.02, figsize = (12,6)):\n        def _show_on_single_plot(ax):\n            if h_v == \"v\":\n                for p in ax.patches:\n                    _x = p.get_x() + p.get_width() / 2\n                    _y = p.get_y() + p.get_height()+ v_space\n                    value = float(p.get_height())\n                    ax.text(_x, _y, f'{value:.1f}', ha=\"center\") \n            elif h_v == \"h\":\n                for p in ax.patches:\n                    _x = p.get_x() + p.get_width() + float(space)\n                    _y = p.get_y() + p.get_height()+ v_space\n                    value = int(p.get_width())\n                    ax.text(_x, _y, value, ha=\"left\")\n\n        if isinstance(axs, np.ndarray):\n            for idx, ax in np.ndenumerate(axs):\n                _show_on_single_plot(ax)\n        else:\n            _show_on_single_plot(axs)\n            \n\n#     fig = plt.gcf()\n#     fig.set_size_inches(12, 8)\n    plt.figure(figsize = figsize)\n    sns.set()\n    plt.title('Probability')\n\n    prob = df*100\n    pal = sns.color_palette(palette='Blues_r', n_colors=len(prob))\n    rank = prob.values.argsort().argsort() \n    #g=sns.barplot(x='day',y='tip',data=groupedvalues, palette=np.array(pal[::-1])[rank])\n    br = sns.barplot(prob.index, prob.values, palette=np.array(pal[::-1])[rank])\n    show_values_on_bars(br, \"v\", .50)\n    plt.show()          ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def kdeplot( df,col_name, figsize = (12,6)):\n\n    plt.figure(figsize = figsize)\n    sns.kdeplot(df[col_name][df.target==0], shade = True, color = 'b', label = '0')\n    sns.kdeplot(df[col_name][df.target==1], shade = True, color = 'r', label = '1')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Age","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['age_approx_mean_enc'] = train_df['age_approx'].map(train_df.groupby(['age_approx'])['target'].mean())\ntest_df['age_approx_mean_enc'] = test_df['age_approx'].map(train_df.groupby(['age_approx'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Probability of melanoma with respect to Age","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_bar_plot(train_df.groupby(['age_approx'])['target'].mean(), (20,10))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Before Mean Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'age_approx'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After Mean Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'age_approx_mean_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Sex","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['sex_mean_enc'] = train_df.sex_enc.map(train_df.groupby(['sex_enc'])['target'].mean())\ntest_df['sex_mean_enc'] = test_df.sex_enc.map(train_df.groupby(['sex_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Probability of melanoma with respect to Sex","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_bar_plot(train_df.groupby(['sex'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Before Mean Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'sex_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After Mean Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'sex_mean_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## n_images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['n_images_mean_enc'] = train_df['n_images_enc'].map(train_df.groupby(['n_images_enc'])['target'].mean())\ntest_df['n_images_mean_enc'] = test_df['n_images_enc'].map(train_df.groupby(['n_images_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Probability of melanoma with respect to Number of Image per Patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_bar_plot(train_df.groupby(['n_images_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'n_images'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'n_images_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After Mean Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'n_images_mean_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Image Size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['image_size_mean_enc'] = train_df['image_size_enc'].map(train_df.groupby(['image_size_enc'])['target'].mean())\ntest_df['image_size_mean_enc'] = test_df['image_size_enc'].map(train_df.groupby(['image_size_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Probability of melanoma with respect to Image Size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_bar_plot(train_df.groupby(['image_size_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'image_size'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'image_size_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'image_size_mean_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Anatom General Challenge","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['anatom_mean_enc'] = train_df['anatom_enc'].map(train_df.groupby(['anatom_enc'])['target'].mean())\ntest_df['anatom_mean_enc'] = test_df['anatom_enc'].map(train_df.groupby(['anatom_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Probability of melanoma with respect to Anatom General Challege","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_bar_plot(train_df.groupby(['anatom_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'anatom_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'anatom_mean_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Mean Color","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['mean_color_mean_enc'] = train_df['mean_color_enc'].map(train_df.groupby(['mean_color_enc'])['target'].mean())\ntest_df['mean_color_mean_enc'] = test_df['mean_color_enc'].map(train_df.groupby(['mean_color_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Probability of melanoma with respect to Mean Color","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_bar_plot(train_df.groupby(['mean_color_enc'])['target'].mean())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Before encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'mean_color'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'mean_color_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## After Mean Encoding","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"col_name = 'mean_color_mean_enc'\nkdeplot(train_df,col_name, figsize = (16,8))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Mean Encoding is Angel or Devil?\n\nAs you can see mean encoding has some interesting effect on data. But there is a good chnace we will end up overfitting because we're depending on target of train data. What if train and test data have different distribution??? ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Correlation Matrix\nWe can extract some interesting features from Corr Matrix. I leave that to the reader to find out some tricky features from Corr Matrix. Let me know you find one. \n\n**Spoiler Alert:** You can easily get **.80** just playing with these features","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"corr = train_df.corr(method = 'pearson')\ncorr = corr.abs()\ncorr.style.background_gradient(cmap='inferno')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plt.figure(figsize = (20,20))\n# sns.heatmap(corr, annot = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr = test_df.corr(method = 'pearson')\ncorr = corr.abs()\ncorr.style.background_gradient(cmap='inferno')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Selecting Features","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.columns","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# I get rid of some features for best LB score","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"features = [\n            'age_approx',\n#             'age_enc',\n#             'age_approx_mean_enc',\n            'age_id_min',\n            'age_id_max',\n            'sex_enc',\n#             'sex_mean_enc',\n            'anatom_enc',\n#             'anatom_mean_enc',\n            'n_images',\n#             'n_images_mean_enc',\n#             'n_images_enc',\n            'image_size_scaled',\n#             'image_size_enc',\n#             'image_size_mean_enc',\n            'mean_color',\n#             'mean_color_enc', \n#             'mean_color_mean_enc'\n           ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trlm_df = pd.read_csv('../input/landscape/train40Features.csv')\ntelm_df = pd.read_csv('../input/landscape/test40Features.csv')\n\nX = pd.concat([train_df[features], trlm_df.iloc[:,3:]], axis=1)\ny = train_df['target']\n\nX_test = pd.concat([test_df[features],telm_df.iloc[:,3:]], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load libraries\nfrom pandas import read_csv\nfrom pandas.plotting import scatter_matrix\nfrom matplotlib import pyplot\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import RandomizedSearchCV, GridSearchCV\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.svm import SVR, SVC\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom sklearn.linear_model import SGDRegressor, BayesianRidge\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training Xgboost\n## Parameters were tunned using Gridsearch","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model = XGBRegressor(base_score=0.5, booster=None, colsample_bylevel=1,\n             colsample_bynode=1, colsample_bytree=0.8, gamma=1, gpu_id=-1,\n             importance_type='gain', interaction_constraints=None,\n             learning_rate=0.002, max_delta_step=0, max_depth=10,\n             min_child_weight=1, missing=None, monotone_constraints=None,\n             n_estimators=700, n_jobs=-1, nthread=-1, num_parallel_tree=1,\n             objective='binary:logistic', random_state=0, reg_alpha=0,\n             reg_lambda=1, scale_pos_weight=1, silent=True, subsample=0.8,\n             tree_method=None, validate_parameters=False, verbosity=None)\n\nkfold = StratifiedKFold(n_splits=5, random_state=1001, shuffle=True)\ncv_results = cross_val_score(model, X, y, cv=kfold, scoring='roc_auc', verbose = 3)\ncv_results.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xgb = XGBRegressor(base_score=0.5, booster=None, colsample_bylevel=1,\n             colsample_bynode=1, colsample_bytree=0.8, gamma=1, gpu_id=-1,\n             importance_type='gain', interaction_constraints=None,\n             learning_rate=0.002, max_delta_step=0, max_depth=10,\n             min_child_weight=1, missing=None, monotone_constraints=None,\n             n_estimators=700, n_jobs=-1, nthread=-1, num_parallel_tree=1,\n             objective='binary:logistic', random_state=0, reg_alpha=0,\n             reg_lambda=1, scale_pos_weight=1, silent=True, subsample=0.8,\n             tree_method=None, validate_parameters=False, verbosity=None)\n\nxgb.fit(X,y)\npred_xgb = xgb.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plot: Feature Importance","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_important = xgb.get_booster().get_score(importance_type='weight')\nkeys = list(feature_important.keys())\nvalues = list(feature_important.values())\n\ndata = pd.DataFrame(data=values, index=keys, columns=[\"score\"]).sort_values(by = \"score\", ascending=False)\nplt.figure(figsize= (12,10))\nsns.barplot(x = data.score , y = data.index, orient = 'h', palette = 'Blues_r')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Dimension Reduction (PCA)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.decomposition import PCA\n\nn_components = 2\npca = PCA(n_components=n_components)\nX_pca = pca.fit_transform(X)\n\npca_df = pd.DataFrame({'x_pca_0':X_pca[:,0],\n             'x_pca_1':X_pca[:,1],\n             'y':y})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (10,10))\nsns.scatterplot(\n    x=\"x_pca_0\", y=\"x_pca_1\",\n    hue=\"y\",\n    data=pca_df,\n    legend=\"full\",\n    alpha=0.9\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prediction","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.DataFrame({'image_name':test_df.image_name.values,\n                    'target':pred_xgb})\nsub.to_csv('submission.csv',index = False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Though it is difficult to predict location of melanoma class from  plot but we look carefully we'll be able to notice that there are some regions where there is no melanoma class at all.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"There might be two possible reasons behind LB score differing from the CV score .\n1. Only 30% test data is used in Public LB\n2. Our Model has been ovefitted\n\nPlease let me know if I can improve my results. \n## Thank You Very Much","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}