{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ni = 0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if i < 10:\n            print(os.path.join(dirname, filename))\n            i = i + 1\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport tensorflow as tf\nfrom torch.utils.data import Dataset,DataLoader\nfrom tqdm import tqdm\nfrom skimage import io","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_data = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.shape, test_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of patients in train dataset : \",train_data['patient_id'].nunique())\nprint(\"Number of patients in test dataset : \",test_data['patient_id'].nunique())\ncommon_data = pd.merge(train_data,test_data,on=['patient_id','patient_id'])\nprint(\"Number of common patients in train and test dataset : \",common_data['patient_id'].nunique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We can only use \"sex\", \"age_approx\", and \"anatom_site_general_challenge\" as features as \"benign_malignant\", and \"diagnosis\" are not present in test_data. Let's do some EDA related to these features","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Dataset is quite unbalanced.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['age_approx'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['anatom_site_general_challenge'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(1,2, figsize=(18,6))\n\nsns.countplot(x='sex',data=train_data,hue='target',ax=ax[0])\nsns.countplot(x='anatom_site_general_challenge',data=train_data,hue='target',ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax1 = plt.subplots(1,1,figsize=(18,6)) \n\nsns.countplot(x='age_approx',data=train_data,hue='target',ax=ax1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = train_data.copy()\n\ndf1 = df.loc[df['target'] == 1][['age_approx']].groupby('age_approx').agg({'age_approx':'count'})\ndf2 = df.loc[df['target'] == 0][['age_approx']].groupby('age_approx').agg({'age_approx':'count'})\ndf1.columns = ['count']\ndf2.columns = ['count']\ndf1.loc[0] = [0]\ndf1.loc[10] = [0]\ndf1 = df1.sort_values('age_approx')\ndf1['total'] = df1['count'] + df2['count']\ndf1['perc_malignant'] = df1['count']/df1['total']\ndf1 = df1.reset_index()\nf, ax2 = plt.subplots(1,1,figsize=(18,6))\n\nsns.barplot(x='age_approx',y='perc_malignant',data=df1,ax=ax2)\nax2.set_title('Percent malignant in age approx');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We see that if a tumor is present, be it either malignant or benign, the percentage of malignant tumors are highers in older people compared to younger people. However, people in the age group of approx 0 or 10 have no cases of malignant tumors.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"image_list = train_data[train_data['target'] == 0].sample(8)['image_name']\nimage_all=[]\nfor image_id in image_list:\n    image_file = f'/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'+image_id+'.jpg' \n    img = np.array(Image.open(image_file))\n    image_all.append(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(2,4,figsize=(18,8))\n\nc = 0\nfor i in range(2):\n    for j in range(4):\n        ax[i][j].imshow(image_all[c])\n        ax[i][j].axis('off')\n        c = c + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"melanoma_list = train_data[train_data['target'] == 0].sample(8)['image_name']\nmelanoma_all=[]\nfor image_id in melanoma_list:\n    image_file = f'/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'+image_id+'.jpg' \n    img = np.array(Image.open(image_file))\n    melanoma_all.append(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax1 = plt.subplots(2,4,figsize=(18,8))\n\nc = 0\nfor i in range(2):\n    for j in range(4):\n        ax1[i][j].imshow(melanoma_all[c])\n        ax1[i][j].axis('off')\n        c = c + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.isna().sum(), test_data.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's impute the null values.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"na_cols = train_data.columns[train_data.isna().any()].tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for col in na_cols:\n    mode = train_data[col].mode().values[0]\n    train_data[col] = train_data[col].fillna(mode)\n    test_data[col] = test_data[col].fillna(mode)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.isna().sum(), test_data.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datasets = [train_data, test_data]\n\nfor df in datasets:\n    df['sex_label'] = np.where(df['sex']=='female',1,0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['anatom_site_general_challenge'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ntrain_data['anatom_label'] = le.fit_transform(train_data['anatom_site_general_challenge'].astype('str'))\ntest_data['anatom_label'] = le.transform(test_data['anatom_site_general_challenge'].astype('str'))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.isna().sum(), test_data.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"No Null values in the training and test data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images = train_data['image_name'].values\ntrain_sizes = np.zeros(train_images.shape[0])\nfor i, img_path in enumerate(tqdm(train_images)):\n    train_sizes[i] = os.path.getsize(os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg'))\n    \ntrain_data['image_size'] = train_sizes\n\n\ntest_images = test_data['image_name'].values\ntest_sizes = np.zeros(test_images.shape[0])\nfor i, img_path in enumerate(tqdm(test_images)):\n    test_sizes[i] = os.path.getsize(os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg'))\n    \ntest_data['image_size'] = test_sizes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nminmax = MinMaxScaler()\n\ntrain_data['image_size_scaled'] = minmax.fit_transform(train_data['image_size'].values.reshape(-1,1))\ntest_data['image_size_scaled'] = minmax.transform(test_data['image_size'].values.reshape(-1,1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import KBinsDiscretizer\ncategorize = KBinsDiscretizer(n_bins = 10, encode = 'ordinal', strategy = 'uniform')\ntrain_data['image_size_enc'] = categorize.fit_transform(train_data.image_size_scaled.values.reshape(-1, 1)).astype(int).squeeze()\ntest_data['image_size_enc'] = categorize.transform(test_data.image_size_scaled.values.reshape(-1, 1)).astype(int).squeeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12,6))\nsns.countplot(x = 'image_size_enc', hue = 'target', data = train_data)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Finding Average colour of each image","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Function for finding average colour of an image. \n# Didn't need to use it after adding data from mean color isic 2020\n\ndef average_color(image_path):\n    from skimage import io\n    img = io.imread(image_path)[:, :, :-1]\n    avg = img.mean(axis=0).mean(axis=0).mean()\n    return avg","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Below i have written a chunk of code for finding average colour values for each image. It was taking too long so i added the values from the [mean color isic 2020](https://www.kaggle.com/awsaf49/mean-color-isic2020). I also used many techniques from this [kernel](https://www.kaggle.com/awsaf49/xgboost-tabular-data-ml-cv-85-lb-787).","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# It was taking too long to calculate the colour values myself. Uncomment for using the code\n\n#avg_colors_train = np.zeros(train_data.shape[0])\n\n#for i, img_path in enumerate(tqdm(train_images)):\n#    image_path = os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg')\n#    avg = average_color(image_path)\n#    avg_colors_train[i] = avg\n    \n#train_image_color['avg_color'] = avg_colors_train\n\n#train_image_color.to_csv('/kaggle/working/created_data/train_image_color.csv',index=False)'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# For test data\n#avg_colors_test = np.zeros(test_data.shape[0])\n\n#for i, img_path in enumerate(tqdm(test_images)):\n#    image_path = os.path.join('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg')\n#    avg = average_color(image_path)\n#    avg_colors_test[i] = avg\n    \n#test_image_color['avg_color'] = avg_colors_test\n\n#test_image_color.to_csv('/kaggle/working/created_data/test_image_color.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_colors = pd.read_csv('/kaggle/input/mean-color-isic2020/train_color.csv')\ntest_colors = pd.read_csv('/kaggle/input/mean-color-isic2020/test_color.csv')\n\ntrain_data['avg_color'] = train_colors['color_mean']\ntest_data['avg_color'] = test_colors['color_mean']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.groupby('patient_id').agg({'age_approx':'min'}).reset_index().head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_min = train_data.groupby('patient_id').agg({'age_approx':'min'}).reset_index()\ndf_min.columns = ['patient_id','age_min_id']\ntrain_data = pd.merge(train_data, df_min, on = 'patient_id', how= 'left')\n\ndf_max = train_data.groupby('patient_id').agg({'age_approx':'max'}).reset_index()\ndf_max.columns = ['patient_id','age_max_id']\ntrain_data = pd.merge(train_data, df_max, on = 'patient_id', how= 'left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_min = test_data.groupby('patient_id').agg({'age_approx':'min'}).reset_index()\ndf_min.columns = ['patient_id','age_min_id']\ntest_data = pd.merge(test_data, df_min, on = 'patient_id', how= 'left')\n\ndf_max = test_data.groupby('patient_id').agg({'age_approx':'max'}).reset_index()\ndf_max.columns = ['patient_id','age_max_id']\ntest_data = pd.merge(test_data, df_max, on = 'patient_id', how= 'left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_X = train_data[test_data.describe().columns]\ntrain_y = train_data[['target']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_X.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrf = RandomForestClassifier(n_estimators=1000, random_state=42, n_jobs=-1)\n\nrf.fit(train_X,train_y)\ntrain_proba = rf.predict_proba(train_X)[:,1]\ntest_proba = rf.predict_proba(test_data[train_X.columns])[:,1]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import roc_curve, roc_auc_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fpr, tpr, thresholds = roc_curve(train_y, train_proba)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_roc_curve(fpr, tpr):\n    plt.figure(figsize=(12,10))\n    plt.plot(fpr, tpr, color='orange', label='ROC')\n    plt.plot([0, 1], [0, 1], color='darkblue', linestyle='--')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic (ROC) Curve')\n    plt.legend()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_roc_curve(fpr,tpr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc_auc_score(train_y, train_proba)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rf = RandomForestClassifier(n_estimators=500, n_jobs = -1, random_state = 42)\nrf.fit(train_X, train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = []\nfor tree in rf.estimators_:\n    predictions.append(tree.predict_proba(train_X)[None, :])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = np.vstack(predictions)\ncum_mean = np.cumsum(predictions, axis=0)/np.arange(1, predictions.shape[0] + 1)[:, None, None]\n\nscores = []\nfor pred in cum_mean:\n    scores.append(roc_auc_score(train_y, np.argmax(pred, axis=1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.plot(scores, linewidth=3)\nplt.xlabel('num_trees')\nplt.ylabel('roc_auc');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"100 Estimators seems to be fine.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"features_list = [5,7,9]\ndepth_list = [3,5,8]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Hyperparameter tuning\nimport time\n\nn_trees = 100\nmodels = {}\nroc_train_dict = {}\n\nfor max_features in features_list:\n    for max_depth in depth_list:\n        start = time.time()\n        model = RandomForestClassifier(n_estimators = n_trees, max_depth = max_depth, max_features=max_features, n_jobs = -1, random_state=42)\n        model.fit(train_X,train_y)\n        train_pred = model.predict_proba(train_X)[:,1]\n        train_roc = roc_auc_score(train_y,train_pred)\n        roc_train_dict[max_features,max_depth] = round(train_roc,3)\n        models[max_features,max_depth] = model\n        end = time.time()\n        time_taken = round(end-start,3)\n        print(\"Time taken for \",max_depth,\" depth and \",max_features,\" features is : \",time_taken,\" seconds.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc_df = pd.DataFrame(roc_train_dict, index=['train_roc']).transpose()\nroc_df = roc_df.reset_index()\nroc_df.columns = ['max_features','max_depth','train_roc']\nroc_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"max_features = roc_df[roc_df['train_roc'] == roc_df['train_roc'].max()]['max_features'].values[0]\nmax_depth = roc_df[roc_df['train_roc'] == roc_df['train_roc'].max()]['max_depth'].values[0]\nmodel = models[max_features,max_depth]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submissions = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submissions.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_proba = model.predict_proba(test_data[train_X.columns])[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submissions.shape, test_proba.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submissions['target'] = test_proba","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submissions.to_csv('/kaggle/working/submissions_melanoma.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}