{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os, random, re, math, time\nrandom.seed(a=42)\nimport numpy as np\n\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nimport PIL\nfrom kaggle_datasets import KaggleDatasets\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Reading in the data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n## import the dataset\ndftrain = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ndftest = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### check for missing values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"## Missing values in training data\ndftrain.isna().any()\ndftrain = dftrain.dropna()\n\n## Missing values in testing data\ndftest.isna().any()\ndftest = dftest.dropna()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### check unique values in each column","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Count of male and female \", dftrain.sex.value_counts())\nprint(\"uniques in anatom_site_general_challenge column \", dftrain.anatom_site_general_challenge.unique())\nprint(\"uniques in diagnosis column \", dftrain.diagnosis.unique())\nprint(\"uniques in benign_malignant column \", dftrain.benign_malignant.unique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### visualizations\n","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Below plot is the distribution of age of those affected","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.kdeplot(dftrain[(dftrain['target'] == 1)].age_approx, shade = True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Below plot is the distribution of sex of those affected","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"dummy = dftrain[dftrain['target'] == 1]\n\nsns.barplot(x = \"sex\", y = \"target\", data = dummy, hue =\"sex\")\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Comparision of affected (0 - blue) to unaffected (1 - orange)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# viewing the distributions of data \n\nsns.barplot(x = \"target\", y = \"target\", data = dftrain, hue =\"target\")\n\n## hence the number of unaffected surpasses the affected, we need to normalise this data. But we need to understand \n# various other columns before normalising this data","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### lets see the distributions of sex  in  unaffected before removing normalizing","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"no of unaffected males  : \", len(dftrain[(dftrain['target'] == 0) & (dftrain['sex'] == 'male')]))\nprint(\"no of unaffected females  : \", len(dftrain[(dftrain['target'] == 0) & (dftrain['sex'] == 'female')]))\n\n## not. much of a difference in. the ratios","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Below chart is the distribution of anatom site challenge","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.barplot(x = \"anatom_site_general_challenge\", y = \"target\", data = dftrain, hue =\"anatom_site_general_challenge\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## looking at the unique values of anatom_site_general_challenge column\n\ndftrain.anatom_site_general_challenge.unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# looking at the number of unique values of anatom_site_general_challenge column\n\ndf = dftrain.copy()\nprint(\"no of head/neck  : \", len(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'head/neck')]))\nprint(\"no of upper extremity  : \", len(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'upper extremity')]))\nprint(\"no of lower extremity  : \", len(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'lower extremity')]))\nprint(\"no of torso  : \", len(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'torso')]))\nprint(\"no of palms/soles'  : \", len(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'palms/soles')]))\nprint(\"no of oral/genital'  : \", len(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'oral/genital')]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## number of affected in the whole dataset\n\nprint(\"no of affected : \", len(dftrain[dftrain['target'] == 1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp = dftrain[dftrain['target'] == 1]\n\n## for training  (500 samples)\nfinal_df = temp[:500]\n\n## for testing (75 samples)\naffected_validationdata = temp[500:]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### selecting few samples randomly from the unaffected to normalize  the data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def random_data_selector(data, n):\n    data =  data.sample(n = n)\n    return data\n\n\ndf = dftrain.copy()\n\n## selecting samples for testing (100 samples)\ntemp  = df[df['target'] == 0]\nunaffected_validationdata = temp[:100]\n\n\ndata = random_data_selector(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'head/neck')], 100)\nfinal_df = final_df.append(data, ignore_index = True)\n\ndata = random_data_selector(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'upper extremity')], 100)\nfinal_df = final_df.append(data, ignore_index = True)\n\ndata = random_data_selector(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'lower extremity')], 100)\nfinal_df = final_df.append(data, ignore_index = True)\n\ndata = random_data_selector(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'torso')], 100)\nfinal_df = final_df.append(data, ignore_index = True)\n\ndata = random_data_selector(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'palms/soles')], 100)\nfinal_df = final_df.append(data, ignore_index = True)\n\ndata = random_data_selector(df[(df['target'] == 0) & (df['anatom_site_general_challenge'] == 'oral/genital')], 100)\nfinal_df = final_df.append(data, ignore_index = True)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"final_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## finding the number of infected\nlen(final_df[final_df['target'] == 1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## finding the number of infected\nlen(final_df[final_df['target'] == 0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## shuffle the dataset\n\nimport sklearn\n\nfinal_df  = sklearn.utils.shuffle(final_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\nimport pathlib\nimport imageio\nfrom skimage.transform import resize\nimport numpy as np\nfrom keras.preprocessing import image\nfrom keras.applications.resnet50 import ResNet50\nfrom keras.preprocessing import image\nfrom keras.applications.resnet50 import preprocess_input, decode_predictions\n\n\n\n#converting normal images into numpy array\n\n\ntraining_paths = pathlib.Path('../input/siim-isic-melanoma-classification/jpeg').glob('train/*.jpg')\ntraining_sorted = sorted([x for x in training_paths])\ndirectory_path = '../input/siim-isic-melanoma-classification/jpeg/train/'\n\n\n\nfor index in range(len(training_sorted)) : training_sorted[index] = str(training_sorted[index])    \n    \ntraining_images = np.zeros(150528)\ntraining_images = training_images.reshape(1,224,224,3)\n\nfor index in range(len(final_df)):\n    img_name = final_df.loc[index].image_name\n    img_name = str(directory_path +  img_name + '.jpg')\n    position_in_list = training_sorted.index(img_name)\n    img_path = training_sorted[position_in_list]\n    \n    img = image.load_img(img_path, target_size = (224,224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis = 0)\n    x = preprocess_input(x)    \n    training_images = np.vstack((training_images, x))\n\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## making the xtrain data ready\n\nxtraindf = training_images.copy()\nxtraindf = xtraindf[1:]\nxtraindf.shape\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfig, axes = plt.subplots(5,5, figsize=(8,8))\n\nfor i,ax in enumerate(axes.flat):\n    ax.imshow(xtraindf[i])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## making the ytrain data ready\n\nytraindf = final_df.target\nytraindf.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Test data analysis","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"affected_validationdata = affected_validationdata.append(unaffected_validationdata)\ntestdata_copy = affected_validationdata.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testdata_copy.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"affected : \", len(testdata_copy[testdata_copy['target'] == 1]))\nprint(\"unaffected : \", len(testdata_copy[testdata_copy['target'] == 0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## shuffle the df \nimport sklearn\n\ntestdata_copy  = sklearn.utils.shuffle(testdata_copy).reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testdata_copy.tail()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### convert test images to numpy arrays","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\n#converting normal images into numpy array\n\n\ntest_paths = pathlib.Path('../input/siim-isic-melanoma-classification/jpeg/').glob('train/*.jpg')\ntest_sorted = sorted([x for x in test_paths])\ndirectory_path = '../input/siim-isic-melanoma-classification/jpeg/train/'\n\n\n\nfor index in range(len(test_sorted)) : test_sorted[index] = str(test_sorted[index])    \n    \ntest_images = np.zeros(150528)\ntest_images = test_images.reshape(1,224,224,3)\n\nfor index in range(len(testdata_copy)):\n    img_name = testdata_copy.loc[index].image_name\n    img_name = str(directory_path +  img_name + '.jpg')\n    \n    position_in_list = test_sorted.index(img_name)\n    img_path = test_sorted[position_in_list]\n    \n    img = image.load_img(img_path, target_size = (224,224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis = 0)\n    x = preprocess_input(x)    \n    test_images = np.vstack((test_images, x))\n        \n\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Making the xtest data ready\n\nxtestdf = test_images.copy()\nxtestdf = xtestdf[1:]\nxtestdf.shape\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## plotting the images\n\nimport matplotlib.pyplot as plt\nfig, axes = plt.subplots(5,5, figsize=(8,8))\n\nfor i,ax in enumerate(axes.flat):\n    ax.imshow(xtestdf[i])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## making the ytest data ready\n\nytestdf = testdata_copy.target\nytestdf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# ResNet50 model\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications.resnet50 import ResNet50\n\nimg_rows, img_cols = 224, 224\n\n\nresnet = ResNet50(weights = 'imagenet',\n                      include_top = False,\n                      input_shape  =  (img_rows,  img_cols, 3))\n\n## lets freeze the last  4 layers as they  are set to be trainable by  default\nfor layer in  resnet.layers:\n  layer.trainable = False\n\n## lets look at our layers\nfor (i, layer) in enumerate(resnet.layers):\n  print(str(i)  + \" \"  +  layer.__class__.__name__, layer.trainable)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n## Let us  create a function to build our top layer\n\ndef addTopMobileNetLayer(bottom_model, num_classes):\n\n  top_model = bottom_model.output\n  top_model = GlobalAveragePooling2D()(top_model)\n  top_model = (Dense(2048, activation = 'relu'))(top_model)\n  top_model = (Dense(2048, activation = 'relu'))(top_model)\n  top_model = (Dense(1024, activation = 'relu'))(top_model)\n  top_model = (Dense(num_classes, activation  = 'softmax'))(top_model)\n\n  return top_model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.layers import Dense, GlobalAveragePooling2D, Dropout, Activation,  Flatten\nfrom keras.layers import Conv2D, MaxPooling2D, ZeroPadding2D\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.models import Model\n\nnum_classes = 2\n\nFC_head = addTopMobileNetLayer(resnet, num_classes)\n\nmodel  = Model(inputs =  resnet.input, outputs = FC_head)\n\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfrom keras.optimizers import Adam\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping\n\nearlystop = EarlyStopping(monitor = 'val_loss', \n                          min_delta = 0, \n                          patience = 3,\n                          verbose = 1,\n                          restore_best_weights = True)\n\ncallbacks = [earlystop]\n\n\nmodel.compile(loss = 'binary_crossentropy', optimizer = Adam(lr = 0.005), metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 10\n\n\nhistory = model.fit(xtraindf, \n          ytraindf,\n          batch_size = 8,\n          epochs = epochs,\n          verbose = 1,\n          callbacks = callbacks,\n          validation_data = (xtestdf, ytestdf)\n          )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### My CNN","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"\nimport tensorflow\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dense, Flatten, Dropout\nfrom keras.optimizers import Adam, RMSprop\nfrom tensorflow.keras.callbacks import TensorBoard","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(24,3,3, input_shape = (224,224,3), activation = 'relu'))\nmodel.add(Conv2D(36,3,3, activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size = (2,2)))\nmodel.add(Conv2D(36,3,3, activation = 'relu'))\nmodel.add(Conv2D(36,3,3,  activation = 'relu'))\n\nmodel.add(Flatten())\nmodel.add(Dense(units = 2028, activation = 'relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(units = 1024, activation = 'relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(units = 512, activation = 'relu'))\nmodel.add(Dense(units = 1, activation = 'sigmoid'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss = 'binary_crossentropy', optimizer = RMSprop(lr = 0.005), metrics = ['accuracy'])\nepochs = 10\n\n\nhistory = model.fit(xtraindf, \n          ytraindf,\n          batch_size = 64,\n          epochs = epochs,\n          verbose = 1,\n          validation_data = (xtestdf, ytestdf))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#plot of validation vs training accuracy over the epochs\n\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.legend(['training', 'validation'])\nplt.title('Loss')\nplt.xlabel('Epoch')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ypred = model.predict(xtestdf)\nevaluation = model.evaluate(xtestdf, ytestdf)\nprint('Test accuracy : {:.3f}'.format(evaluation[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}