{"cells":[{"metadata":{},"cell_type":"markdown","source":"# About this notebook...","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Importing the necessary libraries\nimport pandas as pd\nimport matplotlib.pylab as plt\nfrom matplotlib import pyplot\nimport seaborn as sns\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D , MaxPool2D , Flatten , Dropout\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import classification_report,confusion_matrix\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import roc_curve, auc\nfrom numpy import expand_dims\nimport numpy as np\nimport glob\nimport os\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Read the train csv file\nTrain_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\nTrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Finding unique patient ids from csv file\nprint(f\"The total patient ids are {Train_df['patient_id'].count()}, from those the unique ids are {Train_df['patient_id'].value_counts().shape[0]} \")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Display patient_id column \npatient_id = Train_df['patient_id'].unique()\npatient_id","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Remove the duplicate 'patiend_id'\ndf = Train_df.drop_duplicates(subset = \"patient_id\", keep='first') \ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Check whether any cell is empty or not\ndf.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Replace empty cell with nan \ndf.replace('', np.nan, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Remove all the rows which have null value\ndata = df.dropna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Finding number of benign samples\nbenign = data[data['target'] == 0]\nbenign = benign.sample(800)                               #choose number of samples from benign \nbenign_image = benign['image_name'].tolist()              #convert the columan data into list\nbenign_image = [item + '.jpg' for item in benign_image]   #add the .jpg extension at the end of 'image_name'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"benign_label = benign['target'].tolist()                  #Converted the labels into the list\nbenign_label = np.array(benign_label)                     #convert list into numpy array\nlen(benign_label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Divide the benign images into training, validation and test set\ntrain_b = benign_image[:500]\nval_b = benign_image[500:650]\ntest_b = benign_image[650:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Divide the benign labels into training, validation and test set\ntrain_bl = benign_label[:500]\nval_bl = benign_label[500:650]\ntest_bl = benign_label[650:]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Note: We have only 64 unique melanoma samples. That's why I have used augmentated melanoma images. This is the link of augmented images dataset. ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# dir is your directory path\nM_train = os.listdir('../input/melanoma-512-images/train/') \nfile1 = len(M_train)\nprint(file1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# dir is your directory path\nM_val = os.listdir('../input/melanoma-512-images/val/') \nfile2 = len(M_val)\nprint(file2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# dir is your directory path\nM_test = os.listdir('../input/melanoma-512-images/test/') \nfile3 = len(M_test)\nprint(file3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Divide the malignant images into training, validation and test set\ntrain_m = M_train[:500]\nval_m = M_val[:150]\ntest_m = M_test[:150]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Function to convert the images into grayscale and numpy array. \ndef image(path, data):\n    output = []\n    for i in range(len(data)):\n        img_arr = cv2.imread(path + data[i], cv2.IMREAD_GRAYSCALE)\n        output.append(img_arr)\n    return np.array(output)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#train, validation and test set melanoma images for model training\npath1 = '../input/melanoma-512-images/train/'\ntrain_mimg = image(path1, train_m)\npath2 = '../input/melanoma-512-images/val/'\nval_mimg = image(path2, val_m)\npath3 = '../input/melanoma-512-images/test/' \ntest_mimg = image(path3, test_m)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reshape the melanoma images\nimg_size = 512\ntrain_mimg = train_mimg.reshape(-1, img_size, img_size, 1)\nval_mimg = val_mimg.reshape(-1, img_size, img_size, 1)\ntest_mimg = test_mimg.reshape(-1, img_size, img_size, 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(train_mimg))\nprint(len(val_mimg))\nprint(len(test_mimg))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Divide the malignant labels into training, validation and test set\ntrain_ml = np.ones(500, dtype = int)\nval_ml = np.ones(150, dtype = int)\ntest_ml = np.ones(150, dtype = int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Final train, validation and test set labels for model training\ny_train = np.concatenate((train_bl, train_ml))\ny_val = np.concatenate((val_bl, val_ml))\ny_test = np.concatenate((test_bl, test_ml))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Function to resize the benign images and convert into grayscale and numpy array. \nimg_size = 512\ndef load_image(path, data_dir):\n    data = []\n    for i in range(len(data_dir)):\n        img_arr = cv2.imread(path + data_dir[i], cv2.IMREAD_GRAYSCALE)\n        resized_arr = cv2.resize(img_arr, (img_size, img_size)) # Reshaping images to preferred size\n        data.append(resized_arr)\n    return np.array(data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#train, validation and test set benign images for model training\npath_train = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\ntrain_bimg = load_image(path_train, train_b)\nval_bimg = load_image(path_train, val_b)\ntest_bimg = load_image(path_train, test_b)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reshape the benign images\ntrain_bimg = train_bimg.reshape(-1, img_size, img_size, 1)\nval_bimg = val_bimg.reshape(-1, img_size, img_size, 1)\ntest_bimg = test_bimg.reshape(-1, img_size, img_size, 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Final train, validation and test set images for model training\ntrain = np.concatenate((train_bimg, train_mimg))\nval = np.concatenate((val_bimg, val_mimg))\ntest = np.concatenate((test_bimg, test_mimg))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Normalize the data\nx_train = np.array(train) / 255\nx_val = np.array(val) / 255\nx_test = np.array(test) / 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Train the model\nmodel = Sequential()\nmodel.add(Conv2D(32, (3,3) , strides = 1 , padding = 'same' , activation = 'linear' , input_shape = (512, 512, 1)))\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(64, (3,3) , strides = 1 , padding = 'same' , activation = 'linear'))\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(128, (3,3) , strides = 1 , padding = 'same' , activation = 'linear'))\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(128, (3,3) , strides = 1 , padding = 'same' , activation = 'linear'))\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(256, (3,3) , strides = 1 , padding = 'same' , activation = 'linear'))\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Flatten())\nmodel.add(Dense(units = 256, activation = 'linear'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(units = 1 , activation = 'sigmoid'))\nmodel.compile(optimizer = 'sgd' , loss = 'binary_crossentropy' , metrics = ['accuracy', keras.metrics.AUC()])\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(x_train, y_train, batch_size = 4, epochs = 30 , steps_per_epoch = 100, validation_data = (x_val, y_val))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Predict x_test set\npredictions = model.predict_classes(x_test)\npredictions = predictions.reshape(1,-1)[0]\npredictions[:300]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Calculate roc_auc_score\nrocaucscore = roc_auc_score(y_test, predictions)\nprint('ROC_AUC_SCORE: %.2f' % rocaucscore)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Calculate sensitivity and precision\ntp,fn,fp,tn = confusion_matrix(y_test, predictions, labels=[0,1]).ravel()   \nSensityvity = tp/(tp+fn)\nprint('sensitivity:',Sensityvity)\nprecision = tp/(tp+fp)\nprint('precision:',precision)\nprint('False Negatives:',fn)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Read the test csv file\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\ntest_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_images = test_df['image_name'].tolist()              #convert the columan data into list\nall_images = [item + '.jpg' for item in all_images]      #add the .jpg extension at the end of 'image_name' to read image from the main folder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#load test data and convert it into 512 grayscale images\npath_test = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'\ntest_data = load_image(path_test, all_images[:5491])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#reshape test images\ntest_imgs = test_data.reshape(-1, img_size, img_size, 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#predict the probability of the class of test images\nprobabilities = model.predict(test_imgs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#convert the probability of the class into numpy array\nresult = np.array(probabilities)\nprint(result.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#store the probabitility of array into list\nfor i in range(len(result)):\n    result[i]=result[i][0]\nresult = list(result)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#create data frame to store the result\ndf_result = pd.DataFrame(result, columns=['target'])\ndf_image = test_df['image_name']\n\nfinal_result = pd.concat([df_image, df_result], axis = 1)\nfinal_result.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#convert the data frame into csv file\nfinal_result.to_csv('submission.csv', header=True, index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}