{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"# Handling the Images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom sklearn.model_selection import train_test_split                       # used to split dataset\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator        # used to feed jpg images into model\nfrom tensorflow.keras.applications import Xception                         # load pretrained model\nfrom tensorflow.keras.layers import Dense, Flatten                         # add a normal layer at the end\nfrom matplotlib import pyplot as plt                                       # for data visualization\nfrom skimage.transform import rotate, AffineTransform, warp                # for data augmentation\nfrom skimage.transform import resize                                       # resize image\nfrom skimage import io                                                     # for saving images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntrain","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\ntest_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#make a dataframe that has image location\nimages_df = pd.DataFrame()\nimages_df['image_address'] = train_dir + train['image_name'] + '.jpg'\nimages_df['target'] = train['target']\nimages_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number of malignant cases = ', images_df['target'].sum())\nprint('Number of benign cases    = ', images_df['target'].count() - images_df['target'].sum())\nprint('Ratio of malignant cases   = ', images_df['target'].sum()/images_df['target'].count())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### So only 1.7% of data is true, it means by just outputting all false, our model can reach 98.3% accuracy by just calssifying all as benign!\n\nWe don't want this imbalance\n\nSo lets use only a part of the benign data!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's use only ~1200 benign images\n\n#lets get a mask with all the benign cases as true\nbenign_cases_truth = (images_df['target'] == 0).iloc[:1200]\nprint('we use only', benign_cases_truth.sum(), 'benign cases')\n\n#apply the mask to get ~1200 benign cases\nbenign_cases = (images_df[:].iloc[:1200])[:][benign_cases_truth]\n\n# use all the malignant cases:\nmalignant_cases = images_df[:][images_df['target'] == 1]\nprint('we use', len(malignant_cases.index), 'malignant cases')\n\nimages_df = benign_cases.copy()\nimages_df = images_df.append(malignant_cases)\n# don't worry about the order, train_test_split shuffles by default\nimages_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# split the data into training and dev set so we can validate our model\nX_train, X_dev, y_train, y_dev = train_test_split(images_df,images_df['target'], test_size=0.2, random_state=1234)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_shape = (299, 299)\ninput_shape_with_channels = (299, 299, 3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.makedirs('./train/zero')\nos.makedirs('./train/one')\nos.makedirs('./test/zero')\nos.makedirs('./test/one')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def augment_and_save(path, label, train_or_test):\n    \n    if label == 0:\n        label = 'zero'\n    elif label == 1:\n        label = 'one'\n    image_name = path[-16:-4]\n    save_location = train_or_test+'/'+label+'/'+image_name\n    \n    #read the image\n    image = io.imread(path)\n    #resize image\n    image_resized = resize(image, input_shape)\n    #make rotated image\n    rotated = rotate(image_resized, angle=45, mode = 'wrap')\n    #flipped\n    flipLR = np.fliplr(image_resized)\n    flipUD = np.flipud(image_resized)\n    \n    #save image\n    io.imsave(save_location+'_resized.jpg', image_resized)\n    io.imsave(save_location+'_rotated.jpg', rotated)\n    io.imsave(save_location+'_flipLR.jpg',  flipLR)    \n    io.imsave(save_location+'_flipUD.jpg',  flipUD)\n    \n    global i\n    print(i, end = ', ')\n    i = i+1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# to make training data\nX_train.apply(lambda row : augment_and_save(row['image_address'], \n                                  row['target'], 'train'), axis = 1)\n\n# similarly do the same for dev data\n# Note this must not be done for Test data","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Using this data\n\nif you run this above code you will get data in folders which you can then access following this this tutorial on [tf.data](https://www.tensorflow.org/tutorials/load_data/images) using keras preprocessing or tf.Data!","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}