{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Importing the necessary libraries\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\nimport os\nimport cv2\nimport shutil","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Read the csv file\ndf = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Finding unique patient ids from csv file\nprint(f\"The total patient ids are {df['patient_id'].count()}, from those the unique ids are {df['patient_id'].value_counts().shape[0]} \")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Display patient_id column \npatient_id = df['patient_id'].unique()\npatient_id","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Remove the duplicate 'patiend_id'\ndf = df.drop_duplicates(subset = \"patient_id\", keep='first') \ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Check whether any cell is empty or not\ndf.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Replace empty cell with nan \ndf.replace('', np.nan, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Remove all the rows which have null value\ndata = df.dropna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Finding number of malignant samples\nmalignant = data[data['target'] == 1]\nmalignant_image = malignant['image_name'].tolist()              #convert the columan data into list\nmalignant_image = [item + '.jpg' for item in malignant_image]   #add the .jpg extension at the end of 'image_name'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(malignant_image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Final train set images for model training\ntrain_img = malignant_image[0:40]\nval_img = malignant_image[40:52]\ntest_img = malignant_image[52:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_size = 512\ndef load_image(data_dir):\n    data = []\n    for i in range(len(data_dir)):\n        img_arr = cv2.imread(path + data_dir[i])\n        resized_arr = cv2.resize(img_arr, (img_size, img_size)) # Reshaping images to preferred size\n        data.append(resized_arr)\n    return np.array(data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train = load_image(train_img)\nx_val = load_image(val_img)\nx_test = load_image(test_img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Image augmentation using ImageDataGenerator\ndatagen1 = ImageDataGenerator(rotation_range = 50, zoom_range = 0.3,width_shift_range=0.3,\n                             height_shift_range=0.3,horizontal_flip = True, vertical_flip=True)\ndatagen1.fit(x_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Image augmentation using ImageDataGenerator\ndatagen2 = ImageDataGenerator(rotation_range = 50, zoom_range = 0.3,width_shift_range=0.3,\n                             height_shift_range=0.3,horizontal_flip = True, vertical_flip=True)\ndatagen2.fit(x_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Image augmentation using ImageDataGenerator\ndatagen3 = ImageDataGenerator(rotation_range = 50, zoom_range = 0.3,width_shift_range=0.3,\n                             height_shift_range=0.3,horizontal_flip = True, vertical_flip=True)\ndatagen3.fit(x_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dir1 = 'train_aug'\nif not os.path.exists(dir1):\n    os.mkdir(dir1)\n    print('Directory', dir1, 'created')\nelse:\n    print('Directory', dir1, 'already exists')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dir2 = 'val_aug'\nif not os.path.exists(dir2):\n    os.mkdir(dir2)\n    print('Directory', dir2, 'created')\nelse:\n    print('Directory', dir2, 'already exists')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dir3 = 'test_aug'\nif not os.path.exists(dir3):\n    os.mkdir(dir3)\n    print('Directory', dir3, 'created')\nelse:\n    print('Directory', dir3, 'already exists')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0\nfor batch in datagen1.flow(x_train, batch_size = 4, \n                          save_to_dir = dir1, \n                          save_prefix = 'M',\n                          save_format = 'jpg'):\n    i += 1\n    if i > 250:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"j = 0\nfor batch in datagen2.flow(x_val, batch_size = 4, \n                          save_to_dir = dir2, \n                          save_prefix = 'M',\n                          save_format = 'jpg'):\n    j += 1\n    if j > 50:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"k = 0\nfor batch in datagen3.flow(x_test, batch_size = 4, \n                          save_to_dir = dir3, \n                          save_prefix = 'M',\n                          save_format = 'jpg'):\n    k += 1\n    if k > 50:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.getcwd()\ncollection1 = \"train_aug/\"\nfor i, filename in enumerate(os.listdir(collection1)):\n    os.rename(\"train_aug/\" + filename, \"train_aug/\" + str(i) + \".jpg\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"collection2 = \"val_aug/\"\nfor i, filename in enumerate(os.listdir(collection2)):\n    os.rename(\"val_aug/\" + filename, \"val_aug/\" + str(i) + \".jpg\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"collection3 = \"test_aug/\"\nfor i, filename in enumerate(os.listdir(collection3)):\n    os.rename(\"test_aug/\" + filename, \"test_aug/\" + str(i) + \".jpg\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"l1 = os.listdir('/kaggle/working/train_aug/') # dir is your directory path\nfile1 = len(l1)\nprint(file1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"l2 = os.listdir('/kaggle/working/val_aug/') # dir is your directory path\nfile2 = len(l2)\nprint(file2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"l3 = os.listdir('/kaggle/working/test_aug/') # dir is your directory path\nfile3 = len(l3)\nprint(file3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_read1 = cv2.imread('train_aug/1.jpg')\nplt.imshow(img_read1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_read2 = cv2.imread('val_aug/1.jpg')\nplt.imshow(img_read2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_read3 = cv2.imread('test_aug/1.jpg')\nplt.imshow(img_read3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = 'zip'\nif not os.path.exists(train):\n    os.mkdir(train)\n    print('Directory', train, 'created')\nelse:\n    print('Directory', train, 'already exists')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"zip_name = 'train'\ndirectory_name = 'train_aug'\n\n# Create 'path\\to\\zip_file.zip'\nshutil.make_archive(zip_name, 'zip', directory_name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val = 'zip1'\nif not os.path.exists(val):\n    os.mkdir(val)\n    print('Directory', val, 'created')\nelse:\n    print('Directory', val, 'already exists')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"zip_name2 = 'val'\ndirectory_name2 = 'val_aug'\n\n# Create 'path\\to\\zip_file.zip'\nshutil.make_archive(zip_name2, 'zip', directory_name2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = 'zip2'\nif not os.path.exists(test):\n    os.mkdir(test)\n    print('Directory', test, 'created')\nelse:\n    print('Directory', test, 'already exists')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"zip_name3 = 'test'\ndirectory_name3 = 'test_aug'\n\n# Create 'path\\to\\zip_file.zip'\nshutil.make_archive(zip_name3, 'zip', directory_name3)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}