{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Import necessary packages\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom as pdi\nimport seaborn as sns\nimport os","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '/kaggle/input/siim-isic-melanoma-classification/'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(BASE_DIR + 'train.csv')\ntest_df = pd.read_csv(BASE_DIR + 'test.csv')\nsample_sub_df = pd.read_csv(BASE_DIR + 'sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of Training Samples : \", train_df.shape[0])\nprint(\"Number of Test Samples : \", test_df.shape[0])\nprint(\"Number of Training Features : \", train_df.shape[1])\nprint(\"Number of Test Features : \", test_df.shape[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[train_df['target'] == 1].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[train_df['target'] == 0].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Total number of patients in the training data        : \",train_df['patient_id'].count())\nprint(\"Total number of unique patients in the training data : \",train_df['patient_id'].value_counts().shape[0])\nprint(\"Total number of patients in the testing data         : \",test_df['patient_id'].count())\nprint(\"Total number of unique patients in the testing data  : \",test_df['patient_id'].value_counts().shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['image_name'] = train_df['image_name'] + '.jpg'\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df['image_name'] = train_df['image_name'] + '.jpg'\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample Images on the Negative Class\n# Extract numpy values from Image column in data frame\nimages = train_df[train_df['target'] == 1]['image_name'].values\n\n# Extract 6 random images from it\nrandom_images = [np.random.choice(images) for i in range(6)]\n\n# Location of the image dir\nimg_dir = BASE_DIR + 'jpeg/train/'\n\nprint('Display Random Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(20,10))\n\n# Iterate and plot random images\nfor i in range(6):\n    plt.subplot(3, 2, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout()    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample Images on the Negative Class\n# Extract numpy values from Image column in data frame\nimages = train_df[train_df['target'] == 0]['image_name'].values\n\n# Extract 6 random images from it\nrandom_images = [np.random.choice(images) for i in range(6)]\n\n# Location of the image dir\nimg_dir = BASE_DIR + 'jpeg/train/'\n\nprint('Display Random Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(20,10))\n\n# Iterate and plot random images\nfor i in range(6):\n    plt.subplot(3, 2, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout()    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Import data generator from keras\nfrom keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Normalize images\nimage_generator = ImageDataGenerator(\n    samplewise_center=True, #Set each sample mean to 0.\n    samplewise_std_normalization= True # Divide each input by its standard deviation\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Flow from directory with specified batch size and target image size\ngenerator = image_generator.flow_from_dataframe(\n        dataframe=train_df,\n        directory=BASE_DIR+\"jpeg/train/\",\n        x_col=\"image_name\", # features\n        y_col= ['target'], # labels\n        class_mode=\"raw\", \n        batch_size= 1, # images per batch\n        shuffle=False, # shuffle the rows or not\n        target_size=(320,320) # width and height of output image\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot a processed image\ngenerated_image, label = generator.__getitem__(1)\nplt.imshow(generated_image[0])\nplt.colorbar()\nprint(f'Melanoma Image - Label :' , label)\nprint(f\"The dimensions of the image are {generated_image.shape[1]} pixels width and {generated_image.shape[2]} pixels height\")\nprint(f\"The maximum pixel value is {generated_image.max():.4f} and the minimum is {generated_image.min():.4f}\")\nprint(f\"The mean value of the pixels is {generated_image.mean():.4f} and the standard deviation is {generated_image.std():.4f}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"generated_image.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}