{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom keras.applications.densenet import DenseNet121\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras.models import Model\nfrom sklearn.model_selection import train_test_split\nfrom keras import backend as K\nfrom numpy import asarray\nfrom numpy import savetxt\nfrom numpy import save\n\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import concatenate\nfrom keras.layers import Conv2D, Dense, Dropout, Flatten\nfrom keras.models import model_from_json\nfrom keras.callbacks import EarlyStopping\nfrom keras.regularizers import l2\nfrom keras.applications.inception_v3 import preprocess_input\nfrom keras import applications\nfrom keras import optimizers\n\n\n\nfrom keras.models import load_model\nfrom PIL import Image\nfrom numpy import load\nimport keras\nfrom sklearn.utils import shuffle\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Read csv file containing training datadata\n##/kaggle/input/siim-isic-melanoma-classification/train.csv\ntrain_df = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\")\ndup_df=pd.read_csv(\"../input/duplicates/2020_Challenge_duplicates.csv\")\n# Print first 5 rows\nprint(f'There are {train_df.shape[0]} rows and {train_df.shape[1]} columns in this data frame of training set')\nprint(f'There are {test_df.shape[0]} rows and {test_df.shape[1]} columns in this data frame of test set')\nprint(f'There are {dup_df.shape[0]} rows and {dup_df.shape[1]} columns in this data frame of duplicates set')\nprint('Training set dataframe')\nprint(f'{train_df.head()}')\nprint('Test set dataframe')\nprint(f'{test_df.head()}')\nprint('duplicate set dataframe')\nprint(f'{dup_df.head()}')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Unique ID check for Training Set\nprint(f\"TRAINING SET:The total patient ids are {train_df['patient_id'].count()}, from those the unique ids are {train_df['patient_id'].value_counts().shape[0]} \")\n#Unique ID check FOR TEST SET\nprint(f\"TEST SET:The total patient ids are {test_df['patient_id'].count()}, from those the unique ids are {test_df['patient_id'].value_counts().shape[0]} \")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#to create a list of the names of each patient condition or disease: TRAINING SET.\ncolumns = train_df.keys()\ncolumns = list(columns)\nprint(columns)\n\n#to create a list of the names of each patient condition or disease: TEST SET.\ncolumns_test = test_df.keys()\ncolumns_test = list(columns_test)\nprint(columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Remove unnecesary elements in training set\ncolumns.remove('image_name')\ncolumns.remove('patient_id')\n# Get the total classes\nprint(f\"TRAINING SET: There are {len(columns)} columns of labels for these conditions: {columns}\")\n# Remove unnecesary elements in test set\ncolumns_test.remove('image_name')\ncolumns_test.remove('patient_id')\n# Get the total classes\nprint(f\"TEST_SET: There are {len(columns_test)} columns of labels for these conditions: {columns_test}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# TRAING SET: Look at the data type of each column and whether null values are present\ntrain_df.info()\n#to check for null values in the datase\nprint('Null values in the train dataset:')\nprint(train_df.isnull().values.sum())\nprint('Let s also check the column-wise distribution of null values:')\nprint(train_df.isnull().sum())\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# TEST SET: Look at the data type of each column and whether null values are present\ntest_df.info()\n#to check for null values in the datase\nprint('Null values in the test dataset:')\nprint(test_df.isnull().values.sum())\nprint('Let s also check the column-wise distribution of null values:')\nprint(test_df.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# DUPLICATE SET: Look at the data type of each column and whether null values are present\n#to check for null values in the datase\nprint('Null values in the test dataset:')\nprint(dup_df.isnull().values.sum())\nprint('Let s also check the column-wise distribution of null values:')\nprint(dup_df.isnull().sum())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Extracting values of dup for train and test set\n##Note that ISIC_id and ISI_id_paired forms a pair of identical images, hence one column must taken and removed from train_df and test_df\ndup_train=dup_df[dup_df[dup_df.columns[2]]=='train']\ndup_test=dup_df[dup_df[dup_df.columns[2]]=='test']\n##Removing unecessary columns\ndup_test\ntrain_df.info()\ndup_df.info()\nkey_cols = ['image_name']\n##Changing the column name of dup_train to that one of original set.\n##This column name will be used to remove duplicate id images\ndup_train.rename(columns = {'ISIC_id_paired':'image_name'}, inplace = True) \ndup_test.rename(columns = {'ISIC_id_paired':'image_name'}, inplace = True) \n##Finding intersection of the two sets\nintersection_df_train=train_df.merge(dup_train.loc[:, dup_train.columns.isin(key_cols)])\nintersection_df_test=test_df.merge(dup_test.loc[:, dup_test.columns.isin(key_cols)])\nprint(f'the shape of dup train {dup_train.shape}')\nprint(f'the shape of intersection train {intersection_df_train.shape}')\nprint(f'the shape of intersection test {intersection_df_test.shape}')\nprint(f'the shape of train_df before removing intersection part {train_df.shape}')\nprint(f'the shape of test_df before removing intersection part {test_df.shape}')\n\n##Finding indexes of intersections rows in the original sets\n\nindexNames_train = train_df[train_df.image_name.isin(intersection_df_train.image_name)].index\nindexNames_test = test_df[test_df.image_name.isin(intersection_df_test.image_name)].index\n# Delete these row indexes from dataFrame\ntrain_df.drop(indexNames_train , inplace=True)\n##test_df.drop(indexNames_test , inplace=True) I am removing this one\nprint(f'the shape of train_df after removing intersection part {train_df.shape}')\nprint(f'the shape of test_df after removing intersection part {test_df.shape}')\n\n\n\n##ISIC_id ISIC_id_paired partition","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Removing null values\ntrain_df[\"sex\"].fillna(\"no_sex\", inplace = True)\ntrain_df[\"age_approx\"].fillna(-1, inplace = True)\ntrain_df[\"anatom_site_general_challenge\"].fillna(\"no_anatomy\", inplace = True)\n##train_df = train_df.fillna(train_df['sex'])\nprint(\"counting null values after removing them from the dataset:\")\nprint(train_df.isnull().values.sum())\ntrain_df.head()\nprint('Let s also check the column-wise distribution of null values after null removal:')\nprint(train_df.isnull().sum())\n\n# Look at the data type of each column and whether null values are present after null removal\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Removing null values for test dataset\ntest_df[\"anatom_site_general_challenge\"].fillna(\"no_anatomy\", inplace = True)\nprint(\"counting null values after removing them from the test dataset:\")\nprint(test_df.isnull().values.sum())\ntest_df.head()\nprint('Let s also check the column-wise distribution of null values after null removal:')\nprint(test_df.isnull().sum())\n\n# Look at the data type of each column and whether null values are present after null removal\ntest_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##converting back age_approx into float:\ntrain_df['age_approx'] = train_df['age_approx'].astype(float)\ntrain_df.info()\n##Note for test dataset there nothing to change as nothing is affected","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Finding and counting unique values in categorical column training set:\nprint('unique values for sex:')\nprint(train_df['sex'].value_counts())\nprint('unique values for anatom_site_general_challenge:')\nprint(train_df['anatom_site_general_challenge'].value_counts())\nprint('unique values for diagnosis:')\nprint(train_df['diagnosis'].value_counts())\nprint('unique values for benign_malignant:')\nprint(train_df['benign_malignant'].value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Finding and counting unique values in categorical column for test dataset:\n##In test set there are only two categorical columns:\nprint('unique values for sex:')\nprint(test_df['sex'].value_counts())\nprint('unique values for anatom_site_general_challenge:')\nprint(test_df['anatom_site_general_challenge'].value_counts())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Encoding categorical column:\n\nlabels_sex = train_df['sex'].astype('category').cat.categories.tolist()\nreplace_sex = {'sex' : {k: v for k,v in zip(labels_sex,list(range(0,len(labels_sex)+1)))}}\n\nlabels_anatom_site_general_challenge = train_df['anatom_site_general_challenge'].astype('category').cat.categories.tolist()\nreplace_anatom_site_general_challenge = {'anatom_site_general_challenge' : {k: v for k,v in zip(labels_anatom_site_general_challenge,list(range(0,len(labels_anatom_site_general_challenge)+1)))}}\n\nlabels_diagnosis = train_df['diagnosis'].astype('category').cat.categories.tolist()\nreplace_diagnosis = {'diagnosis' : {k: v for k,v in zip(labels_diagnosis,list(range(0,len(labels_diagnosis)+1)))}}\n\nlabels_benign_malignant  = train_df['benign_malignant'].astype('category').cat.categories.tolist()\nreplace_benign_malignant= {'benign_malignant' : {k: v for k,v in zip(labels_benign_malignant ,list(range(0,len(labels_benign_malignant)+1)))}}\nprint(replace_sex )\nprint(replace_anatom_site_general_challenge )\nprint(replace_diagnosis )\nprint(replace_benign_malignant)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Encoding categorical column for test set:\n\nlabels_sex_test = test_df['sex'].astype('category').cat.categories.tolist()\nreplace_sex_test = {'sex' : {k: v for k,v in zip(labels_sex,list(range(0,len(labels_sex_test)+1)))}}\n\nlabels_anatom_site_general_challenge_test = test_df['anatom_site_general_challenge'].astype('category').cat.categories.tolist()\nreplace_anatom_site_general_challenge_test = {'anatom_site_general_challenge' : {k: v for k,v in zip(labels_anatom_site_general_challenge_test,list(range(0,len(labels_anatom_site_general_challenge_test)+1)))}}\n\nprint(replace_sex_test)\nprint(replace_anatom_site_general_challenge_test)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Replacing train_df by trainCopy_df to preserve original train_df\ntrainCopy_df = train_df.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Replacing test_df by testCopy_df to preserve original test_df\ntestCopy_df = test_df.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainCopy_df.replace(replace_sex, inplace=True)\nprint(trainCopy_df.head())\nprint(trainCopy_df['sex'].dtypes)\n\ntrainCopy_df_1 = trainCopy_df.copy()\ntrainCopy_df_1.replace(replace_anatom_site_general_challenge, inplace=True)\nprint(trainCopy_df_1.head())\nprint(trainCopy_df_1['anatom_site_general_challenge'].dtypes)\n\ntrainCopy_df_2 = trainCopy_df_1.copy()\ntrainCopy_df_2.replace(replace_diagnosis, inplace=True)\nprint(trainCopy_df_2.head())\nprint(trainCopy_df_2['diagnosis'].dtypes)\n\ntrainCopy_df_3 = trainCopy_df_2.copy()\ntrainCopy_df_3.replace(replace_benign_malignant, inplace=True)\nprint(trainCopy_df_3.head())\nprint(trainCopy_df_3['benign_malignant'].dtypes)\n\ntrainCopy_df_3.info()\ntrainCopy_df_3.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testCopy_df.replace(replace_sex, inplace=True)\nprint(testCopy_df.head())\nprint(testCopy_df['sex'].dtypes)\n\ntestCopy_df_1 = testCopy_df.copy()\ntestCopy_df_1.replace(replace_anatom_site_general_challenge_test, inplace=True)\nprint(testCopy_df_1.head())\nprint(testCopy_df_1['anatom_site_general_challenge'].dtypes)\n\n\n\ntestCopy_df_1.info()\ntestCopy_df_1.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_lc = train_df.copy()\ntrain_df_lc.info()\n\ntest_df_lc = test_df.copy()\ntest_df_lc.info()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_lc['sex'] = train_df_lc['sex'].astype('category')\ntrain_df_lc['anatom_site_general_challenge'] = train_df_lc['anatom_site_general_challenge'].astype('category')\ntrain_df_lc['diagnosis'] = train_df_lc['diagnosis'].astype('category')\ntrain_df_lc['benign_malignant'] = train_df_lc['benign_malignant'].astype('category')\n\ntrain_df_lc['sex'] = train_df_lc['sex'].cat.codes\ntrain_df_lc['anatom_site_general_challenge'] = train_df_lc['anatom_site_general_challenge'].cat.codes\ntrain_df_lc['diagnosis'] = train_df_lc['diagnosis'].cat.codes\ntrain_df_lc['benign_malignant'] = train_df_lc['benign_malignant'].cat.codes\nprint(train_df_lc.isnull().sum())\ntrain_df_lc.info()\n\n\n\nprint(train_df_lc.isnull().sum())\ntrain_df_lc.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df_lc['sex'] = test_df_lc['sex'].astype('category')\ntest_df_lc['anatom_site_general_challenge'] = test_df_lc['anatom_site_general_challenge'].astype('category')\n\n\ntest_df_lc['sex'] = test_df_lc['sex'].cat.codes\ntest_df_lc['anatom_site_general_challenge'] = test_df_lc['anatom_site_general_challenge'].cat.codes\nprint(test_df_lc.isnull().sum())\ntest_df_lc.info()\ntest_df_lc.head()\n\n\n\nprint(test_df_lc.isnull().sum())\ntest_df_lc.head()\n##test_df_lc.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Extract numpy values from Image column in data frame\ntrain_df_lc['image_name']=train_df_lc['image_name'].values+('.jpg') ##concatenating jpg as file extension\nimages = train_df_lc['image_name'].values\n##MY COMMENTS: images contains file names of images from column \"Image\" taken from dataframe train_df\nprint(f\"The shape of images is{images.shape}\")\n\n\n# Extract 9 random images from it\nrandom_images = [np.random.choice(images) for i in range(9)]\n#location of image directory\n\nimg_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\nprint('Display Random Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(20,10))\n\n# Iterate and plot random images\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Extract numpy values from Image column in data frame\ntest_df_lc['image_name']=test_df_lc['image_name'].values+('.jpg') ##concatenating jpg as file extension\nimages_test = test_df_lc['image_name'].values\n##MY COMMENTS: images contains file names of images from column \"Image\" taken from dataframe train_df\nprint(f\"The shape of images is{images.shape}\")\n\n\n# Extract 9 random images from it\nrandom_images_test = [np.random.choice(images_test) for i in range(9)]\n#location of image directory\n\nimg_dir_test='/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'\nprint('Display Random Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(20,10))\n\n# Iterate and plot random images\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir_test, random_images_test[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get the first image that was listed in the train_df dataframe\ntrain_df_lc.info()\ntrain_df_lc.head()\n\n# Get the first image that was listed in the train_df dataframe\nsample_img = train_df_lc.image_name[10]\nraw_image = plt.imread(os.path.join(img_dir, sample_img))\nplt.imshow(raw_image)\nplt.colorbar()\nplt.title('Skin Color RGB Image:TRAIN')\nprint(f\"The dimensions of the image are {raw_image.shape[0]} pixels width, {raw_image.shape[1]} pixels height and pixels depth {raw_image.shape[2]}, 3 color channel\")\nprint(f\"The maximum pixel value is {raw_image.max():.4f} and the minimum is {raw_image.min():.4f}\")\nprint(f\"The mean value of the pixels is {raw_image.mean():.4f} and the standard deviation is {raw_image.std():.4f}\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get the first image that was listed in the test_df dataframe\ntest_df_lc.info()\ntest_df_lc.head()\n\n# Get the first image that was listed in the train_df dataframe\nsample_img_test = test_df_lc.image_name[10]\nraw_image_test = plt.imread(os.path.join(img_dir_test, sample_img_test))\nplt.imshow(raw_image_test)\nplt.colorbar()\nplt.title('Skin Color RGB Image:TEST')\nprint(f\"The dimensions of the image are {raw_image_test.shape[0]} pixels width, {raw_image_test.shape[1]} pixels height and pixels depth {raw_image_test.shape[2]}, 3 color channel\")\nprint(f\"The maximum pixel value is {raw_image_test.max():.4f} and the minimum is {raw_image_test.min():.4f}\")\nprint(f\"The mean value of the pixels is {raw_image_test.mean():.4f} and the standard deviation is {raw_image_test.std():.4f}\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot a histogram of the distribution of the pixels\nsns.distplot(raw_image.ravel(), \n             label=f'Pixel Mean {np.mean(raw_image):.4f} & Standard Deviation {np.std(raw_image):.4f}', kde=False)\nplt.legend(loc='upper center')\nplt.title('Distribution of Pixel Intensities in the Image')\nplt.xlabel('Pixel Intensity')\nplt.ylabel('# Pixels in Image')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Import data generator from keras\nfrom keras.preprocessing.image import ImageDataGenerator\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Normalize images\n\"\"\"\nimage_generator = ImageDataGenerator(\n    samplewise_center=True, #Set each sample mean to 0.\n    samplewise_std_normalization= True # Divide each input by its standard deviation\n)\n\"\"\"\n\nimage_generator = ImageDataGenerator(rescale=1.0/255.0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Flow from directory with specified batch size and target image size\ngenerator = image_generator.flow_from_dataframe(\n        dataframe=train_df_lc,\n        directory=\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\",\n        x_col=\"image_name\", # features\n        y_col= ['target'], # labels\n        class_mode=\"raw\", # 'Mass' column should be in train_df\n        batch_size= 1, # images per batch\n        shuffle=False, # shuffle the rows or not\n        target_size=(320,320) # width and height of output image\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot a processed image\ngenerated_image, label = generator.__getitem__(0)\nplt.imshow(generated_image[0])\nplt.colorbar()\nplt.title('RGB Melanoma Image')\nprint(f\"The dimensions of the image are {generated_image.shape}\")\nprint(f\"The maximum pixel value is {generated_image.max():.4f} and the minimum is {generated_image.min():.4f}\")\nprint(f\"The mean value of the pixels is {generated_image.mean():.4f} and the standard deviation is {generated_image.std():.4f}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Include a histogram of the distribution of the pixels\nsns.set()\nplt.figure(figsize=(20, 10))\n\"\"\"\n# Plot histogram for original iamge\nsns.distplot(raw_image.ravel(), \n             label=f'Original Image: mean {np.mean(raw_image):.4f} - Standard Deviation {np.std(raw_image):.4f} \\n '\n             f'Min pixel value {np.min(raw_image)} - Max pixel value {np.max(raw_image)}',\n             color='blue', \n             kde=False)\n\"\"\"\n# Plot histogram for generated image\nsns.distplot(generated_image[0].ravel(), \n             label=f'Generated Image: mean {np.mean(generated_image[0]):.4f} - Standard Deviation {np.std(generated_image[0]):.4f} \\n'\n             f'Min pixel value {np.min(generated_image[0])} - Max pixel value {np.max(generated_image[0])}', \n             color='red', \n             kde=False)\n\n# Place legends\nplt.legend()\nplt.title('Distribution of Pixel Intensities in the Image')\nplt.xlabel('Pixel Intensity')\nplt.ylabel('# Pixel')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Include a histogram of the distribution of the pixels\nsns.set()\nplt.figure(figsize=(20, 10))\n\n# Plot histogram for original iamge\nsns.distplot(raw_image.ravel(), \n             label=f'Original Image: mean {np.mean(raw_image):.4f} - Standard Deviation {np.std(raw_image):.4f} \\n '\n             f'Min pixel value {np.min(raw_image)} - Max pixel value {np.max(raw_image)}',\n             color='blue', \n             kde=False)\n\n\n# Place legends\nplt.legend()\nplt.title('Distribution of Pixel Intensities in the Image')\nplt.xlabel('Pixel Intensity')\nplt.ylabel('# Pixel')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**The follow section will check class imbalance and remove them**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Count up the number of instances of each class (drop non-class columns from the counts)\nclass_counts = train_df_lc['target'].value_counts() \nprint(class_counts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot up the distribution of counts\nsns.barplot(class_counts.index, class_counts.values, color='b')\nplt.title('Distribution of Classes for Training Dataset', fontsize=15)\nplt.xlabel('Diseases Types', fontsize=15)\nplt.ylabel('Number of Patients', fontsize=15)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_lc.info()\ntest_df_lc.info()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Removing columns from train_df_lc so that train and test have the same number of column\n\ntrain_df_lc = train_df_lc.drop(\"diagnosis\", axis=1)\ntrain_df_lc = train_df_lc.drop(\"benign_malignant\", axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_lc.info()\ntest_df_lc.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# split the data into train and validation set:\n##First of all let's shuffle train_df_lc:\ntrain_df_lc = shuffle(train_df_lc)\ntrain_df_lc.reset_index(inplace=True, drop=True)\n##Split(Train:80% and validation:20%):\nnew_train_df_lc, new_validation_df_lc = train_test_split(train_df_lc, test_size=0.4, random_state=42, shuffle=True)\nnew_train_df_lc.info()\nnew_validation_df_lc.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##new_train_df_lc, new_validation_df_lc\n#Splitting validation and training sets into 3 sub sets each to facilitate training\nnum_split=3\nnew_train_df_lc_splitted=np.array_split(new_train_df_lc, num_split)\nnew_validation_df_lc_splitted=np.array_split(new_validation_df_lc, num_split)\n\n\nfor num_split_temp in range(num_split):\n    ##new_train_df_lc_splitted[num_split_temp].info()\n    ##new_validation_df_lc_splitted[num_split_temp].info()\n    new_train_df_lc_splitted[num_split_temp].to_csv(r'/kaggle/working/new_train_df_lc_splitted_'+str(num_split_temp)+'.csv',index = False)\n    new_validation_df_lc_splitted[num_split_temp].to_csv(r'/kaggle/working/new_validation_df_lc_splitted_'+str(num_split_temp)+'.csv',index = False)\n    \n\n\n\n##train_df_lc.to_csv(r'/kaggle/working/train_df_lc.csv',index = False)\n##test_df_lc.to_csv(r'/kaggle/working/test_df_lc.csv',index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features='image_name'\n##features=['image_name','sex','age_approx','anatom_site_general_challenge']\nlabels=['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##function to check whether there is leakage between two datasets. \n##We'll use this to make sure there are no patients in the test set \n##that are also present in either the train or validation sets\n# UNQ_C1 (UNIQUE CELL IDENTIFIER, DO NOT EDIT)\ndef check_for_leakage(df1, df2, patient_col):\n    \"\"\"\n    Return True if there any patients are in both df1 and df2.\n\n    Args:\n        df1 (dataframe): dataframe describing first dataset\n        df2 (dataframe): dataframe describing second dataset\n        patient_col (str): string name of column with patient IDs\n    \n    Returns:\n        leakage (bool): True if there is leakage, otherwise False\n    \"\"\"\n\n    \n    df1_patients_unique =df1[patient_col]\n    df2_patients_unique =df2[patient_col]\n    \n    patients_in_both_groups =pd.merge(df1_patients_unique, df2_patients_unique, how='inner', on=[patient_col])\n\n    leakage = len(patients_in_both_groups) > 0 # boolean (true if there is at least 1 patient in both groups)\n    \n  \n    \n    return leakage\n\n##We finally decided to remove this part due to the fact that if there are leakages, this would not have effect since the same patient can have skin cancer a different skin surface","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"leak=check_for_leakage(new_train_df_lc,test_df_lc,'patient_id')\nprint(leak)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Saving the dataframe to output\ntrain_df_lc.to_csv(r'/kaggle/working/train_df_lc.csv',index = False)\ntest_df_lc.to_csv(r'/kaggle/working/test_df_lc.csv',index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n##array_num stands for categorical and numerical inputs\n##The purpose of this function is to create both image generator and a numerical and categorical\n##Here we are trying to make sure the image and its corresponding inputs are in the same batch with the same label when shuffled\nclass DataGenerator(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, list_IDs, labels,img_dir,array_num, batch_size=32, height=320,width=320,depth=3,\n                 n_classes=1,num_size=3, shuffle=True):\n        'Initialization'\n        self.height=height\n        self.width=width\n        self.depth=depth\n        self.batch_size = batch_size\n        self.labels = labels\n        self.list_IDs = list_IDs\n        self.n_classes = n_classes\n        self.num_size=num_size\n        self.shuffle = shuffle\n        self.img_dir=img_dir\n        self.array_num=array_num\n        self.n = 0\n        self.max = self.__len__()\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n\n    def __getIndexes__(self):\n        return self.indexes\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Find list of IDs\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n\n        # Generate data\n        X, y = self.__data_generation(list_IDs_temp)\n\n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.list_IDs))\n        ##print(f'indexes within generator and after shuffle{self.indexes}')\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n            print(f'indexes within generator and after shuffle{self.indexes}')\n\n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        dimension=self.height*self.width*self.depth\n        ##X = np.empty((self.batch_size,1,dimension))\n        X = np.empty((self.batch_size,self.height,self.width,self.depth))\n        y = np.empty((self.batch_size), dtype=int)\n        X_num=np.empty((self.batch_size,1,self.num_size))\n        y_num = np.empty((self.batch_size), dtype=int) ##y_num and y are the same\n\n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            # Store sample\n            \n            image_directory=ID\n            ##print(image_directory)\n            image = Image.open(os.path.join(self.img_dir, image_directory)) \n            image = image.resize((self.height,self.width)) \n            data = asarray(image)\n            data=data/255.0\n            # summarize shape\n            ##print(data.shape[0])\n            ##print(data.shape[0])\n            t_flatten = data.flatten().reshape(1,data.shape[0]*data.shape[1]*data.shape[2])\n            \n            \n            \n            ##X[i,] = t_flatten \n            X[i,] = data\n            ##interm=self.array_num[np.where(self.array_num[:,0] == ID)]\n            ##X_num[i,]=interm[:,[1,2,3]]\n            ##print(f'the shape of X_num[i,]: {X_num[i,].shape}')\n            ##print('I am good')\n\n            # Store class\n            y[i] = self.labels[ID]\n            ##y_num[i] = self.labels[ID]\n            ##y=y.reshape((y[i].shape[0],1))\n\n        ##return X, keras.utils.to_categorical(y, num_classes=self.n_classes)\n        y=y.reshape((y.shape[0],1))\n        ##y_num=y_num.reshape((y_num.shape[0],1))\n        return X,y\n\n    def __next__(self):\n        if self.n >= self.max:\n            self.n = 0\n        result = self.__getitem__(self.n)\n        self.n += 1\n        return result\n    \n    \n\n##array_num stands for categorical and numerical inputs\n##The purpose of this function is to create both image generator and a numerical and categorical\n##Here we are trying to make sure the image and its corresponding inputs are in the same batch with the same label when shuffled\n##through indexes_from_images, which is from image generator, we are making sure both batches(from image and numerical) are both shuffled in the same manner\nclass DataGeneratorNumericalCategorical(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, list_IDs, labels,img_dir,array_num,indexes_from_images, batch_size=32, height=320,width=320,depth=3,\n                 n_classes=1,num_size=3, shuffle=True):\n        'Initialization'\n        self.height=height\n        self.width=width\n        self.depth=depth\n        self.batch_size = batch_size\n        self.labels = labels\n        self.list_IDs = list_IDs\n        self.n_classes = n_classes\n        self.num_size=num_size\n        self.shuffle = shuffle\n        self.img_dir=img_dir\n        self.array_num=array_num\n        self.indexes_from_images=indexes_from_images\n        self.n = 0\n        self.max = self.__len__()\n        self.on_epoch_end()\n        print('I am good')\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n    \n    def __getIndexes__(self):\n        return self.indexes\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Find list of IDs\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n\n        # Generate data\n        X_num,y_num = self.__data_generation(list_IDs_temp)\n\n        return X_num,y_num\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        ##self.indexes = np.arange(len(self.list_IDs))\n        self.indexes =self.indexes_from_images\n        ##Below we are avoiding any shuffle since we shuffled in the image generator\n        ##if self.shuffle == True:\n            ##np.random.shuffle(self.indexes)\n\n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        dimension=self.height*self.width*self.depth\n        ##X = np.empty((self.batch_size,1,dimension))\n        X = np.empty((self.batch_size,self.height,self.width,self.depth))\n        y = np.empty((self.batch_size), dtype=int)\n        ##X_num=np.empty((self.batch_size,1,self.num_size))\n        X_num=np.empty((self.batch_size,self.num_size))\n        y_num = np.empty((self.batch_size), dtype=int) ##y_num and y are the same\n\n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            # Store sample\n            \n            image_directory=ID\n            ##print(image_directory)\n            image = Image.open(os.path.join(self.img_dir, image_directory)) \n            image = image.resize((self.height,self.width)) \n            data = asarray(image)\n            data=data/255.0\n            # summarize shape\n            ##print(data.shape[0])\n            ##print(data.shape[0])\n            t_flatten = data.flatten().reshape(1,data.shape[0]*data.shape[1]*data.shape[2])\n            \n            \n            \n            ##X[i,] = t_flatten \n            X[i,] = data\n            interm=self.array_num[np.where(self.array_num[:,0] == ID)]\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY:TRAIN{self.array_num[:,0]}')\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY:TRAIN{X[i,].shape}')\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY:TRAIN{interm[:,[1,2,3]].shape}')\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY self.array_num:TRAIN{self.array_num}')\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY:TRAIN{X_num.shape}')\n            X_num[i,]=interm[:,[1,2,3]]\n            ##print(f'the shape of X_num[i,]: {X_num[i,].shape}')\n            ##print(f'the shape of interm[:,[1,2,3]] interm: {interm[:,[1,2,3]].shape}')\n            ##print(f'the valueof interm[:,[1,2,3]] interm: {interm[:,[1,2,3]]}')\n            ##print('I am good')\n\n            # Store class\n            y[i] = self.labels[ID]\n            y_num[i] = self.labels[ID]\n            ##y=y.reshape((y[i].shape[0],1))\n\n        ##return X, keras.utils.to_categorical(y, num_classes=self.n_classes)\n        y=y.reshape((y.shape[0],1))\n        y_num=y_num.reshape((y_num.shape[0],1))\n        return X_num,y_num\n    \n    def __next__(self):\n        if self.n >= self.max:\n            self.n = 0\n        result = self.__getitem__(self.n)\n        self.n += 1\n        return result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n##array_num stands for categorical and numerical inputs\n##The purpose of this function is to create both image generator and a numerical and categorical\n##Here we are trying to make sure the image and its corresponding inputs are in the same batch with the same label when shuffled\nclass DataGeneratorTest(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, list_IDs,img_dir,array_num, batch_size=32, height=320,width=320,depth=3,\n                 n_classes=1,num_size=3, shuffle=False):\n        'Initialization'\n        self.height=height\n        self.width=width\n        self.depth=depth\n        self.batch_size = batch_size\n        ##self.labels = labels\n        self.list_IDs = list_IDs\n        self.n_classes = n_classes\n        self.num_size=num_size\n        self.shuffle = shuffle\n        self.img_dir=img_dir\n        self.array_num=array_num\n        self.n = 0\n        self.max = self.__len__()\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n\n    def __getIndexes__(self):\n        return self.indexes\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Find list of IDs\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n\n        # Generate data\n        X = self.__data_generation(list_IDs_temp)\n\n        return X\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.list_IDs))\n        ##print(f'indexes within generator and after shuffle{self.indexes}')\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n            print(f'indexes within generator and after shuffle{self.indexes}')\n\n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        dimension=self.height*self.width*self.depth\n        ##X = np.empty((self.batch_size,1,dimension))\n        X = np.empty((self.batch_size,self.height,self.width,self.depth))\n        ##y = np.empty((self.batch_size), dtype=int)\n        X_num=np.empty((self.batch_size,1,self.num_size))\n        ##y_num = np.empty((self.batch_size), dtype=int) ##y_num and y are the same\n\n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            # Store sample\n            \n            image_directory=ID\n            ##print(f'image directory for test image generator:{image_directory}')\n            ##print(f'image directory for test for image generator:{self.img_dir}')\n            image = Image.open(os.path.join(self.img_dir, image_directory)) \n            image = image.resize((self.height,self.width)) \n            data = asarray(image)\n            data=data/255.0\n            # summarize shape\n            ##print(data.shape[0])\n            ##print(data.shape[0])\n            t_flatten = data.flatten().reshape(1,data.shape[0]*data.shape[1]*data.shape[2])\n            \n            \n            \n            ##X[i,] = t_flatten \n            X[i,] = data\n            ##interm=self.array_num[np.where(self.array_num[:,0] == ID)]\n            ##X_num[i,]=interm[:,[1,2,3]]\n            ##print(f'the shape of X_num[i,]: {X_num[i,].shape}')\n            ##print('I am good')\n\n            # Store class\n            ##y[i] = self.labels[ID]\n            ##y_num[i] = self.labels[ID]\n            ##y=y.reshape((y[i].shape[0],1))\n\n        ##return X, keras.utils.to_categorical(y, num_classes=self.n_classes)\n        ##y=y.reshape((y.shape[0],1))\n        ##y_num=y_num.reshape((y_num.shape[0],1))\n        return X\n\n    def __next__(self):\n        if self.n >= self.max:\n            self.n = 0\n        result = self.__getitem__(self.n)\n        self.n += 1\n        return result\n    \n    \n\n##array_num stands for categorical and numerical inputs\n##The purpose of this function is to create both image generator and a numerical and categorical\n##Here we are trying to make sure the image and its corresponding inputs are in the same batch with the same label when shuffled\n##through indexes_from_images, which is from image generator, we are making sure both batches(from image and numerical) are both shuffled in the same manner\nclass DataGeneratorNumericalCategoricalTest(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, list_IDs,img_dir,array_num,indexes_from_images, batch_size=32, height=320,width=320,depth=3,\n                 n_classes=1,num_size=3, shuffle=False):\n        'Initialization'\n        self.height=height\n        self.width=width\n        self.depth=depth\n        self.batch_size = batch_size\n        ##self.labels = labels\n        self.list_IDs = list_IDs\n        self.n_classes = n_classes\n        self.num_size=num_size\n        self.shuffle = shuffle\n        self.img_dir=img_dir\n        self.array_num=array_num\n        self.indexes_from_images=indexes_from_images\n        self.n = 0\n        self.max = self.__len__()\n        self.on_epoch_end()\n        print('I am good')\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n    \n    def __getIndexes__(self):\n        return self.indexes\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Find list of IDs\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n\n        # Generate data\n        X_num = self.__data_generation(list_IDs_temp)\n\n        return X_num\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        ##self.indexes = np.arange(len(self.list_IDs))\n        self.indexes =self.indexes_from_images\n        ##Below we are avoiding any shuffle since we shuffled in the image generator\n        ##if self.shuffle == True:\n            ##np.random.shuffle(self.indexes)\n\n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        dimension=self.height*self.width*self.depth\n        ##X = np.empty((self.batch_size,1,dimension))\n        X = np.empty((self.batch_size,self.height,self.width,self.depth))\n        ##y = np.empty((self.batch_size), dtype=int)\n        ##X_num=np.empty((self.batch_size,1,self.num_size))\n        X_num=np.empty((self.batch_size,self.num_size))\n        ##y_num = np.empty((self.batch_size), dtype=int) ##y_num and y are the same\n\n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            # Store sample\n            \n            image_directory=ID\n            ##print(f'image directory for test for numerica generator:{image_directory}')\n            ##print(f'image directory for test for numerica generator:{self.img_dir}')\n            image = Image.open(os.path.join(self.img_dir, image_directory)) \n            image = image.resize((self.height,self.width)) \n            data = asarray(image)\n            data=data/255.0\n            # summarize shape\n            ##print(data.shape[0])\n            ##print(data.shape[0])\n            t_flatten = data.flatten().reshape(1,data.shape[0]*data.shape[1]*data.shape[2])\n            \n            \n            \n            ##X[i,] = t_flatten \n            X[i,] = data\n            interm=self.array_num[np.where(self.array_num[:,0] == ID)]\n            print(ID)\n            ##print(f'FUNCTION:size of IMAGE NUMERICAL ARRAY:TEST{self.array_num[:,0]}')\n            ##print(f'FUNCTION:size of IMAGE NUMERICAL ARRAY:TEST{X[i,].shape}')\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY INTERIM:Test{interm[:,[1,2,3]]}')\n            ##print(f'FUNCTION size of  NUMERICAL ARRAY:Test{X_num.shape}')\n           \n            \n            X_num[i,]=interm[:,[1,2,3]]\n            ##print(f'the shape of X_num[i,]: {X_num[i,].shape}')\n            ##print('I am good')\n\n            # Store class\n            ##y[i] = self.labels[ID]\n            ##y_num[i] = self.labels[ID]\n            ##y=y.reshape((y[i].shape[0],1))\n\n        ##return X, keras.utils.to_categorical(y, num_classes=self.n_classes)\n        ##y=y.reshape((y.shape[0],1))\n        ##y_num=y_num.reshape((y_num.shape[0],1))\n        return X_num\n    \n    def __next__(self):\n        if self.n >= self.max:\n            self.n = 0\n        result = self.__getitem__(self.n)\n        self.n += 1\n        return result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def createList_ids_labels(df):\n    train_array =df.to_numpy()\n    print(train_array.shape)\n    ##Retrieving the first row of train_array\n    column_of_imageIDs=train_array[:,0]\n    ##converting the column_of_imageIDs into a list\n    list_IDs = column_of_imageIDs.tolist()\n    print(f'Value of the list at 0th index: {list_IDs[1000]}')\n    ##Creating a numpy array of labels combined with image_name\n    ##image_name\tpatient_id\tsex\tage_approx\tanatom_site_general_challenge\ttarget\n    labels_array = train_array[:, [0, 5]]\n    numerical_array=train_array[:,[0,2,3,4]]\n    print(f'shape of numerical_array as an array train and validation: {numerical_array.shape}')\n    ##Creating a dictionary of labels where image_name are keys\n    labels = dict(labels_array)\n    print(f'shape of labels as an array: {labels_array.shape}')\n    print(f'shape of labels as a dictionary: {len(labels)}')\n    ##print(labo['ISIC_2637011.jpg'])\n    \n    return list_IDs,numerical_array,labels\n\ndef createList_ids_labelsTest(df):\n    test_array =df.to_numpy()\n    print(test_array.shape)\n    ##Retrieving the first row of train_array\n    column_of_imageIDs=test_array[:,0]\n    ##converting the column_of_imageIDs into a list\n    list_IDs = column_of_imageIDs.tolist()\n    print(f'Value of the list at 0th index: {list_IDs[1000]}')\n    ##Creating a numpy array of labels combined with image_name\n    ##image_name\tpatient_id\tsex\tage_approx\tanatom_site_general_challenge\ttarget\n    ##labels_array = train_array[:, [0, 5]]\n    numerical_array=test_array[:,[0,2,3,4]]\n    print(f'shape of numerical_array as an array test: {numerical_array.shape}')\n    ##Creating a dictionary of labels where image_name are keys\n    ##labels = dict(labels_array)\n    ##print(f'shape of labels as an array: {labels_array.shape}')\n    ##print(f'shape of labels as a dictionary: {len(labels)}')\n    ##print(labo['ISIC_2637011.jpg'])\n    \n    return list_IDs,numerical_array","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Parameters\nparams = {'height':320,\n          'width': 320,\n          'depth':3,\n          'batch_size':8,\n          'n_classes': 1,\n          'shuffle': True}\n\nparams_test = {'height':320,\n          'width': 320,\n          'depth':3,\n          'batch_size':1,\n          'n_classes': 1,\n          'shuffle': True}\n\n# Datasets\n##new_train_df_lc, new_validation_df_lc\npartition_train,train_numerical_array, labo_train=createList_ids_labels(new_train_df_lc) ##labo instead of labels to avoid confunsion with labels\npartition_validation,validation_numerical_array,labo_validation=createList_ids_labels(new_validation_df_lc)\n\npartition_test,test_numerical_array=createList_ids_labelsTest(test_df_lc) ##labo instead of labels to avoid confunsion with labels\nprint(f'list of IDs shape within test array:{len(partition_test)}')\nprint(f'list of IDs shape within test_numerical_array:{len(test_numerical_array.shape)}')\n\n\n# Generators\n##DataGeneratorNumericalCategorical\ntraining_generator = DataGenerator(partition_train, labo_train,img_dir,train_numerical_array,**params)\n##print(f'size of NUMERICAL ARRAY:TRAIN{train_numerical_array.shape}')\n##train_generator_images, train_generator_numericalCategorical=....\nvalidation_generator = DataGenerator(partition_validation, labo_validation,img_dir,validation_numerical_array,**params)\ntest_generator = DataGeneratorTest(partition_test,img_dir_test,test_numerical_array,**params_test)\n##print(f'size of IMAGE NUMERICAL ARRAY:TEST{test_numerical_array.shape}')\n##validation_generator = DataGenerator(partition['validation'], labels, **params)\n##validation_generator_images, validation_generator_numericalCategorical=...\n##print(f'shape of labe train: {labo_train}')\ntrain_indexes=training_generator.__getIndexes__()\nvalidation_indexes=validation_generator.__getIndexes__()\ntest_indexes=test_generator.__getIndexes__()\ntraining_generator_numericalCategorical = DataGeneratorNumericalCategorical(partition_train, labo_train,img_dir,train_numerical_array,train_indexes,**params)\n##train_generator_images, train_generator_numericalCategorical=....\nvalidation_generator_numericalCategorical = DataGeneratorNumericalCategorical(partition_validation, labo_validation,img_dir,validation_numerical_array,validation_indexes,**params)\n\ntest_generator_numericalCategorical = DataGeneratorNumericalCategoricalTest(partition_test,img_dir_test,test_numerical_array,test_indexes,**params_test)\n                                                                        ##(list_IDs,img_dir,array_num,indexes_from_images, batch_size=32, height=320,width=320,depth=3,n_classes=1,num_size=3, shuffle=True)\n\ntrain_indexes_num=training_generator_numericalCategorical.__getIndexes__()\nvalidation_indexes_num=validation_generator_numericalCategorical.__getIndexes__()\ntest_indexes_num=test_generator_numericalCategorical.__getIndexes__()\n\n\nnew_train_df_lc.info()\nnew_validation_df_lc.info()\ntest_df_lc.info()\n\n##To check if indexes are the same both in image and numerical generator(train and validation)\nprint(train_indexes)\nprint(train_indexes_num)\nprint(validation_indexes)\nprint(validation_indexes_num)\nprint(test_indexes)\nprint(test_indexes_num)\nprint(img_dir_test)\nprint(img_dir)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n##IMAGE_DIR_TRAIN=\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\"\n##train_generator = get_generator(new_train_df_lc, IMAGE_DIR_TRAIN,features, labels)\n##training_generator_numercalCategorical\n##validation_generator_numercalCategorical\n\nx= test_generator.__getitem__(0)\nx_num= test_generator_numericalCategorical.__getitem__(0)\nx_validation, y_validation = validation_generator.__getitem__(0)\n\nx_validation_num, y_validation_num = validation_generator_numericalCategorical.__getitem__(0)\n\nplt.imshow(x[0])\nprint(f'x train:{x.shape}')\n##(f'y train:{y.shape}')\nprint(f'x train num:{x_num.shape}')\n##print(f'y train num:{y_num.shape}')\n##print(f'y train:{y[0,:]}')\nprint(f'validation:{y_validation.shape}')\n\nindexes=training_generator.__getIndexes__()\nprint(indexes)\ntest_df_lc.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##This function combine two generators: image and numerical generator\n\ndef combine_generator(gen1, gen2):\n    while True:\n        xy = next(gen1) #or next(train_generator)\n        xy1 = next(gen2) #or next(train_generator1)\n        print(xy[0].shape)\n        yield [xy[0],xy1[0]]\n\n\nclass MultipleInputGenerator(keras.utils.Sequence):\n    \"\"\"Wrapper of 2 ImageDataGenerator\"\"\"\n\n    def __init__(self, gen1,gen2):\n        # Keras generator\n       \n        # Real time multiple input data augmentation\n        self.genX1 = gen1\n        self.genX2 = gen2\n\n    def __len__(self):\n        \"\"\"It is mandatory to implement it on Keras Sequence\"\"\"\n        return self.genX1.__len__()\n\n    def __getitem__(self, index):\n        \"\"\"Getting items from the 2 generators and packing them\"\"\"\n        X1_batch, Y_batch = self.genX1.__getitem__(index)\n        X2_batch, Y_batch = self.genX2.__getitem__(index)\n\n        X_batch = [X1_batch, X2_batch]\n       \n\n        return X_batch, Y_batch\n        ##return x,array,y_array\n        \n        \nclass MultipleInputGeneratorTest(keras.utils.Sequence):\n    \"\"\"Wrapper of 2 ImageDataGenerator\"\"\"\n\n    def __init__(self, gen1,gen2):\n        # Keras generator\n       \n        # Real time multiple input data augmentation\n        self.genX1 = gen1\n        self.genX2 = gen2\n\n    def __len__(self):\n        \"\"\"It is mandatory to implement it on Keras Sequence\"\"\"\n        return self.genX1.__len__()\n\n    def __getitem__(self, index):\n        \"\"\"Getting items from the 2 generators and packing them\"\"\"\n        X1_batch = self.genX1.__getitem__(index)\n        X2_batch = self.genX2.__getitem__(index)\n\n        X_batch = [X1_batch, X2_batch]\n       \n\n        return X_batch\n        ##return x,array,y_array\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##train_generator_overall = combine_generator(training_generator_numericalCategorical,training_generator)\n##validation_generator_overall = combine_generator(validation_generator_numericalCategorical, validation_generator)\n##train_generator_overall\n##train_generator_overall=zip(training_generator_numericalCategorical,training_generator)\n##validation_generator_overall=zip(validation_generator_numericalCategorical, validation_generator)\ntrain_generator_overall=MultipleInputGenerator(training_generator_numericalCategorical,training_generator)\nvalidation_generator_overall=MultipleInputGenerator(validation_generator_numericalCategorical, validation_generator)\ntest_generator_overall=MultipleInputGeneratorTest(test_generator_numericalCategorical,test_generator)\nx_batch,y_batch=train_generator_overall.__getitem__(0)\nx_batchtest=test_generator_overall.__getitem__(0)\n\"\"\"\nprint(type(x_batch))\nprint(type(y_batch))\nprint(len(x_batch))\nprint(y_batch.shape)\nprint(type(x_batch[0]))\nprint(type(x_batch[1]))\n\"\"\"\nprint(x_batchtest[0])\nprint(x_batchtest[1])\nprint(len(x_batchtest))\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_train_generator(df, image_dir, x_col, y_cols, shuffle=True, batch_size=8, seed=1, target_w = 320, target_h = 320):\n    \"\"\"\n    Return generator for training set, normalizing using batch\n    statistics.\n\n    Args:\n      train_df (dataframe): dataframe specifying training data.\n      image_dir (str): directory where image files are held.\n      x_col (str): name of column in df that holds filenames.\n      y_cols (list): list of strings that hold y labels for images.\n      sample_size (int): size of sample to use for normalization statistics.\n      batch_size (int): images per batch to be fed into model during training.\n      seed (int): random seed.\n      target_w (int): final width of input images.\n      target_h (int): final height of input images.\n    \n    Returns:\n        train_generator (DataFrameIterator): iterator over training set\n    \"\"\"        \n    print(\"getting train generator...\") \n    # normalize images\n    ##image_generator = ImageDataGenerator(samplewise_center=True,samplewise_std_normalization= True)\n    image_generator =ImageDataGenerator(rescale=1./255)\n    \n    # flow from directory with specified batch size\n    # and target image size\n    generator = image_generator.flow_from_dataframe(\n            dataframe=df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=shuffle,\n            seed=seed,\n            target_size=(target_w,target_h))\n    \n    return generator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_test_and_valid_generator(valid_df, test_df, train_df, img_dir,test_dir, x_col, y_cols, sample_size=100, batch_size=8, seed=1, target_w = 320, target_h = 320):\n    \"\"\"\n    Return generator for validation set and test test set using \n    normalization statistics from training set.\n\n    Args:\n      valid_df (dataframe): dataframe specifying validation data.\n      test_df (dataframe): dataframe specifying test data.\n      train_df (dataframe): dataframe specifying training data.\n      image_dir (str): directory where image files are held.\n      x_col (str): name of column in df that holds filenames.\n      y_cols (list): list of strings that hold y labels for images.\n      sample_size (int): size of sample to use for normalization statistics.\n      batch_size (int): images per batch to be fed into model during training.\n      seed (int): random seed.\n      target_w (int): final width of input images.\n      target_h (int): final height of input images.\n    \n    Returns:\n        test_generator (DataFrameIterator) and valid_generator: iterators over test set and validation set respectively\n    \"\"\"\n    print(\"getting train and valid generators...\")\n    # get generator to sample dataset\n    raw_train_generator = ImageDataGenerator().flow_from_dataframe(\n        dataframe=train_df, \n        directory=img_dir,\n        x_col=\"image_name\", \n        y_col=labels, \n        class_mode=\"raw\", \n        batch_size=sample_size, \n        shuffle=True, \n        target_size=(target_w, target_h))\n    \n    # get data sample\n    batch = raw_train_generator.next()\n    data_sample = batch[0]\n\n    # use sample to fit mean and std for test set generator\n    ##image_generator = ImageDataGenerator(featurewise_center=True,featurewise_std_normalization= True)\n    image_generator =ImageDataGenerator(rescale=1./255)\n    \n    # fit generator to sample from training data\n    image_generator.fit(data_sample)\n\n    # get test generator\n    valid_generator = image_generator.flow_from_dataframe(\n            dataframe=valid_df,\n            directory=img_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=True,\n            seed=seed,\n            target_size=(target_w,target_h))\n   \n    test_generator = image_generator.flow_from_dataframe(\n            dataframe=test_df,\n            directory=test_dir,\n            x_col=x_col,\n            class_mode=None,\n            batch_size=batch_size,\n            shuffle=True,\n            seed=seed,\n            target_size=(target_w,target_h))\n    \n    return valid_generator,test_generator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator_course = get_train_generator(new_train_df_lc, img_dir, \"image_name\", labels)\nvalid_generator_course,test_generator_course= get_test_and_valid_generator(new_validation_df_lc, test_df_lc, new_train_df_lc, img_dir,img_dir_test, \"image_name\", labels)\n##new_train_df_lc_splitted_loaded = pd.read_csv(\"../input/splitted-datasets/new_train_df_lc_splitted_0.csv\")\n##new_validation_df_lc_splitted_loaded = pd.read_csv(\"../input/splitted-datasets/new_validation_df_lc_splitted_0.csv\")\ntrain_generator_course = get_train_generator(new_train_df_lc_splitted_loaded, img_dir, \"image_name\", labels)\nvalid_generator_course,test_generator_course= get_test_and_valid_generator(new_validation_df_lc_splitted_loaded, test_df_lc, new_train_df_lc, img_dir,img_dir_test, \"image_name\", labels)\n\n\nx, y = train_generator_course.__getitem__(0)\nprint(np.max(x[0]))\nprint(np.min(x[0]))\nprint(x[0].shape)\nplt.imshow(x[0]);\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# UNQ_C2 (UNIQUE CELL IDENTIFIER, DO NOT EDIT)\ndef compute_class_freqs(labels):\n    \"\"\"\n    Compute positive and negative frequences for each class.\n\n    Args:\n        labels (np.array): matrix of labels, size (num_examples, num_classes)\n    Returns:\n        positive_frequencies (np.array): array of positive frequences for each\n                                         class, size (num_classes)\n        negative_frequencies (np.array): array of negative frequences for each\n                                         class, size (num_classes)\n    \"\"\"\n    ### START CODE HERE (REPLACE INSTANCES OF 'None' with your code) ###\n    \n    # total number of patients (rows)\n    N = labels.shape[0]\n    \n    positive_frequencies = np.sum(labels, axis=0) / labels.shape[0]\n    negative_frequencies = 1-positive_frequencies\n\n    ### END CODE HERE ###\n    return positive_frequencies, negative_frequencies","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_label = list(training_generator.labels)\ns = pd.Series(training_generator.labels)\nlabel_array = s.values\nlabel_array=label_array.reshape((label_array.shape[0],1))\n##print(label_array)\n##print(training_generator.labels)\nprint(f'the shape of label matrix is: {label_array.shape[0]}')\nfreq_pos, freq_neg = compute_class_freqs(label_array)\nfreq_pos\n##NOTE: Note that these frequencies are the same for both inputs(Images and(numerical categorical))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.DataFrame({\"Class\": labels, \"Label\": \"Positive\", \"Value\": freq_pos})\ndata = data.append([{\"Class\": labels[l], \"Label\": \"Negative\", \"Value\": v} for l,v in enumerate(freq_neg)], ignore_index=True)\nplt.xticks(rotation=90)\nf = sns.barplot(x=\"Class\", y=\"Value\", hue=\"Label\" ,data=data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pos_weights = freq_neg\nneg_weights = freq_pos\npos_contribution = freq_pos * pos_weights \nneg_contribution = freq_neg * neg_weights","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.DataFrame({\"Class\": labels, \"Label\": \"Positive\", \"Value\": pos_contribution})\ndata = data.append([{\"Class\": labels[l], \"Label\": \"Negative\", \"Value\": v} \n                        for l,v in enumerate(neg_contribution)], ignore_index=True)\nplt.xticks(rotation=90)\nsns.barplot(x=\"Class\", y=\"Value\", hue=\"Label\" ,data=data);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# UNQ_C3 (UNIQUE CELL IDENTIFIER, DO NOT EDIT)\ndef get_weighted_loss(pos_weights, neg_weights, epsilon=1e-7):\n    \"\"\"\n    Return weighted loss function given negative weights and positive weights.\n\n    Args:\n      pos_weights (np.array): array of positive weights for each class, size (num_classes)\n      neg_weights (np.array): array of negative weights for each class, size (num_classes)\n    \n    Returns:\n      weighted_loss (function): weighted loss function\n    \"\"\"\n    def weighted_loss(y_true, y_pred):\n        \"\"\"\n        Return weighted loss value. \n\n        Args:\n            y_true (Tensor): Tensor of true labels, size is (num_examples, num_classes)\n            y_pred (Tensor): Tensor of predicted labels, size is (num_examples, num_classes)\n        Returns:\n            loss (Float): overall scalar loss summed across all classes\n        \"\"\"\n        # initialize loss to zero\n        loss = 0.0\n        \n        ### START CODE HERE (REPLACE INSTANCES OF 'None' with your code) ###\n\n        for i in range(len(pos_weights)):\n            # for each class, add average weighted loss for that class \n            loss += -(K.mean( pos_weights[i] * y_true[:,i] * K.log(y_pred[:,i] + epsilon) + neg_weights[i] * (1 - y_true[:,i]) * K.log(1 - y_pred[:,i] + epsilon), axis = 0))#complete this line\n        return loss\n    \n        ### END CODE HERE ###\n    return weighted_loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n##../input/chestxray8-dataframe/densenet.hdf5\n\n# create the base pre-trained model\n\ndef create_cnn_model():\n    base_model = DenseNet121(weights='../input/chestxray8-dataframe/densenet.hdf5', include_top=False)\n\n    x = base_model.output\n\n    # add a global spatial average pooling layer\n    x = GlobalAveragePooling2D()(x)\n\n    # and a logistic layer\n    ##predictions = Dense(len(labels), activation=\"sigmoid\")(x)\n    ##predictions = Dense(2, activation=\"relu\")(x)\n    predictions = x\n    print(f' shape of base_model input cnn:{base_model.input.shape}')\n\n    model = Model(inputs=base_model.input, outputs=predictions)\n    ##model.compile(optimizer='adam', loss=get_weighted_loss(pos_weights, neg_weights),metrics=['accuracy'])\n    return model\ndef create_cnn_model2():\n    base_model = DenseNet121(weights='../input/chestxray8-dataframe/densenet.hdf5', include_top=False)\n\n    x = base_model.output\n\n    # add a global spatial average pooling layer\n   \n    x = GlobalAveragePooling2D()(x)\n    \"\"\"\n    x = Dense(512, activation=\"relu\")(x)\n    x=Dropout(0.2)(x)\n    x = Dense(256, activation=\"relu\")(x)\n    x=Dropout(0.2)(x)\n    x = Dense(128, activation=\"relu\")(x)\n    x = Dense(64, activation=\"relu\")(x)\n    x=Dropout(0.2)(x)\n    \"\"\"\n\n    # and a logistic layer\n    predictions = Dense(len(labels), activation=\"sigmoid\")(x)\n    model = Model(inputs=base_model.input, outputs=predictions)\n    ##opt=keras.optimizers.RMSprop()\n    opt = keras.optimizers.Adam(learning_rate=0.01)\n    model.compile(optimizer=opt, loss=get_weighted_loss(pos_weights, neg_weights),metrics=['accuracy'])\n    return model\n\ndef create_cnn_model_mine(target_dims):\n    model = Sequential()\n    model.add(Conv2D(64, kernel_size=4, strides=1, activation='relu', input_shape=target_dims))\n    model.add(Conv2D(64, kernel_size=4, strides=2, activation='relu'))\n    model.add(Dropout(0.5))\n    model.add(Conv2D(128, kernel_size=4, strides=1, activation='relu'))\n    model.add(Conv2D(128, kernel_size=4, strides=2, activation='relu'))\n    model.add(Dropout(0.5))\n    model.add(Conv2D(256, kernel_size=4, strides=1, activation='relu'))\n    model.add(Conv2D(256, kernel_size=4, strides=2, activation='relu'))\n    model.add(Flatten())\n    model.add(Dense(512, activation='relu'))\n    model.add(Dense(1, activation='sigmoid'))\n    return model;\n\ndef create_mlp_model(dim):\n    # define our MLP network\n    model = Sequential()\n    model.add(Dense(512, input_dim=dim, activation=\"relu\"))\n    model.add(Dropout(0.2))\n    model.add(Dense(128, activation=\"relu\"))\n    model.add(Dropout(0.2))\n    model.add(Dense(64, activation=\"relu\"))\n    model.add(Dropout(0.2))\n    model.add(Dense(8, activation=\"relu\"))\n    model.add(Dropout(0.2))\n    print(f' shape of base_model input mlp:{model.input.shape}')\n    # check to see if the regression node should be added\n    ##model.add(Dense(1, activation=\"sigmoid\"))\n    ##model.add(Dense(8, activation=\"relu\"))\n        # return our model\n    return model\n\ndef createInceptionModel():\n    base_model = applications.InceptionV3(weights='imagenet', \n                                include_top=False, \n                                input_shape=(320,320,3))\n    base_model.trainable = False\n\n    add_model = Sequential()\n    add_model.add(base_model)\n    add_model.add(GlobalAveragePooling2D())\n    add_model.add(Dropout(0.5))\n    add_model.add(Dense(1, \n                        activation='sigmoid'))\n\n    model = add_model\n    model.compile(loss='binary_crossentropy', \n                  optimizer=optimizers.SGD(lr=1e-4, \n                                           momentum=0.9),\n                  metrics=['accuracy'])\n    model.summary()\n    return model\n\ntrain_df_lc.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mlp = create_mlp_model(dim=3)\n##cnn = create_cnn_model()\n##cnn2=create_cnn_model2()\ninception=createInceptionModel()\ninception.summary()\n##cnn = create_cnn_model_mine((320,320,3))\n##combinedInput = concatenate([mlp.output, cnn.output])\n##x = Dense(512, activation=\"relu\")(combinedInput)\n##x=Dropout(0.2)(x)\n##x = Dense(256, activation=\"relu\")(x)\n##x=Dropout(0.2)(x)\n##x = Dense(128, activation=\"relu\")(combinedInput)\n##x = Dense(64, activation=\"relu\")(x)\n##x=Dropout(0.2)(x)\n##x = Dense(1, activation=\"sigmoid\")(x)\n##combined_model = Model(inputs=[mlp.input, cnn.input], outputs=x)\n##combined_model = Model(inputs=[train_generator_numericalCategorical, train_generator_images], outputs=x)\n##opt = keras.optimizers.Adam(learning_rate=0.001)\n##combined_model.compile(optimizer=opt, loss=get_weighted_loss(pos_weights, neg_weights),metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def createDataset(img_dir,df,height,width,depth):\n    t_array=df.to_numpy()\n    df.info()\n    print(t_array.shape)\n\n    ##t_array=t_array[0:5,:]\n    t_rows=t_array.shape[0]\n    final_dimension=height*width*depth+t_array.shape[1]\n    final_array=np.zeros((t_array.shape[0],final_dimension))\n    imageColumn=np.zeros((t_array.shape[0],height*width*depth))\n    ##print(img_dir)\n    for i in range(t_rows):\n        image_id=t_array[i,0]\n        image_directory=image_id\n        ##print(image_directory)\n        image = Image.open(os.path.join(img_dir, image_directory)) \n        image = image.resize((height,width)) \n        data = asarray(image)\n        # summarize shape\n        ##print(data.shape[0])\n        ##print(data.shape[0])\n        t_flatten = data.flatten().reshape(1,data.shape[0]*data.shape[1]*data.shape[2])\n        ##print(f' the shape of flatten image is: {t_flatten.shape}')\n        imageColumn[i,:]=t_flatten\n        ##print(f'image column is: {imageColumn[i,:]}')\n        \n    final_array=np.concatenate((imageColumn, t_array[:,1:t_array.shape[1]]), axis=1)\n    \n    ##print(f'final array is: {final_array}')\n    ##print(final_array.shape)\n    ##print(imageColumn.shape)\n    \n    return final_array\n\"\"\"\ntrain_array=createDataset(img_dir,train_df_lc,height=320,width=320,depth=3)\nsave('/kaggle/working/train_array.npy', train_array)\n\ntest_array=createDataset(img_dir_test,test_df_lc,height=320,width=320,depth=3)\nsave('/kaggle/working/test_array.npy', test_array)\n\n##print(train_array.shape)\n\nloaded_array_train = np.load('/kaggle/working/train_array.npy',allow_pickle=True)\nloaded_array_test = np.load('/kaggle/working/test_array.npy',allow_pickle=True)\n# print the array\n##print(loaded_array.shape)\n##print(loaded_array)\n\ncomparison = np.equal(loaded_array_train,train_array)\ncomparison1 = np.equal(loaded_array_test,test_array)\nequal_arrays = comparison.all()\nequal_arrays1 = comparison1.all()\nprint(equal_arrays)\nprint(equal_arrays1)\n\n\"\"\"\n    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##cnn.summary()\n##cnn.compile(optimizer='adam', loss=get_weighted_loss(pos_weights, neg_weights),metrics=['accuracy'])\n##history0 = cnn.fit_generator(training_generator, steps_per_epoch=100, validation_steps=25, epochs = 3)\n\n##mlp.summary()\n##mlp.compile(optimizer='adam', loss=get_weighted_loss(pos_weights, neg_weights),metrics=['accuracy'])\n##history = mlp.fit_generator(training_generator_numericalCategorical, steps_per_epoch=100, validation_steps=25, epochs = 3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Calculating steps_per_epcohs\nprint(len(train_generator_overall))\ndeno=params['batch_size']\nstepsPerEpochs_train=len(train_generator_overall)/deno\nstepsPerEpochs_validation=len(validation_generator_overall)/deno\nprint(stepsPerEpochs_validation)\nprint(stepsPerEpochs_train)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ACCURACY_THRESHOLD=0.98\nclass myCallback(keras.callbacks.Callback): \n    def on_batch_end(self, epoch, logs={}): \n        if(logs.get('acc') > ACCURACY_THRESHOLD):   \n          print(\"\\nWe have reached %2.2f%% accuracy, so we will stopping training.\" %(acc_thresh*100))   \n        self.model.stop_training = True\n        \n##callbacks = myCallback()\n\ncallbacks = EarlyStopping(monitor = 'accuracy',\n                          min_delta = 0.0001,\n                          patience = 1,\n                          verbose = 1,\n                          restore_best_weights = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Testing cnn model\n##cnn_model=create_cnn_model()\n##cnn_model.compile(optimizer='adam', loss=get_weighted_loss(pos_weights, neg_weights),metrics=['accuracy'])\n##train_generator_overall\n##validation_generator_overall\n##print(type(combined_model.inputs))\n##print(len(combined_model.inputs))\n##combined_model.summary()\n##history = combined_model.fit_generator(train_generator_overall,validation_data=validation_generator_overall, steps_per_epoch=5, validation_steps=2,use_multiprocessing=True,workers=6, epochs =3)\n##history = combined_model.fit_generator(train_generator_overall,validation_data=validation_generator_overall, steps_per_epoch=stepsPerEpochs_train, validation_steps=stepsPerEpochs_validation, epochs =3)\n##history = combined_model.fit_generator(train_generator_overall,validation_data=validation_generator_overall,epochs = 10)\n##history = combined_model.fit_generator(train_generator_overall,validation_data=validation_generator_overall, steps_per_epoch=stepsPerEpochs_train, validation_steps=stepsPerEpochs_validation, epochs =3)\nhistory = inception.fit_generator(train_generator_course, validation_data=valid_generator_course,steps_per_epoch=stepsPerEpochs_train, validation_steps=stepsPerEpochs_validation,epochs = 8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# summarize history for accuracy\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# save model and architecture to single file\n\n\n# serialize model to JSON\nmodel_json =inception.to_json()\nwith open(\"./model.json\", \"w\") as json_file:\n    json_file.write(model_json)\n# serialize weights to HDF5\ninception.save_weights(\"./model.h5\")\nprint(\"Saved model to disk\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##loaded_model = load_model('../input/loaded-model-1/model.h5',custom_objects={'weighted_loss': weighted_loss})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##model.load_weights(\"./input/trained-models-melanoma/model.h5\")\n\n# load json and create model\njson_file = open('../input/loaded-model15/model.json', 'r')\nloaded_model_json = json_file.read()\njson_file.close()\nloaded_model = model_from_json(loaded_model_json)\n# load weights into new model\nloaded_model.load_weights(\"../input/loaded-model15/model.h5\")\nprint(\"Loaded model from disk\")\nprint(len(test_generator_overall))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\npredicted_vals = loaded_model.predict_generator(test_generator_course, steps = len(test_generator_overall))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(predicted_vals)\nprint(predicted_vals.shape)\nnp.amin(predicted_vals)\nfinal_predict = predicted_vals.flatten()\nprint(final_predict.shape)\nsub=pd.DataFrame({'image_name':test_df['image_name'],'target':final_predict})\nsub.to_csv(r'/kaggle/working/submission.csv',index=False)\nprint(test_df_lc.shape)\nprint(np.amax(predicted_vals))\ntest_df.head()\n\n##pd.DataFrame(predicted_vals, columns=['predictions']).to_csv('../input/output/prediction.csv')\n##np.savetxt('C:/Users/Administrator/Downloads/siim-isic-melanoma-classification/trainedModels/predictions.csv',predicted_vals,delimiter=',')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}