{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimg=np.array(Image.open(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/ISIC_0079038.jpg\"))\n#plt.imshow(img)\nprint(img.shape)\nprint(type(img))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['image_name'] = df['image_name'].apply(lambda x: f'/kaggle/input/siim-isic-melanoma-classification/jpeg/train/{x}.jpg')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=np.array(Image.open(df['image_name'][0]))\nX=np.resize(X,[32,32,3])\nfor i in range(1,len(df)):\n    img=np.array(Image.open(df['image_name'][i]))\n    img=np.resize(img,[32,32,3])\n    X=np.dstack((X,img))\n    print(i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class config:\n    image_width = 128\n    image_height = 128\n    batch_size = 32\n    num_epochs = 20\n    learning_rate = 1e-3\n    valid_size = 0.2\n    base_model = 'se_resnext50_32x4d'\n    seed = 0\n    verbose_step = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Split train and valid while making sure there is no patients simultaneously in train and valid\nimport random\nunique_patient_ids = set(df['patient_id'])\nunique_patient_ids = list(unique_patient_ids)\nrandom.shuffle(unique_patient_ids)\n\ntrain_ids = unique_patient_ids[:int( (1 - config.valid_size) * len(unique_patient_ids))]\nvalid_ids = unique_patient_ids[int( (1 - config.valid_size) * len(unique_patient_ids)):]\n\ntrain_df = df[df['patient_id'].isin(train_ids)].sample(frac=1).reset_index(drop=True)\nvalid_df = df[df['patient_id'].isin(valid_ids)].sample(frac=1).reset_index(drop=True)\n\n# Checking that there is no common patient id\na = set(train_df['patient_id'])\nb = set(valid_df['patient_id'])\nc = a.intersection(b)\n\nassert len(c) == 0, 'Patients simultaneously in training and validation set'\n\n# Checking the size\nprint(f'There are {len(train_df)} samples in the training set.')\nprint(f'There are {len(valid_df)} samples in the validation set.')\nprint(f'There are {len(train_df.query(\"target==1\"))} in training set ({len(train_df.query(\"target==1\")) / len(train_df) * 100: .2f} %)')\nprint(f'There are {len(valid_df.query(\"target==1\"))} in validation set ({len(valid_df.query(\"target==1\")) / len(valid_df) * 100: .2f} %)')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nclass MelanomaDataset:\n    def __init__(self, image_paths, config, resize=True, augmentations=None):\n        self.image_paths = image_paths\n        #self.targets = targets\n        self.augmentations = augmentations\n        self.config = config\n        self.resize = resize\n    \n    def __len__(self):\n        return len(self.image_paths)\n    \n    def __getitem__(self, item):\n        image = Image.open(self.image_paths[item])\n        #targets = self.targets[item]\n        \n        if self.resize:\n            image = image.resize(\n                (self.config.image_width, self.config.image_height), resample=Image.BILINEAR\n            )\n        \n        image = np.array(image)\n        \n        if self.augmentations is not None:\n            augmented = self.augmentations(image=image)\n            image = augmented['image']\n        \n        image = np.transpose(image, (2, 0, 1)).astype(np.float32)\n        \n        return {\n            'image': tf.tensor(image, dtype=tf.float),\n            #'targets': tf.tensor(targets, dtype=tf.long),\n        }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset = MelanomaDataset(\n    image_paths=train_df['image_name'],\n    #targets=train_df['target'],\n    config=config,\n    resize=True,\n    augmentations=None,\n)\n\n#train_targets=tf.Tensor(train_df['target'], shape=(26316,), dtype=tf.float32)\n\nvalid_dataset = MelanomaDataset(\n    image_paths=valid_df['image_name'],\n    #targets=valid_df['target'],\n    config=config,\n    resize=True,\n    augmentations=None,\n )\n\n#test_targets=tf.Tensor(valid_df['target'],shape=(6810,), dtype=tf.float32)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_df['image_name'].shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}