{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":104884,"sourceType":"datasetVersion","datasetId":54339},{"sourceId":1150616,"sourceType":"datasetVersion","datasetId":649927},{"sourceId":1193409,"sourceType":"datasetVersion","datasetId":679322},{"sourceId":1243687,"sourceType":"datasetVersion","datasetId":690737}],"dockerImageVersionId":29926,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Main Idea\n\nLets merge these datasets from [topic](https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/154296#864656) by [@andrewmvd](https://www.kaggle.com/andrewmvd):\n\n---\n- [Melanoma Detection Dataset](https://www.kaggle.com/wanderdust/skin-lesion-analysis-toward-melanoma-detection)\n- [Skin Lesion Images for Melanoma Classification](https://www.kaggle.com/andrewmvd/isic-2019)\n- [Skin Cancer MNIST: HAM10000](https://www.kaggle.com/kmader/skin-cancer-mnist-ham10000)\n---\n\n- [SIIM-ISIC Melanoma Classification](https://www.kaggle.com/c/siim-isic-melanoma-classification/data)\n","metadata":{}},{"cell_type":"markdown","source":"# Changelog\n\n\n- v2 initial version\n- v4 add StratifiedGroupKFold\n- v5 remove: skin-lesion-analysis-toward-melanoma-detection (see [here](https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/155859#878163))\n- v7 exclude duplicates (see [here](https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/157701)), add stratification by `count of target_id`, add `folds_13062020.csv` ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom glob import glob\nimport cv2\nfrom skimage import io\nfrom tqdm import tqdm\nimport seaborn as sns","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-14T17:23:36.901934Z","iopub.execute_input":"2024-03-14T17:23:36.902271Z","iopub.status.idle":"2024-03-14T17:23:38.159890Z","shell.execute_reply.started":"2024-03-14T17:23:36.902241Z","shell.execute_reply":"2024-03-14T17:23:38.159183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NEED_IMAGE_SAVE = True\nIM_SIZE = 512","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:24:43.484855Z","iopub.execute_input":"2024-03-14T17:24:43.485256Z","iopub.status.idle":"2024-03-14T17:24:43.489543Z","shell.execute_reply.started":"2024-03-14T17:24:43.485223Z","shell.execute_reply":"2024-03-14T17:24:43.488489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p '512x512-dataset-melanoma'\n!mkdir -p '512x512-test'","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2024-03-14T17:24:47.198180Z","iopub.execute_input":"2024-03-14T17:24:47.198543Z","iopub.status.idle":"2024-03-14T17:24:49.182621Z","shell.execute_reply.started":"2024-03-14T17:24:47.198513Z","shell.execute_reply":"2024-03-14T17:24:49.181522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:24:52.619578Z","iopub.execute_input":"2024-03-14T17:24:52.620017Z","iopub.status.idle":"2024-03-14T17:24:52.721995Z","shell.execute_reply.started":"2024-03-14T17:24:52.619969Z","shell.execute_reply":"2024-03-14T17:24:52.720918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:24:55.555172Z","iopub.execute_input":"2024-03-14T17:24:55.555550Z","iopub.status.idle":"2024-03-14T17:24:55.579504Z","shell.execute_reply.started":"2024-03-14T17:24:55.555515Z","shell.execute_reply":"2024-03-14T17:24:55.578585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['diagnosis'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:24:59.420169Z","iopub.execute_input":"2024-03-14T17:24:59.420503Z","iopub.status.idle":"2024-03-14T17:24:59.433446Z","shell.execute_reply.started":"2024-03-14T17:24:59.420475Z","shell.execute_reply":"2024-03-14T17:24:59.432543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['diagnosis'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:03.651638Z","iopub.execute_input":"2024-03-14T17:25:03.652022Z","iopub.status.idle":"2024-03-14T17:25:03.868685Z","shell.execute_reply.started":"2024-03-14T17:25:03.651988Z","shell.execute_reply":"2024-03-14T17:25:03.867641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# isic 2019","metadata":{}},{"cell_type":"code","source":"df_gt = pd.read_csv('../input/isic-2019/ISIC_2019_Training_GroundTruth.csv')\nimage_id = df_gt.iloc[25]['image']\nimage = cv2.imread(f'../input/isic-2019/ISIC_2019_Training_Input/ISIC_2019_Training_Input/{image_id}.jpg', cv2.IMREAD_COLOR)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\nio.imshow(image);","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:07.807142Z","iopub.execute_input":"2024-03-14T17:25:07.807503Z","iopub.status.idle":"2024-03-14T17:25:08.218656Z","shell.execute_reply.started":"2024-03-14T17:25:07.807473Z","shell.execute_reply":"2024-03-14T17:25:08.217451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_gt.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:12.383767Z","iopub.execute_input":"2024-03-14T17:25:12.384164Z","iopub.status.idle":"2024-03-14T17:25:12.401805Z","shell.execute_reply.started":"2024-03-14T17:25:12.384129Z","shell.execute_reply":"2024-03-14T17:25:12.401047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df_gt == 1.0).idxmax(axis=1).unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:16.188542Z","iopub.execute_input":"2024-03-14T17:25:16.189053Z","iopub.status.idle":"2024-03-14T17:25:16.280877Z","shell.execute_reply.started":"2024-03-14T17:25:16.189005Z","shell.execute_reply":"2024-03-14T17:25:16.280152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_downsampled = df_gt[df_gt['image'].str.contains('downsampled')]\ndf_downsampled.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:19.620035Z","iopub.execute_input":"2024-03-14T17:25:19.620491Z","iopub.status.idle":"2024-03-14T17:25:19.657720Z","shell.execute_reply.started":"2024-03-14T17:25:19.620452Z","shell.execute_reply":"2024-03-14T17:25:19.656787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('[ALL]:', df_gt.shape[0])\nprint('[∩ isic2020]:', len(set(df_train['image_name'].values).intersection(df_gt['image'].values)))\nprint('[downsampled isic2019 ∩ isic2020]:', len(set(df_train['image_name'].values).intersection([\n    image_id[:-12] for image_id in df_downsampled['image'].values\n])))\nprint('[downsampled isic2019 ∩ isic2019]:', len(set(df_gt['image'].values).intersection([\n    image_id[:-12] for image_id in df_downsampled['image'].values\n])))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-14T17:25:23.292900Z","iopub.execute_input":"2024-03-14T17:25:23.293271Z","iopub.status.idle":"2024-03-14T17:25:23.316001Z","shell.execute_reply.started":"2024-03-14T17:25:23.293240Z","shell.execute_reply":"2024-03-14T17:25:23.314722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SLATMD [Almost completely repeated] [Removed]","metadata":{}},{"cell_type":"code","source":"paths = glob('../input/skin-lesion-analysis-toward-melanoma-detection/skin-lesions/*/*/*.jpg')\nimage = cv2.imread(paths[777], cv2.IMREAD_COLOR)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\nio.imshow(image);","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:26.908439Z","iopub.execute_input":"2024-03-14T17:25:26.908773Z","iopub.status.idle":"2024-03-14T17:25:27.903900Z","shell.execute_reply.started":"2024-03-14T17:25:26.908743Z","shell.execute_reply":"2024-03-14T17:25:27.902854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = [path.split('/')[-1][:-4] for path in paths]\nprint('[ALL]:', len(image_ids))\nprint('[∩ isic2020]:', len(set(image_ids).intersection(df_train['image_name'].values)))\nprint('[∩ isic2019]:', len(set(image_ids).intersection(df_gt['image'].values)))\nprint('[∩ isic2019 downsampled]:', len(set(image_ids).intersection([image_id[:-12] for image_id in df_gt[df_gt['image'].str.contains('downsampled')]['image'].values])))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-14T17:25:32.364504Z","iopub.execute_input":"2024-03-14T17:25:32.365080Z","iopub.status.idle":"2024-03-14T17:25:32.400592Z","shell.execute_reply.started":"2024-03-14T17:25:32.365027Z","shell.execute_reply":"2024-03-14T17:25:32.399745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Skin Cancer MNIST: HAM10000 [Repeated]","metadata":{}},{"cell_type":"code","source":"df_meta = pd.read_csv('../input/skin-cancer-mnist-ham10000/HAM10000_metadata.csv')\ndf_meta.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:35.772609Z","iopub.execute_input":"2024-03-14T17:25:35.772961Z","iopub.status.idle":"2024-03-14T17:25:35.814928Z","shell.execute_reply.started":"2024-03-14T17:25:35.772919Z","shell.execute_reply":"2024-03-14T17:25:35.814058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta['localization'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:38.887236Z","iopub.execute_input":"2024-03-14T17:25:38.887580Z","iopub.status.idle":"2024-03-14T17:25:38.894712Z","shell.execute_reply.started":"2024-03-14T17:25:38.887551Z","shell.execute_reply":"2024-03-14T17:25:38.893634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = df_meta.iloc[777]['image_id']\nimage = cv2.imread(f'../input/skin-cancer-mnist-ham10000/HAM10000_images_part_1/{image_id}.jpg', cv2.IMREAD_COLOR)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\nio.imshow(image);","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:25:41.828345Z","iopub.execute_input":"2024-03-14T17:25:41.828684Z","iopub.status.idle":"2024-03-14T17:25:42.132165Z","shell.execute_reply.started":"2024-03-14T17:25:41.828654Z","shell.execute_reply":"2024-03-14T17:25:42.131176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('[ALL]:', df_meta.shape[0])\nprint('[∩ isic2020]:', len(set(df_meta['image_id'].values).intersection(df_train['image_name'].values)))\nprint('[∩ isic2019]:', len(set(df_meta['image_id'].values).intersection(df_gt['image'].values)))\nprint('[∩ slatmd]:', len(set(df_meta['image_id'].values).intersection(image_ids)))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-14T17:25:45.200585Z","iopub.execute_input":"2024-03-14T17:25:45.201008Z","iopub.status.idle":"2024-03-14T17:25:45.216984Z","shell.execute_reply.started":"2024-03-14T17:25:45.200970Z","shell.execute_reply":"2024-03-14T17:25:45.215990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merge datasets & metadata","metadata":{}},{"cell_type":"markdown","source":"* Mel  -> Melanoma \n* NV   -> Melanocytic nevus \n* BCC  -> Basal cell carcinoma \n* AK   -> Actinic keratosis \n* BKL  -> Benign keratosis (solar lentigo / seborrheic keratosis / lichen planus-like keratosis) \n* DF   -> Dermatofibroma\n* VASC -> Vascular lesion \n* SCC  -> Squamous cell carcinoma\n* UNK  -> Unknown","metadata":{}},{"cell_type":"code","source":"# isic2020\ndf_train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv', index_col='image_name')\ndf_train['diagnosis'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T19:58:28.113151Z","iopub.execute_input":"2024-03-14T19:58:28.113496Z","iopub.status.idle":"2024-03-14T19:58:28.202019Z","shell.execute_reply.started":"2024-03-14T19:58:28.113467Z","shell.execute_reply":"2024-03-14T19:58:28.201230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['diagnosis'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T19:58:31.056339Z","iopub.execute_input":"2024-03-14T19:58:31.056698Z","iopub.status.idle":"2024-03-14T19:58:31.071955Z","shell.execute_reply.started":"2024-03-14T19:58:31.056668Z","shell.execute_reply":"2024-03-14T19:58:31.071064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('seborrheic keratosis', 'BKL'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('lichenoid keratosis', 'BKL'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('solar lentigo', 'BKL'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('lentigo NOS', 'BKL'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('cafe-au-lait macule', 'UNK'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('atypical melanocytic proliferation', 'UNK'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('unknown', 'UNK'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('nevus', 'NV'))\ndf_train['diagnosis'] = df_train['diagnosis'].apply(lambda x: x.replace('melanoma', 'MEL'))\ndf_train['diagnosis'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T19:58:33.866620Z","iopub.execute_input":"2024-03-14T19:58:33.867034Z","iopub.status.idle":"2024-03-14T19:58:34.007267Z","shell.execute_reply.started":"2024-03-14T19:58:33.866996Z","shell.execute_reply":"2024-03-14T19:58:34.006426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = {\n    'patient_id' : [],\n    'image_id': [],\n    'target': [],\n    'source': [],\n    'sex': [],\n    'diagnosis': [],\n    'age_approx': [],\n    'anatom_site_general_challenge': [],\n    'height': [],\n    'width': []\n}\n\nfor image_id, row in tqdm(df_train.iterrows(), total=df_train.shape[0]):\n    if image_id in dataset['image_id']:\n        continue\n    dataset['patient_id'].append(row['patient_id'])\n    dataset['image_id'].append(image_id)\n    dataset['target'].append(row['target'])\n    dataset['source'].append('ISIC20')\n    dataset['sex'].append(row['sex'])\n    dataset['diagnosis'].append(row['diagnosis'])\n    dataset['age_approx'].append(row['age_approx'])\n    dataset['anatom_site_general_challenge'].append(row['anatom_site_general_challenge'])\n    dataset['height'] = \"NaN\"\n    dataset['width'] = \"NaN\"\n    if NEED_IMAGE_SAVE:\n        image = cv2.imread(f'../input/siim-isic-melanoma-classification/jpeg/train/{image_id}.jpg', cv2.IMREAD_COLOR)\n        dataset['height'] = image.shape[0]\n        dataset['width'] = image.shape[1]\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (IM_SIZE, IM_SIZE), cv2.INTER_AREA)\n        cv2.imwrite(f'./{IM_SIZE}x{IM_SIZE}-dataset-melanoma/{image_id}.jpg', image)\n\n# isic2019\ndf_gt = pd.read_csv('../input/isic-2019/ISIC_2019_Training_GroundTruth.csv', index_col='image')\ndf_meta = pd.read_csv('../input/isic-2019/ISIC_2019_Training_Metadata.csv', index_col='image')\ndf_meta['diagnosis'] = (df_gt == 1.0).idxmax(axis=1)\n\nfor image_id, row in tqdm(df_meta.iterrows(), total=df_meta.shape[0]):\n    if image_id in dataset['image_id']:\n        continue\n\n    dataset['patient_id'].append(row['lesion_id'])\n    dataset['image_id'].append(image_id)\n    dataset['target'].append(int(df_gt.loc[image_id]['MEL']))\n    dataset['source'].append('ISIC19')\n    dataset['sex'].append(row['sex'])\n    dataset['age_approx'].append(row['age_approx'])\n    dataset['diagnosis'].append(row['diagnosis'])\n    dataset['anatom_site_general_challenge'].append(\n        {'anterior torso': 'torso', 'posterior torso': 'torso'}.get(row['anatom_site_general'], row['anatom_site_general'])\n    )\n    dataset['height'] = \"NaN\"\n    dataset['width'] = \"NaN\"\n    if NEED_IMAGE_SAVE:\n        image = cv2.imread(f'../input/isic-2019/ISIC_2019_Training_Input/ISIC_2019_Training_Input/{image_id}.jpg', cv2.IMREAD_COLOR)\n        dataset['height'] = image.shape[0]\n        dataset['width'] = image.shape[1]\n        image = cv2.resize(image, (IM_SIZE, IM_SIZE), cv2.INTER_AREA)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        cv2.imwrite(f'./{IM_SIZE}x{IM_SIZE}-dataset-melanoma/{image_id}.jpg', image)\n        \n    \ndataset = pd.DataFrame(dataset).set_index('image_id')    ","metadata":{"execution":{"iopub.status.busy":"2024-03-14T19:58:37.832190Z","iopub.execute_input":"2024-03-14T19:58:37.832553Z","iopub.status.idle":"2024-03-14T21:58:23.915870Z","shell.execute_reply.started":"2024-03-14T19:58:37.832517Z","shell.execute_reply":"2024-03-14T21:58:23.914057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:58:36.095110Z","iopub.execute_input":"2024-03-14T21:58:36.095464Z","iopub.status.idle":"2024-03-14T21:58:36.112051Z","shell.execute_reply.started":"2024-03-14T21:58:36.095430Z","shell.execute_reply":"2024-03-14T21:58:36.111227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Excluding duplicates (with [clustering approach](https://www.kaggle.com/shonenkov/dbscan-clustering-check-marking))","metadata":{}},{"cell_type":"code","source":"df_duplicates = pd.read_csv('../input/melanoma-merged-external-data-512x512-jpeg/duplicates_13062020.csv', index_col='image_ids')\ndf_duplicates.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:58:39.133697Z","iopub.execute_input":"2024-03-14T21:58:39.134339Z","iopub.status.idle":"2024-03-14T21:58:39.154676Z","shell.execute_reply.started":"2024-03-14T21:58:39.134288Z","shell.execute_reply":"2024-03-14T21:58:39.153858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_value(duplicate_data, row, field):\n    if row[field] == 1 or duplicate_data.shape[0] <= 2:\n        return duplicate_data.iloc[0][field]\n    if 'ISIC20' in duplicate_data.source.values:\n        duplicate_data = duplicate_data[duplicate_data.source == 'ISIC20']\n    return sorted(duplicate_data[field].value_counts().items(), key=lambda x: -x[1])[0][0]","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:58:42.609041Z","iopub.execute_input":"2024-03-14T21:58:42.609384Z","iopub.status.idle":"2024-03-14T21:58:42.616250Z","shell.execute_reply.started":"2024-03-14T21:58:42.609355Z","shell.execute_reply":"2024-03-14T21:58:42.615351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cleaned_duplicates = {\n    'image_id': [],\n    'patient_id': [],\n    'target': [],\n    'source': [],\n    'sex': [],\n    'age_approx': [],\n    'anatom_site_general_challenge': [],\n}\ndrop_image_ids = []\nfor image_ids, row in df_duplicates.iterrows():\n    image_ids = image_ids.split('.')\n    drop_image_ids.extend(image_ids)\n    duplicate_data = dataset.loc[image_ids].sort_values('source', ascending=False)\n    for field in [    \n        'patient_id',\n        'target',\n        'source',\n        'sex',\n        'age_approx',\n        'anatom_site_general_challenge',\n    ]:\n        cleaned_duplicates[field].append(get_value(duplicate_data, row, field))\n    cleaned_duplicates['image_id'].append(duplicate_data.index.values[0])\n\ncleaned_duplicates = pd.DataFrame(cleaned_duplicates).set_index('image_id')\ncleaned_duplicates.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:58:46.830053Z","iopub.execute_input":"2024-03-14T21:58:46.830413Z","iopub.status.idle":"2024-03-14T21:58:49.999284Z","shell.execute_reply.started":"2024-03-14T21:58:46.830381Z","shell.execute_reply":"2024-03-14T21:58:49.998280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset.drop(drop_image_ids)\ndataset = dataset.append(cleaned_duplicates)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:58:55.012145Z","iopub.execute_input":"2024-03-14T21:58:55.012491Z","iopub.status.idle":"2024-03-14T21:58:55.043767Z","shell.execute_reply.started":"2024-03-14T21:58:55.012462Z","shell.execute_reply":"2024-03-14T21:58:55.042863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:58:58.262677Z","iopub.execute_input":"2024-03-14T21:58:58.263078Z","iopub.status.idle":"2024-03-14T21:58:58.281236Z","shell.execute_reply.started":"2024-03-14T21:58:58.263040Z","shell.execute_reply":"2024-03-14T21:58:58.280474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.to_csv('marking.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:00.813111Z","iopub.execute_input":"2024-03-14T21:59:00.813496Z","iopub.status.idle":"2024-03-14T21:59:01.617257Z","shell.execute_reply.started":"2024-03-14T21:59:00.813463Z","shell.execute_reply":"2024-03-14T21:59:01.616320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple EDA:","metadata":{}},{"cell_type":"code","source":"dataset['source'].hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:05.374337Z","iopub.execute_input":"2024-03-14T21:59:05.374715Z","iopub.status.idle":"2024-03-14T21:59:05.562324Z","shell.execute_reply.started":"2024-03-14T21:59:05.374681Z","shell.execute_reply":"2024-03-14T21:59:05.561303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(dataset['target'].value_counts())\ndataset['target'].hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:08.343506Z","iopub.execute_input":"2024-03-14T21:59:08.343882Z","iopub.status.idle":"2024-03-14T21:59:08.529873Z","shell.execute_reply.started":"2024-03-14T21:59:08.343850Z","shell.execute_reply":"2024-03-14T21:59:08.528819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['diagnosis'].hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:11.198097Z","iopub.execute_input":"2024-03-14T21:59:11.198490Z","iopub.status.idle":"2024-03-14T21:59:11.378764Z","shell.execute_reply.started":"2024-03-14T21:59:11.198459Z","shell.execute_reply":"2024-03-14T21:59:11.377856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['sex'].fillna('unknown').hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:14.318620Z","iopub.execute_input":"2024-03-14T21:59:14.318981Z","iopub.status.idle":"2024-03-14T21:59:14.484012Z","shell.execute_reply.started":"2024-03-14T21:59:14.318930Z","shell.execute_reply":"2024-03-14T21:59:14.483032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['age_approx'].hist(bins=50);","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:17.127210Z","iopub.execute_input":"2024-03-14T21:59:17.127558Z","iopub.status.idle":"2024-03-14T21:59:17.373258Z","shell.execute_reply.started":"2024-03-14T21:59:17.127529Z","shell.execute_reply":"2024-03-14T21:59:17.372399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['anatom_site_general_challenge'].fillna('unknown').value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:20.062225Z","iopub.execute_input":"2024-03-14T21:59:20.062579Z","iopub.status.idle":"2024-03-14T21:59:20.090921Z","shell.execute_reply.started":"2024-03-14T21:59:20.062550Z","shell.execute_reply":"2024-03-14T21:59:20.089902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stratify GroupKFold Splitting\n\nhttps://www.kaggle.com/jakubwasikowski/stratified-group-k-fold-cross-validation","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport random\nimport pandas as pd\nfrom collections import Counter, defaultdict\n\ndef stratified_group_k_fold(X, y, groups, k, seed=None):\n    \"\"\" https://www.kaggle.com/jakubwasikowski/stratified-group-k-fold-cross-validation \"\"\"\n    labels_num = np.max(y) + 1\n    y_counts_per_group = defaultdict(lambda: np.zeros(labels_num))\n    y_distr = Counter()\n    for label, g in zip(y, groups):\n        y_counts_per_group[g][label] += 1\n        y_distr[label] += 1\n\n    y_counts_per_fold = defaultdict(lambda: np.zeros(labels_num))\n    groups_per_fold = defaultdict(set)\n\n    def eval_y_counts_per_fold(y_counts, fold):\n        y_counts_per_fold[fold] += y_counts\n        std_per_label = []\n        for label in range(labels_num):\n            label_std = np.std([y_counts_per_fold[i][label] / y_distr[label] for i in range(k)])\n            std_per_label.append(label_std)\n        y_counts_per_fold[fold] -= y_counts\n        return np.mean(std_per_label)\n    \n    groups_and_y_counts = list(y_counts_per_group.items())\n    random.Random(seed).shuffle(groups_and_y_counts)\n\n    for g, y_counts in tqdm(sorted(groups_and_y_counts, key=lambda x: -np.std(x[1])), total=len(groups_and_y_counts)):\n        best_fold = None\n        min_eval = None\n        for i in range(k):\n            fold_eval = eval_y_counts_per_fold(y_counts, i)\n            if min_eval is None or fold_eval < min_eval:\n                min_eval = fold_eval\n                best_fold = i\n        y_counts_per_fold[best_fold] += y_counts\n        groups_per_fold[best_fold].add(g)\n\n    all_groups = set(groups)\n    for i in range(k):\n        train_groups = all_groups - groups_per_fold[i]\n        test_groups = groups_per_fold[i]\n\n        train_indices = [i for i, g in enumerate(groups) if g in train_groups]\n        test_indices = [i for i, g in enumerate(groups) if g in test_groups]\n\n        yield train_indices, test_indices","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-14T21:59:23.398306Z","iopub.execute_input":"2024-03-14T21:59:23.398647Z","iopub.status.idle":"2024-03-14T21:59:23.417771Z","shell.execute_reply.started":"2024-03-14T21:59:23.398619Z","shell.execute_reply":"2024-03-14T21:59:23.416804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf_folds = pd.read_csv('marking.csv')\ndf_folds['patient_id'] = df_folds['patient_id'].fillna(df_folds['image_id'])\ndf_folds['sex'] = df_folds['sex'].fillna('unknown')\ndf_folds['diagnosis'] = df_folds['diagnosis'].fillna('UNK')\ndf_folds['anatom_site_general_challenge'] = df_folds['anatom_site_general_challenge'].fillna('unknown')\ndf_folds['age_approx'] = df_folds['age_approx'].fillna(round(df_folds['age_approx'].median()))\n\npatient_id_2_count = df_folds[['patient_id', 'image_id']].groupby('patient_id').count()['image_id'].to_dict()\n\ndf_folds = df_folds.set_index('image_id')\n\ndef get_stratify_group(row):\n    stratify_group = row['sex']\n    stratify_group += f'_{row[\"anatom_site_general_challenge\"]}'\n    stratify_group += f'_{row[\"source\"]}'\n    stratify_group += f'_{row[\"target\"]}'\n    stratify_group += f'_{row[\"diagnosis\"]}'\n    patient_id_count = patient_id_2_count[row[\"patient_id\"]]\n    if patient_id_count > 80:\n        stratify_group += f'_80'\n    elif patient_id_count > 60:\n        stratify_group += f'_60'\n    elif patient_id_count > 50:\n        stratify_group += f'_50'\n    elif patient_id_count > 30:\n        stratify_group += f'_30'\n    elif patient_id_count > 20:\n        stratify_group += f'_20'\n    elif patient_id_count > 10:\n        stratify_group += f'_10'\n    else:\n        stratify_group += f'_0'\n    return stratify_group\n\ndf_folds['stratify_group'] = df_folds.apply(get_stratify_group, axis=1)\ndf_folds['stratify_group'] = df_folds['stratify_group'].astype('category').cat.codes","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:26.990017Z","iopub.execute_input":"2024-03-14T21:59:26.990379Z","iopub.status.idle":"2024-03-14T21:59:33.190515Z","shell.execute_reply.started":"2024-03-14T21:59:26.990344Z","shell.execute_reply":"2024-03-14T21:59:33.189656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf_folds.loc[:, 'fold'] = 0\n\nskf = stratified_group_k_fold(X=df_folds.index, y=df_folds['stratify_group'], groups=df_folds['patient_id'], k=5, seed=42)\n\nfor fold_number, (train_index, val_index) in enumerate(skf):\n    df_folds.loc[df_folds.iloc[val_index].index, 'fold'] = fold_number","metadata":{"execution":{"iopub.status.busy":"2024-03-14T21:59:38.309686Z","iopub.execute_input":"2024-03-14T21:59:38.310055Z","iopub.status.idle":"2024-03-14T22:22:20.854741Z","shell.execute_reply.started":"2024-03-14T21:59:38.310019Z","shell.execute_reply":"2024-03-14T22:22:20.853965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(df_folds[df_folds['fold'] == 0]['patient_id'].values).intersection(df_folds[df_folds['fold'] == 1]['patient_id'].values)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:24.765533Z","iopub.execute_input":"2024-03-14T22:23:24.765954Z","iopub.status.idle":"2024-03-14T22:23:24.782659Z","shell.execute_reply.started":"2024-03-14T22:23:24.765905Z","shell.execute_reply":"2024-03-14T22:23:24.781721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_folds[df_folds['fold'] == 0]['target'].hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:28.132220Z","iopub.execute_input":"2024-03-14T22:23:28.132553Z","iopub.status.idle":"2024-03-14T22:23:28.299986Z","shell.execute_reply.started":"2024-03-14T22:23:28.132526Z","shell.execute_reply":"2024-03-14T22:23:28.299040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_folds[df_folds['fold'] == 1]['target'].hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:30.724320Z","iopub.execute_input":"2024-03-14T22:23:30.724656Z","iopub.status.idle":"2024-03-14T22:23:30.883604Z","shell.execute_reply.started":"2024-03-14T22:23:30.724628Z","shell.execute_reply":"2024-03-14T22:23:30.882753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_folds['diagnosis'].hist();","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:34.621926Z","iopub.execute_input":"2024-03-14T22:23:34.622305Z","iopub.status.idle":"2024-03-14T22:23:34.794806Z","shell.execute_reply.started":"2024-03-14T22:23:34.622268Z","shell.execute_reply":"2024-03-14T22:23:34.793935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_folds['diagnosis'].unique().to_list()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:37.331104Z","iopub.execute_input":"2024-03-14T22:23:37.331446Z","iopub.status.idle":"2024-03-14T22:23:37.360729Z","shell.execute_reply.started":"2024-03-14T22:23:37.331418Z","shell.execute_reply":"2024-03-14T22:23:37.359548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_folds.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:40.536871Z","iopub.execute_input":"2024-03-14T22:23:40.537245Z","iopub.status.idle":"2024-03-14T22:23:40.555912Z","shell.execute_reply.started":"2024-03-14T22:23:40.537210Z","shell.execute_reply":"2024-03-14T22:23:40.555115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_folds.to_csv('folds_13062020.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:43.035099Z","iopub.execute_input":"2024-03-14T22:23:43.035469Z","iopub.status.idle":"2024-03-14T22:23:43.467947Z","shell.execute_reply.started":"2024-03-14T22:23:43.035433Z","shell.execute_reply":"2024-03-14T22:23:43.467252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"# test isic2020\ndf_test = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv', index_col='image_name')\nfor image_id, row in tqdm(df_test.iterrows(), total=df_test.shape[0]):   \n    if NEED_IMAGE_SAVE:\n        image = cv2.imread(f'../input/siim-isic-melanoma-classification/jpeg/test/{image_id}.jpg', cv2.IMREAD_COLOR)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (512, 512), cv2.INTER_AREA)\n        cv2.imwrite(f'../input/{IM_SIZE}x{IM_SIZE}-test/{image_id}.jpg', image)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:23:49.088778Z","iopub.execute_input":"2024-03-14T22:23:49.089126Z","iopub.status.idle":"2024-03-14T22:53:41.898476Z","shell.execute_reply.started":"2024-03-14T22:23:49.089097Z","shell.execute_reply":"2024-03-14T22:53:41.897567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -czf /tmp/data.tar.gz .","metadata":{"execution":{"iopub.status.busy":"2024-03-01T17:38:22.655445Z","iopub.execute_input":"2024-03-01T17:38:22.655933Z","iopub.status.idle":"2024-03-01T17:39:18.034244Z","shell.execute_reply.started":"2024-03-01T17:38:22.655862Z","shell.execute_reply":"2024-03-01T17:39:18.032238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf ./*","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mv /tmp/data.tar.gz ./","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# split the data into 80% training data, 10% validation data, and 10% test data","metadata":{}},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:12.707812Z","iopub.execute_input":"2024-03-14T22:56:12.708218Z","iopub.status.idle":"2024-03-14T22:56:12.712282Z","shell.execute_reply.started":"2024-03-14T22:56:12.708184Z","shell.execute_reply":"2024-03-14T22:56:12.711344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the folds CSV file\ndf_folds = pd.read_csv('folds_13062020.csv', index_col='image_id')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:16.925925Z","iopub.execute_input":"2024-03-14T22:56:16.926300Z","iopub.status.idle":"2024-03-14T22:56:17.043824Z","shell.execute_reply.started":"2024-03-14T22:56:16.926264Z","shell.execute_reply":"2024-03-14T22:56:17.042881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the total number of images\ntotal_images = len(df_folds)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:23.702997Z","iopub.execute_input":"2024-03-14T22:56:23.703368Z","iopub.status.idle":"2024-03-14T22:56:23.707415Z","shell.execute_reply.started":"2024-03-14T22:56:23.703332Z","shell.execute_reply":"2024-03-14T22:56:23.706492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of images for each set\ntrain_size = int(0.8 * total_images)\nval_size = int(0.1 * total_images)\ntest_size = total_images - train_size - val_size","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:26.676837Z","iopub.execute_input":"2024-03-14T22:56:26.677183Z","iopub.status.idle":"2024-03-14T22:56:26.681828Z","shell.execute_reply.started":"2024-03-14T22:56:26.677153Z","shell.execute_reply":"2024-03-14T22:56:26.680843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select images for training, validation, and test sets based on fold numbers\ntrain_images = df_folds[df_folds['fold'].isin([0, 1, 2, 3])].index[:train_size]\nval_images = df_folds[df_folds['fold'] == 4].index[:val_size]\ntest_images = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv', index_col='image_name').index[:test_size]","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:30.231434Z","iopub.execute_input":"2024-03-14T22:56:30.231895Z","iopub.status.idle":"2024-03-14T22:56:30.274622Z","shell.execute_reply.started":"2024-03-14T22:56:30.231853Z","shell.execute_reply":"2024-03-14T22:56:30.273650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confirming that there is no overlap between the test set and the training/validation set\nassert len(set(train_images).intersection(val_images)) == 0\nassert len(set(train_images).intersection(test_images)) == 0\nassert len(set(val_images).intersection(test_images)) == 0","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:33.501597Z","iopub.execute_input":"2024-03-14T22:56:33.501927Z","iopub.status.idle":"2024-03-14T22:56:33.528195Z","shell.execute_reply.started":"2024-03-14T22:56:33.501897Z","shell.execute_reply":"2024-03-14T22:56:33.527245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the number of images in each set\nprint(\"Number of images in training set:\", len(train_images))\nprint(\"Number of images in validation set:\", len(val_images))\nprint(\"Number of images in test set:\", len(test_images))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:36.652724Z","iopub.execute_input":"2024-03-14T22:56:36.653085Z","iopub.status.idle":"2024-03-14T22:56:36.659021Z","shell.execute_reply.started":"2024-03-14T22:56:36.653051Z","shell.execute_reply":"2024-03-14T22:56:36.657927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"========>","metadata":{}},{"cell_type":"code","source":"import cv2\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:42.511004Z","iopub.execute_input":"2024-03-14T22:56:42.511357Z","iopub.status.idle":"2024-03-14T22:56:42.515624Z","shell.execute_reply.started":"2024-03-14T22:56:42.511325Z","shell.execute_reply":"2024-03-14T22:56:42.514562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(image_path, target_size):\n    # Read the image\n    image = cv2.imread(image_path)\n    \n    # Convert the image to RGB color space\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Resize the image to the target size\n    image = cv2.resize(image, target_size)\n    \n    # Normalize the pixel values to be between 0 and 1\n    image = image.astype(np.float32) / 255.0\n    \n    # You can add more preprocessing steps here, such as data augmentation\n    \n    return image","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:46.655915Z","iopub.execute_input":"2024-03-14T22:56:46.656305Z","iopub.status.idle":"2024-03-14T22:56:46.662815Z","shell.execute_reply.started":"2024-03-14T22:56:46.656270Z","shell.execute_reply":"2024-03-14T22:56:46.661796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:49.894831Z","iopub.execute_input":"2024-03-14T22:56:49.895202Z","iopub.status.idle":"2024-03-14T22:56:49.899411Z","shell.execute_reply.started":"2024-03-14T22:56:49.895169Z","shell.execute_reply":"2024-03-14T22:56:49.898359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage:\nimage_path = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/ISIC_0052212.jpg'","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:52.352362Z","iopub.execute_input":"2024-03-14T22:56:52.352869Z","iopub.status.idle":"2024-03-14T22:56:52.357255Z","shell.execute_reply.started":"2024-03-14T22:56:52.352819Z","shell.execute_reply":"2024-03-14T22:56:52.356415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess the image\npreprocessed_image = preprocess_image(image_path, target_size=(224, 224))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:55.079579Z","iopub.execute_input":"2024-03-14T22:56:55.079912Z","iopub.status.idle":"2024-03-14T22:56:55.115486Z","shell.execute_reply.started":"2024-03-14T22:56:55.079883Z","shell.execute_reply":"2024-03-14T22:56:55.114525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the original and preprocessed images\nplt.figure(figsize=(8, 4))\nplt.subplot(1, 2, 1)\nplt.title('Original Image')\nplt.imshow(cv2.imread(image_path))\nplt.axis('off')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:56:57.838583Z","iopub.execute_input":"2024-03-14T22:56:57.838929Z","iopub.status.idle":"2024-03-14T22:56:58.095838Z","shell.execute_reply.started":"2024-03-14T22:56:57.838899Z","shell.execute_reply":"2024-03-14T22:56:58.094877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplot(1, 2, 2)\nplt.title('Preprocessed Image')\nplt.imshow(preprocessed_image)\nplt.axis('off')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:57:00.431531Z","iopub.execute_input":"2024-03-14T22:57:00.431898Z","iopub.status.idle":"2024-03-14T22:57:00.515374Z","shell.execute_reply.started":"2024-03-14T22:57:00.431862Z","shell.execute_reply":"2024-03-14T22:57:00.514536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"======","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:46:07.287977Z","iopub.execute_input":"2024-03-03T18:46:07.288473Z","iopub.status.idle":"2024-03-03T18:46:07.293504Z","shell.execute_reply.started":"2024-03-03T18:46:07.288434Z","shell.execute_reply":"2024-03-03T18:46:07.292579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the folds CSV file\ndf_folds = pd.read_csv('folds_13062020.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:54:18.800316Z","iopub.execute_input":"2024-03-03T18:54:18.800660Z","iopub.status.idle":"2024-03-03T18:54:18.922521Z","shell.execute_reply.started":"2024-03-03T18:54:18.800629Z","shell.execute_reply":"2024-03-03T18:54:18.921786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'target' column to string\ndf_folds['image_id'] = df_folds['image_id'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:54:21.299499Z","iopub.execute_input":"2024-03-03T18:54:21.299845Z","iopub.status.idle":"2024-03-03T18:54:21.315888Z","shell.execute_reply.started":"2024-03-03T18:54:21.299810Z","shell.execute_reply":"2024-03-03T18:54:21.315026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'target' column to strings\ndf_folds['target'] = df_folds['target'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:54:24.803196Z","iopub.execute_input":"2024-03-03T18:54:24.803544Z","iopub.status.idle":"2024-03-03T18:54:24.869228Z","shell.execute_reply.started":"2024-03-03T18:54:24.803515Z","shell.execute_reply":"2024-03-03T18:54:24.868454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define image dimensions and batch size\nIMAGE_SIZE = (512, 512)\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:53:48.382649Z","iopub.execute_input":"2024-03-03T18:53:48.382999Z","iopub.status.idle":"2024-03-03T18:53:48.388321Z","shell.execute_reply.started":"2024-03-03T18:53:48.382967Z","shell.execute_reply":"2024-03-03T18:53:48.387426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data generators for training, validation, and testing\ntrain_datagen = ImageDataGenerator(rescale=1./255)\nval_datagen = ImageDataGenerator(rescale=1./255)\ntest_datagen = ImageDataGenerator(rescale=1./255)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:53:50.920300Z","iopub.execute_input":"2024-03-03T18:53:50.920636Z","iopub.status.idle":"2024-03-03T18:53:50.925849Z","shell.execute_reply.started":"2024-03-03T18:53:50.920606Z","shell.execute_reply":"2024-03-03T18:53:50.925011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    dataframe=df_folds[df_folds['fold'].isin([0, 1, 2, 3])],\n    directory='./512x512-dataset-melanoma/',\n    x_col='image_id',\n    y_col='target',\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='binary'\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:54:30.498432Z","iopub.execute_input":"2024-03-03T18:54:30.498783Z","iopub.status.idle":"2024-03-03T18:54:30.710738Z","shell.execute_reply.started":"2024-03-03T18:54:30.498748Z","shell.execute_reply":"2024-03-03T18:54:30.709833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_generator = val_datagen.flow_from_dataframe(\n    dataframe=df_folds[df_folds['fold'] == 4],\n    directory='./512x512-dataset-melanoma/',\n    x_col='image_id',\n    y_col='target',\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='binary'\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:48:37.218882Z","iopub.execute_input":"2024-03-03T18:48:37.219281Z","iopub.status.idle":"2024-03-03T18:48:37.280964Z","shell.execute_reply.started":"2024-03-03T18:48:37.219245Z","shell.execute_reply":"2024-03-03T18:48:37.280116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = test_datagen.flow_from_dataframe(\n    dataframe=pd.read_csv('../input/siim-isic-melanoma-classification/test.csv'),\n    directory='/kaggle/input/melanoma-merged-external-data-512x512-jpeg/512x512-test/512x512-test',\n    x_col='image_name',\n    y_col=None,\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode=None,\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:48:40.130626Z","iopub.execute_input":"2024-03-03T18:48:40.131005Z","iopub.status.idle":"2024-03-03T18:48:40.199647Z","shell.execute_reply.started":"2024-03-03T18:48:40.130968Z","shell.execute_reply":"2024-03-03T18:48:40.198869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(512, 512, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dense(1, activation='sigmoid')\n])","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:49:43.636617Z","iopub.execute_input":"2024-03-03T18:49:43.636981Z","iopub.status.idle":"2024-03-03T18:49:43.726047Z","shell.execute_reply.started":"2024-03-03T18:49:43.636950Z","shell.execute_reply":"2024-03-03T18:49:43.725345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print model summary\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:49:46.524773Z","iopub.execute_input":"2024-03-03T18:49:46.525145Z","iopub.status.idle":"2024-03-03T18:49:46.532635Z","shell.execute_reply.started":"2024-03-03T18:49:46.525111Z","shell.execute_reply":"2024-03-03T18:49:46.531852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:49:49.220005Z","iopub.execute_input":"2024-03-03T18:49:49.220359Z","iopub.status.idle":"2024-03-03T18:49:49.255864Z","shell.execute_reply.started":"2024-03-03T18:49:49.220329Z","shell.execute_reply":"2024-03-03T18:49:49.255263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n    train_generator,\n    steps_per_epoch=train_generator.n // BATCH_SIZE,\n    epochs=10,\n    validation_data=validation_generator,\n    validation_steps=validation_generator.n // BATCH_SIZE\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:49:51.527481Z","iopub.execute_input":"2024-03-03T18:49:51.527850Z","iopub.status.idle":"2024-03-03T18:49:51.567322Z","shell.execute_reply.started":"2024-03-03T18:49:51.527809Z","shell.execute_reply":"2024-03-03T18:49:51.565244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Debugging: Print out relevant information\nprint(\"Number of validated image filenames found:\", len(validation_generator.filenames))\nprint(\"Validation generator directory:\", validation_generator.directory)\nprint(\"Sample filenames from validation generator:\", validation_generator.filenames[:10])\nprint(\"DataFrame containing validation data:\")\nprint(df_folds[df_folds['fold'] == 4])\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:20:04.877896Z","iopub.execute_input":"2024-03-03T18:20:04.878270Z","iopub.status.idle":"2024-03-03T18:20:04.900797Z","shell.execute_reply.started":"2024-03-03T18:20:04.878239Z","shell.execute_reply":"2024-03-03T18:20:04.900016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_folds.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-03T17:31:49.915468Z","iopub.execute_input":"2024-03-03T17:31:49.915804Z","iopub.status.idle":"2024-03-03T17:31:49.926284Z","shell.execute_reply.started":"2024-03-03T17:31:49.915775Z","shell.execute_reply":"2024-03-03T17:31:49.925514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"======","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-03-03T19:32:09.755061Z","iopub.execute_input":"2024-03-03T19:32:09.755415Z","iopub.status.idle":"2024-03-03T19:32:09.761047Z","shell.execute_reply.started":"2024-03-03T19:32:09.755385Z","shell.execute_reply":"2024-03-03T19:32:09.760067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data (adjust the paths as per your data location)\ndf_folds = pd.read_csv('folds_13062020.csv', index_col='image_id')\ndf_train = df_folds[df_folds['fold'].isin([0, 1, 2, 3])]\ndf_val = df_folds[df_folds['fold'] == 4]","metadata":{"execution":{"iopub.status.busy":"2024-03-03T19:32:18.554089Z","iopub.execute_input":"2024-03-03T19:32:18.554454Z","iopub.status.idle":"2024-03-03T19:32:18.689151Z","shell.execute_reply.started":"2024-03-03T19:32:18.554419Z","shell.execute_reply":"2024-03-03T19:32:18.688307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have a directory containing images, you need to prepare your data accordingly\n# Here, 'image_directory' should be replaced with the actual directory containing your images\n# You may also need to adjust image loading and preprocessing according to your specific dataset\nimage_directory = './512x512-dataset-melanoma/'\nX_train = [image_directory + img_name for img_name in df_train.index]\nX_val = [image_directory + img_name for img_name in df_val.index]\ny_train = df_train['target']\ny_val = df_val['target']","metadata":{"execution":{"iopub.status.busy":"2024-03-03T19:40:17.782954Z","iopub.execute_input":"2024-03-03T19:40:17.783349Z","iopub.status.idle":"2024-03-03T19:40:17.809968Z","shell.execute_reply.started":"2024-03-03T19:40:17.783313Z","shell.execute_reply":"2024-03-03T19:40:17.808870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define data preprocessing and augmentation if needed\n# Example: Here, we'll use basic image resizing and rescaling\ndef preprocess_image(image):\n    # Load and preprocess your image here (e.g., resizing, normalization)\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-03-03T19:40:20.347998Z","iopub.execute_input":"2024-03-03T19:40:20.348380Z","iopub.status.idle":"2024-03-03T19:40:20.352676Z","shell.execute_reply.started":"2024-03-03T19:40:20.348345Z","shell.execute_reply":"2024-03-03T19:40:20.351798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply preprocessing to training and validation data\nX_train = np.array([preprocess_image(tf.keras.preprocessing.image.load_img(img, target_size=(512, 512))) for img in X_train])\nX_val = np.array([preprocess_image(tf.keras.preprocessing.image.load_img(img, target_size=(512, 512))) for img in X_val])","metadata":{"execution":{"iopub.status.busy":"2024-03-03T19:40:23.122419Z","iopub.execute_input":"2024-03-03T19:40:23.122778Z","iopub.status.idle":"2024-03-03T19:40:23.174794Z","shell.execute_reply.started":"2024-03-03T19:40:23.122742Z","shell.execute_reply":"2024-03-03T19:40:23.173452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"=======>","metadata":{}},{"cell_type":"code","source":"# Required Libraries\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nimport cv2\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:38:52.392402Z","iopub.execute_input":"2024-03-04T17:38:52.392762Z","iopub.status.idle":"2024-03-04T17:38:52.398778Z","shell.execute_reply.started":"2024-03-04T17:38:52.392733Z","shell.execute_reply":"2024-03-04T17:38:52.397644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to preprocess image\ndef preprocess_image(image_path, target_size):\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = cv2.resize(image, target_size)\n    image = image.astype(np.float32) / 255.0\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:39:04.097484Z","iopub.execute_input":"2024-03-04T17:39:04.097841Z","iopub.status.idle":"2024-03-04T17:39:04.103461Z","shell.execute_reply.started":"2024-03-04T17:39:04.097810Z","shell.execute_reply":"2024-03-04T17:39:04.102605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to build CNN model\ndef build_model(input_shape):\n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', input_shape=input_shape),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Conv2D(128, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')  # Output layer with sigmoid activation for binary classification\n    ])\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:39:16.911991Z","iopub.execute_input":"2024-03-04T17:39:16.912387Z","iopub.status.idle":"2024-03-04T17:39:16.922088Z","shell.execute_reply.started":"2024-03-04T17:39:16.912342Z","shell.execute_reply":"2024-03-04T17:39:16.921006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load folds CSV file\ndf_folds = pd.read_csv('/kaggle/input/melanoma-merged-external-data-512x512-jpeg/folds_13062020.csv', index_col='image_id')","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:40:40.252719Z","iopub.execute_input":"2024-03-04T17:40:40.253098Z","iopub.status.idle":"2024-03-04T17:40:40.413780Z","shell.execute_reply.started":"2024-03-04T17:40:40.253063Z","shell.execute_reply":"2024-03-04T17:40:40.412741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define input shape based on preprocessed image size\ninput_shape = (224, 224, 3)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:40:54.671842Z","iopub.execute_input":"2024-03-04T17:40:54.672241Z","iopub.status.idle":"2024-03-04T17:40:54.676683Z","shell.execute_reply.started":"2024-03-04T17:40:54.672200Z","shell.execute_reply":"2024-03-04T17:40:54.675414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into train, validation, and test sets\ntrain_images = df_folds[df_folds['fold'].isin([0, 1, 2, 3])].index\nval_images = df_folds[df_folds['fold'] == 4].index\ntest_images = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv', index_col='image_name').index","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:41:15.974985Z","iopub.execute_input":"2024-03-04T17:41:15.975420Z","iopub.status.idle":"2024-03-04T17:41:16.023339Z","shell.execute_reply.started":"2024-03-04T17:41:15.975377Z","shell.execute_reply":"2024-03-04T17:41:16.022669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess images\ntrain_data = [preprocess_image('../input/siim-isic-melanoma-classification/jpeg/train/' + img_name + '.jpg', input_shape[:2]) for img_name in train_images]\ntrain_labels = df_folds.loc[train_images, 'target'].values\nval_data = [preprocess_image('../input/siim-isic-melanoma-classification/jpeg/train/' + img_name + '.jpg', input_shape[:2]) for img_name in val_images]\nval_labels = df_folds.loc[val_images, 'target'].values\ntest_data = [preprocess_image('../input/siim-isic-melanoma-classification/jpeg/test/' + img_name + '.jpg', input_shape[:2]) for img_name in test_images]","metadata":{"execution":{"iopub.status.busy":"2024-03-04T17:41:40.808117Z","iopub.execute_input":"2024-03-04T17:41:40.808552Z","iopub.status.idle":"2024-03-04T18:49:51.073634Z","shell.execute_reply.started":"2024-03-04T17:41:40.808511Z","shell.execute_reply":"2024-03-04T18:49:51.071965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data augmentation","metadata":{}},{"cell_type":"code","source":"pip install albumentations","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:57:19.335682Z","iopub.execute_input":"2024-03-14T22:57:19.336063Z","iopub.status.idle":"2024-03-14T22:57:27.664113Z","shell.execute_reply.started":"2024-03-14T22:57:19.336029Z","shell.execute_reply":"2024-03-14T22:57:27.662923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1","metadata":{}},{"cell_type":"code","source":"import albumentations as A","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:57:31.381713Z","iopub.execute_input":"2024-03-14T22:57:31.382098Z","iopub.status.idle":"2024-03-14T22:57:32.072862Z","shell.execute_reply.started":"2024-03-14T22:57:31.382059Z","shell.execute_reply":"2024-03-14T22:57:32.072186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as A\n\ndef preprocess_image(image_path, target_size, augment=True):\n    # Read the image\n    image = cv2.imread(image_path)\n    \n    # Convert the image to RGB color space\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Resize the image to the target size\n    image = cv2.resize(image, target_size)\n    \n    if augment:\n        # Define augmentation transformations\n        transform = A.Compose([\n            A.HorizontalFlip(p=0.5),  # Apply horizontal flip with probability 0.5\n            A.VerticalFlip(p=0.5),    # Apply vertical flip with probability 0.5\n            A.Rotate(limit=30, p=0.5),  # Rotate the image up to 30 degrees with probability 0.5\n            A.RandomBrightnessContrast(p=0.2),  # Randomly adjust brightness and contrast with probability 0.2\n            # You can add more augmentation techniques as needed\n        ])\n        \n        # Apply augmentation\n        augmented = transform(image=image)\n        image = augmented['image']\n    \n    # Normalize the pixel values to be between 0 and 1\n    image = image.astype(np.float32) / 255.0\n    \n    return image\n\n# Example usage:\nimage_path = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/ISIC_0076262.jpg'\n\n# Preprocess the image without augmentation\npreprocessed_image_without_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=False)\n\n# Preprocess the image with augmentation\npreprocessed_image_with_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=True) ","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:57:41.618082Z","iopub.execute_input":"2024-03-14T22:57:41.618437Z","iopub.status.idle":"2024-03-14T22:57:42.258259Z","shell.execute_reply.started":"2024-03-14T22:57:41.618406Z","shell.execute_reply":"2024-03-14T22:57:42.257437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Load the original image\noriginal_image = cv2.imread(image_path)\noriginal_image = cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB)\n\n# Preprocess the original image without augmentation\npreprocessed_image_without_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=False)\n\n# Preprocess the original image with augmentation\npreprocessed_image_with_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=True)\n\n# Plot the images\nfig, axes = plt.subplots(1, 3, figsize=(15, 5))\naxes[0].imshow(original_image)\naxes[0].set_title('Original Image')\n\naxes[1].imshow(preprocessed_image_without_augmentation)\naxes[1].set_title('Preprocessed Image (Without Augmentation)')\n\naxes[2].imshow(preprocessed_image_with_augmentation)\naxes[2].set_title('Preprocessed Image (With Augmentation)')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:57:45.963249Z","iopub.execute_input":"2024-03-14T22:57:45.963619Z","iopub.status.idle":"2024-03-14T22:57:48.905406Z","shell.execute_reply.started":"2024-03-14T22:57:45.963584Z","shell.execute_reply":"2024-03-14T22:57:48.904475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Define augmentation transformations\ntransformations = [\n    A.HorizontalFlip(p=0.5),  # Horizontal flip with probability 0.5\n    A.VerticalFlip(p=0.5),    # Vertical flip with probability 0.5\n    A.Rotate(limit=30, p=0.5),  # Rotate the image up to 30 degrees with probability 0.5\n    A.RandomBrightnessContrast(p=0.2),  # Randomly adjust brightness and contrast with probability 0.2\n]\n\n# Load the original image\noriginal_image = cv2.imread(image_path)\noriginal_image = cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB)\n\n# Preprocess the original image without augmentation\npreprocessed_image_without_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=False)\n\n# Plot the original image\nplt.figure(figsize=(15, 5))\nplt.subplot(1, len(transformations) + 1, 1)\nplt.imshow(original_image)\nplt.title('Original Image')\n\n# Apply and plot each augmentation separately\nfor i, transform in enumerate(transformations):\n    augmented = transform(image=original_image)\n    augmented_image = augmented['image']\n    plt.subplot(1, len(transformations) + 1, i + 2)\n    plt.imshow(augmented_image)\n    plt.title(f'Augmentation {i + 1}')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-05T17:43:16.738334Z","iopub.execute_input":"2024-03-05T17:43:16.738698Z","iopub.status.idle":"2024-03-05T17:43:24.759123Z","shell.execute_reply.started":"2024-03-05T17:43:16.738663Z","shell.execute_reply":"2024-03-05T17:43:24.758074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Example usage:\nimage_path = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/ISIC_0076545.jpg'\n\n# Define augmentation transformations\ntransformations = [\n    A.HorizontalFlip(p=0.5),  # Horizontal flip with probability 0.5\n    A.VerticalFlip(p=0.5),    # Vertical flip with probability 0.5\n    A.Rotate(limit=30, p=0.5),  # Rotate the image up to 30 degrees with probability 0.5\n    A.RandomBrightnessContrast(brightness_limit=0.4, contrast_limit=0.3, p=0.2),  # Randomly adjust brightness and contrast with probability 0.2\n]\n\n# Load the original image\noriginal_image = cv2.imread(image_path)\noriginal_image = cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB)\n\n# Preprocess the original image without augmentation\npreprocessed_image_without_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=False)\n\n\n# Plot the original image\nplt.figure(figsize=(15, 5))\nplt.subplot(1, len(transformations) + 1, 1)\nplt.imshow(original_image)\nplt.title('Original Image')\n\n# Apply and plot each augmentation separately\nfor i, transform in enumerate(transformations):\n    augmented = transform(image=original_image)\n    augmented_image = augmented['image']\n    plt.subplot(1, len(transformations) + 1, i + 2)\n    plt.imshow(augmented_image)\n    plt.title(f'Augmentation {i + 1}')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:57:59.477670Z","iopub.execute_input":"2024-03-14T22:57:59.478049Z","iopub.status.idle":"2024-03-14T22:58:04.341706Z","shell.execute_reply.started":"2024-03-14T22:57:59.478016Z","shell.execute_reply":"2024-03-14T22:58:04.340773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Example usage:\nimage_path = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/ISIC_0076742.jpg'\n\n# Define augmentation transformations\ntransformations = [\n    A.HorizontalFlip(p=0.5),  # Horizontal flip with probability 0.5\n    A.VerticalFlip(p=0.5),    # Vertical flip with probability 0.5\n    A.Rotate(limit=30, p=0.5),  # Rotate the image up to 30 degrees with probability 0.5\n    A.RandomBrightnessContrast(brightness_limit=0.3, contrast_limit=0.3, p=0.2),  # Randomly adjust brightness and contrast with probability 0.2\n]\n\n# Load the original image\noriginal_image = cv2.imread(image_path)\noriginal_image = cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB)\n\n# Preprocess the original image without augmentation\npreprocessed_image_without_augmentation = preprocess_image(image_path, target_size=(224, 224), augment=False)\n\n\n# Plot the original image\nplt.figure(figsize=(15, 5))\nplt.subplot(1, len(transformations) + 1, 1)\nplt.imshow(original_image)\nplt.title('Original Image')\n\n# Apply and plot each augmentation separately\nfor i, transform in enumerate(transformations):\n    augmented = transform(image=original_image)\n    augmented_image = augmented['image']\n    plt.subplot(1, len(transformations) + 1, i + 2)\n    plt.imshow(augmented_image)\n    plt.title(f'Augmentation {i + 1}')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-05T19:03:17.771671Z","iopub.execute_input":"2024-03-05T19:03:17.772008Z","iopub.status.idle":"2024-03-05T19:03:26.072325Z","shell.execute_reply.started":"2024-03-05T19:03:17.771978Z","shell.execute_reply":"2024-03-05T19:03:26.071526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ======>","metadata":{}},{"cell_type":"code","source":"pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:58:37.280614Z","iopub.execute_input":"2024-03-14T22:58:37.281019Z","iopub.status.idle":"2024-03-14T22:58:44.209019Z","shell.execute_reply.started":"2024-03-14T22:58:37.280981Z","shell.execute_reply":"2024-03-14T22:58:44.207483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Required Libraries\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# Define CNN model architecture\ndef build_model(input_shape):\n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', input_shape=input_shape),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Conv2D(128, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')  # Output layer with sigmoid activation for binary classification\n    ])\n    return model\n\n# Load the folds CSV file\ndf_folds = pd.read_csv('folds_13062020.csv', index_col='image_id')\n\n# Get the total number of images\ntotal_images = len(df_folds)\n\n# Calculate the number of images for each set\ntrain_size = int(0.8 * total_images)\nval_size = int(0.1 * total_images)\ntest_size = total_images - train_size - val_size\n\n# Select images for training, validation, and test sets based on fold numbers\ntrain_images = df_folds[df_folds['fold'].isin([0, 1, 2, 3])].index[:train_size]\nval_images = df_folds[df_folds['fold'] == 4].index[:val_size]\ntest_images = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv', index_col='image_name').index[:test_size]\n\n# Preprocess images\ntrain_data = [preprocess_image('../input/siim-isic-melanoma-classification/jpeg/train/' + img_name + '.jpg', (224, 224)) for img_name in tqdm(train_images)]\nval_data = [preprocess_image('../input/siim-isic-melanoma-classification/jpeg/train/' + img_name + '.jpg', (224, 224)) for img_name in tqdm(val_images)]\ntest_data = [preprocess_image('../input/siim-isic-melanoma-classification/jpeg/test/' + img_name + '.jpg', (224, 224)) for img_name in tqdm(test_images)]\n\n# Define input shape based on preprocessed image size\ninput_shape = (224, 224, 3)\n\n# Build model\nmodel = build_model(input_shape)\n\n# Compile model\nmodel.compile(optimizer=Adam(learning_rate=0.001), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train model\nhistory = model.fit(np.array(train_data), df_folds.loc[train_images, 'target'].values, epochs=10, batch_size=32, validation_data=(np.array(val_data), df_folds.loc[val_images, 'target'].values))\n\n# Evaluate model\nloss, accuracy = model.evaluate(np.array(val_data), df_folds.loc[val_images, 'target'].values)\nprint(\"Validation Accuracy:\", accuracy)\n\n# Make predictions on test set\npredictions = model.predict(np.array(test_data))\n\n# Post-processing code can be added here if necessary\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T22:58:55.798593Z","iopub.execute_input":"2024-03-14T22:58:55.799006Z","iopub.status.idle":"2024-03-15T00:18:16.025707Z","shell.execute_reply.started":"2024-03-14T22:58:55.798965Z","shell.execute_reply":"2024-03-15T00:18:16.023890Z"},"trusted":true},"execution_count":null,"outputs":[]}]}