{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":63056,"databundleVersionId":8940774,"sourceType":"competition"},{"sourceId":7457180,"sourceType":"datasetVersion","datasetId":4313900}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Run this to enable CSS types\nfrom IPython.core.display import HTML\n\n# Font stuff\nfont_to_use = \"Barlow\" # \"Verdana\"\nfallback = \"Verdana\"\nfont_import_str = f\"\"\"\n@import url('https://fonts.googleapis.com/css2?family={font_to_use.replace(' ', '+')}:ital,wght@0,100;0,200;0,300;0,400;0,500;0,600;0,700;0,800;0,900;1,100;1,200;1,300;1,400;1,500;1,600;1,700;1,800;1,900&display=swap');\n\"\"\"\n\ndef css_styling(verbose=True):\n    styles = open(\"/kaggle/input/my-css-styles/kaggle_styles.css\", \"r\").read().replace('Verdana', font_to_use) #+f\", {fallback}\")\n    html_str = \"<style>\"+font_import_str+styles+\"\\nb{font-weight:bold;}\\n\"+\"</style>\"\n    if verbose: print(html_str)\n    return HTML(html_str)\n\ncss_styling(False)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-07-07T22:32:14.080334Z","iopub.execute_input":"2024-07-07T22:32:14.080750Z","iopub.status.idle":"2024-07-07T22:32:14.139485Z","shell.execute_reply.started":"2024-07-07T22:32:14.080718Z","shell.execute_reply":"2024-07-07T22:32:14.137744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/skin_cancer_isic_competition_banner.png?raw=true\" width=100% style=\"padding: 0 0 !important; margin: 0 0 !important;\"></center>\n\n<br style=\"margin: 15px;\">\n\n<p style=\"text-align: center; font-size: 15px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</p>\n\n<hr>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em;\">\n    <b style=\"font-size: 16px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n<hr>\n\n<center><b><font size=5 color=\"red\">⚠️ WIP - UNDERGOING FREQUEST UPDATES - WIP ⚠️</font></b></center>\n\n<hr>","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #41C0BA; background-color: #ffffff;\">\n    CHANGELOG\n</h1>\n\n<ul>\n    <li>\n        <b>Version 1-5</b>\n        <ul>\n            <li>Initial Versions</li>\n            <li>Just getting things up and running ...</li>\n        </ul>\n    </li>\n    <li>\n        <b>Version 6</b>\n        <ul>\n            <li>First public version</li>\n            <li>Still super WIP</li>\n        </ul>\n    </li>\n    <li>\n        <b>Version 7-8</b>\n        <ul>\n            <li>Add box plot functionality as an option for continuous variables</li>\n            <li>Add border highlighting to image plot</li>\n            <li>Add batch plotting for feature investigation</li>\n        </ul>\n    </li>\n    <li>\n        <b>Version 9</b>\n        <ul>\n            <li>Add feature preprocessing and investigation and correlation plot</li>\n            <li>Cross validation for random and area based submission</li>\n            <li>Do area based submission to test pipeline...</li>\n        </ul>\n    </li>\n    <li>\n        <b>Version 10-11</b>\n        <ul>\n            <li>Work on LGBM and understanding</li>\n        </ul>\n    </li>\n\n</ul>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #41C0BA; background-color: #ffffff;\">\n    TABLE OF CONTENTS\n</h1>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#introduction\" style=\"text-decoration: none; color: #375B6D;\">1&nbsp;&nbsp;&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#background_information\" style=\"text-decoration: none; color: #375B6D;\">2&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#imports\" style=\"text-decoration: none; color: #375B6D;\">3&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#setup\" style=\"text-decoration: none; color: #375B6D;\">4&nbsp;&nbsp;&nbsp;&nbsp;SETUP AND HELPER FUNCTIONS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#eda\" style=\"text-decoration: none; color: #375B6D;\">5&nbsp;&nbsp;&nbsp;&nbsp;EXPLORATORY DATA ANALYSIS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#baseline\" style=\"text-decoration: none; color: #375B6D;\">6&nbsp;&nbsp;&nbsp;&nbsp;BASELINE SUBMISSION</a></h3>\n\n<hr>\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"introduction\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #41C0BA;\" id=\"introduction\">1&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #375B6D;\" href=\"#toc\">&#10514;</a></h1>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">1.1 <b>WHAT</b> IS THIS?</h3>\n<hr>\n\n<ul>\n    <li>This notebook will follow the authors learning path and highlight relevant terms, information, and useful content about the competition.</li>\n    <li>This notebook will conduct an <b>E</b>xploratory <b>D</b>ata <b>A</b>nalysis for the competition.</li>\n    <li>This notebook <i>may</i> propose an open-source baseline solution.</li>\n</ul>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">1.2 <b>WHY</b> IS THIS?</h3>\n<hr>\n\n<ul>\n    <li>Writing and sharing my learning path and the resulting exploratory data analysis can help improve my own understanding of the competition and the data.</li>\n    <li>Sharing my work may help others who are interested in the competition (or the data). This help may take the form of:\n        <ul>\n            <li>Better understanding the problem and potential common solutions (incl. my baseline).</li>\n            <li>Better understanding of the provided dataset.</li>\n            <li>Better understanding of the background information and research.</li>\n            <li>Better ability to hypothesize new solutions.</li>\n        </ul>\n    </li>\n    <li>Exploratory data analysis is a critical step in any data science project. Sharing my EDA might help others in the competition.</li>\n    <li>Writing and sharing my work is often a fun and rewarding experience! It not only allows me to explore and try different techniques, ideas, and visualizations but also encourages and supports other learners and participants.</li>\n</ul>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">1.3 <b>WHO</b> IS THIS FOR?</h3>\n<hr>\n\n\n<ul>\n    <li>The primary purpose of this notebook is to educate <b>MYSELF</b>, however, my review/learning might be beneficial to others:\n        <ul>\n            <li>Other Kagglers (aka. current and future competition participants).</li>\n            <li>Anyone interested in learning more about using artificial intelligence to tackle biomedical imaging problems.</li>\n        </ul>\n    </li>\n</ul>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">1.4 <b>HOW</b> WILL THIS WORK?</h3>\n<hr>\n\n\n<p>I'm going to assemble some markdown cells (like this one) at the beginning of the notebook to go over some concepts/details/etc.</p>\n\n<p>Following this, I will attempt to walk through the data and understand it better prior to composing a baseline solution.</p>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #41C0BA;\" id=\"background_information\">2&nbsp;&nbsp;BACKGROUND INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #375B6D;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n\nIn an effort to advance the field of automated skin cancer detection, the <b>International Skin Imaging Collaboration (ISIC)</b> has launched <b>this competition</b>. This challenge aims to:\n- Improve accuracy in distinguishing between malignant and benign lesions\n- Enhance efficiency in clinical workflows\n- Develop algorithms capable of prioritizing high-risk lesions\n- Ultimately reduce mortality rates associated with skin cancer through early detection\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">2.1 COMPETITION <b>OVERVIEW</b></h3>\n<hr>\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">PRIMARY TASK DESCRIPTION</b>\n<br>\n<br>\nThe <b>goal of this competition</b> is to <b>detect skin cancer</b> using <b>smartphone-quality images</b> of skin lesions. \n\nThese models could help <b>identify potential cancer cases (top K)</b> in areas without specialized dermatological care, <b>improving early detection and triage</b>.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">HOST TASK DESCRIPTION</b>\n<br>\n<br>\n\nSkin cancer can be deadly if not caught early, but many populations lack specialized dermatologic care. Over the past several years, dermoscopy-based AI algorithms have been shown to benefit clinicians in diagnosing melanoma, basal cell, and squamous cell carcinoma. \n\n<b>However,</b> determining which individuals should see a clinician in the first place has great potential impact. <b>Triaging applications have a significant potential to benefit underserved populations and improve early skin cancer detection, the key factor in long-term patient outcomes.</b>\n\nDermatoscope images reveal morphologic features not visible to the naked eye, but these images are typically <b>only captured in dermatology clinics</b>. Algorithms that benefit people in primary care or non-clinical settings must be <b>adept to evaluating lower quality images</b>. \n\nThis competition <b>leverages 3D TBP (Total Body Photography)</b> to present a novel dataset of every single lesion from thousands of patients across three continents with images resembling cell phone photos. \n\nYou must differentiate histologically-confirmed malignant skin lesions from benign lesions on a patient. Your work will help to improve early diagnosis and disease prognosis by extending the benefits of automated skin cancer detection to a broader population and settings.\n\n<hr>\n\nThe <b>ISIC 2024 Challenge</b> focuses on:\n- Binary classification of skin lesions (malignant vs. benign/intermediate)\n- Utilization of 3D Total Body Photography (TBP) derived images\n- Integration of patient metadata for improved diagnosis\n- Prioritization of lesions for clinical review\n\n<b>3D Total Body Photography (TBP)</b> plays a pivotal role in this challenge, offering comprehensive imaging of a patient's entire skin surface. This technology is crucial for:\n- Detecting new lesions\n- Monitoring changes in existing lesions\n- Providing context for individual lesion assessment\n- Enabling efficient full-body skin examinations\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">HOST PROVIDED BACKGROUND - IMPORTANCE</b>\n<br>\n<br>\n\nThe three major types of skin cancer are:\n1. Basal Cell Carcinoma (BCC)\n2. Squamous Cell Carcinoma (SCC)\n3. Melanoma\n\n<b>BCC and SCC are very common, with over 5 million estimated cases in the US each year, but relatively unlikely to be lethal</b>. The Skin Cancer Foundation estimates that <b>melanoma, the deadliest form of skin cancer, will be diagnosed over 200,000 times in the US in 2024 and that almost 9,000 people will die from the disease</b>. As with other cancers, early and accurate detection—potentially aided by data science—can make treatment more effective.\n\nAdvanced skin cancer is a disfiguring and potentially deadly disease, but if caught early, most can be cured with minor surgery. Automated image analysis tools that allow individuals to assess their own skin lesions may expedite clinical presentation and diagnosis. Better detection of skin cancer presents the opportunity to positively impact hundreds of thousands of people every year.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">HOST PROVIDED BACKGROUND - CLINICAL CONTEXT</b>\n<br>\n<br>\n\n<b>Ugly Duckling Sign for Melanoma Diagnosis</b>\n\n<b>Benign moles <mark>on an individual</mark> tend to resemble each other in terms of color, shape, size, and pattern. Outlier lesions are more likely to be melanoma, an observation known as the “ugly duckling sign”.</b> \n\nHowever, most skin lesion classification algorithms are trained for independent analysis of individual skin lesions. <b>The dataset presented here is novel because it represents each person's lesion phenotype more completely.</b> Algorithms may be able to enhance their diagnostic accuracy when taking into account “context” within the same patient to determine which images represent a cancerous outlier.\n\n<b>Representation of common lesions</b>\n\nDermatologists normally use digital dermoscopy to document the more atypical lesions such as those that undergo biopsy or short-term monitoring. Utilizing this dataset, which includes every lesion from thousands of patients across six continents, helps <b>circumvent the lesion-selection bias inherent to large routinely collected dermoscopy image datasets, where the ordinary benign examples tend to be underrepresented, leading to a theoretical risk of low algorithm specificity when used in non-specialized settings</b>.\n\n<b>Telemedicine</b>\n\nSince the start of COVID-19, telemedicine has become very common. Telemedicine patients are often asked by their providers to submit cellphone photos of their skin conditions. These photos are typically captured by the patient or a family member and the quality of the photos tend to be worse than photos taken in a clinic. AI algorithms that are robust to varying degrees of photo quality could improve quality of care in these situations.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">HOST PROVIDED BACKGROUND - IMAGE MODALITY</b>\n<br>\n<br>\n\n<b>3D Whole Body Photography</b>\n\nDesigned specifically for dermatology, the <b><a href=\"https://www.canfieldsci.com/imaging-systems/vectra-wb360-imaging-system/\">VECTRA WB360</a></b> whole body 3D imaging system captures the entire skin surface in macro quality resolution with a single capture by processing 92 camera images (46 bi-camera positions).\n\n<b>Tiles</b>\n\n* The location of each lesion on the patient is detected automatically and exported as individual 15x15 mm field-of-view cropped images. \n* <b>The test set and training sets are comprised of tiles.</b> \n* Teams are permitted to use other public datasets for developing algorithms.\n\n<b>Dermoscopy</b>\n\n* <b>Dermoscopy refers to the examination of the skin using skin surface microscopy.</b>\n* Dermoscopy requires a high quality magnifying lens and a powerful lighting system (a <b><a href=\"https://dermoscopedia.org/Dermoscopic_equipment\">dermatoscope</a></b>), which illuminate morphologic features not otherwise visible to the naked eye. * Large dermoscopy datasets are available on <b><a href=\"https://www.isic-archive.com/\">ISIC</a></b>\n\n<center><img src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-user-content/o/inbox%2F4972760%2F349a3ae1149d15dc5642063a2d742c88%2Fimage%20type_noexif_240425.jpg?generation=1714060307710359&alt=media\"></center>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">2.2 <b>DATASET</b> OVERVIEW</h3>\n<hr>\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">HIGH LEVEL DATA SUMMARY</b>\n\nIn this challenge you are differentiating benign from malignant cases. For each image (<b><code>isic_id</code></b>) you are assigning the probability (<b><code>target</code></b>) ranging <b><code>[0, 1]</code></b> that the case is malignant.\n\nThe dataset – the SLICE-3D dataset, containing skin lesion image crops extracted from 3D TBP for skin cancer detection – consists of <b>diagnostically labelled images with additional metadata</b>. \n* The images are JPEGs.\n* The associated .csv file contains:\n    * a binary diagnostic label (target)\n    * potential input variables (e.g. age_approx, sex, anatom_site_general, etc.)\n    * additional attributes (e.g. image source and precise diagnosis).\n\n<br>\n\n\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">DATASET DETAILS AND COMPOSITION</b>\n\nTo <b>mimic non-dermoscopic images</b>, this competition uses standardized cropped lesion-images of lesions from 3D Total Body Photography (TBP). Vectra WB360, a 3D TBP product from Canfield Scientific, <b>captures the complete visible cutaneous surface area in one macro-quality resolution tomographic image</b>. An AI-based software then identifies individual lesions on a given 3D capture. This allows for the <b>image capture and identification of all lesions on a patient</b>, which are <b>exported as individual 15x15 mm field-of-view cropped photos</b>. \n\nThe dataset contains every lesion from a subset of thousands of patients seen between the years 2015 and 2024 across nine institutions and three continents.\n\nThe following are examples from the training set:\n* **'Strongly-labelled tiles'** are those whose labels were **derived through histopathology assessment**. \n* **'Weak-labelled tiles'** are those who were **not biopsied** and were considered **'benign'** by a doctor.\n\n<center><img src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-user-content/o/inbox%2F4972760%2F169b1f691322233e7b31aabaf6716ff3%2Fex-tiles.png?generation=1717700538524806&alt=media\"></center>\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">DIRECTORY STRUCTURE</b>\n\n```\n├── /kaggle/working\n└── /kaggle/input/isic-2024-challenge/train-image\n    ├── train-image/\n    │   └── image/\n    │       ├── ISIC_0015670.jpg\n    │       ├── ISIC_0015845.jpg\n    │       └── ...\n    │    \n    ├── (?) test-image/\n    │   └── image/\n    │       └── ...\n    │    \n    ├── sample_submission.csv\n    ├── test-image.hdf5\n    ├── test-metadata.csv\n    ├── train-image.hdf5\n    └── train-metadata.csv\n```\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">DATA FILE DESCRIPTIONS</b>\n\n<b><code>train-image/</code></b>:\n* Contains image files for the training set (available only for training purposes).\n\n<b><code>train-image.hdf5</code></b>:\n* A single HDF5 file containing training image data, with `isic_id` as the key.\n\n<b><code>train-metadata.csv</code></b>:\n* Metadata for the training set\n\n<b><code>test-image.hdf5</code></b>:\n* A single HDF5 file containing test image data with `isic_id` as the key. Initially contains 3 test examples to ensure the inference pipeline works correctly. For final submission, this file is replaced with a hidden test set containing approximately 500k images.\n\n<b><code>test-metadata.csv</code></b>:\n* Metadata for the test subset.\n\n<b><code>sample_submission.csv</code></b>:\n* A sample submission file in the correct format.\n\n<br>\n\n<b>Columns in the <code>train-metadata.csv</code></b>\n\n<table>\n<thead>\n<tr>\n<th>field name</th>\n<th>description</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td><code>target</code></td>\n<td>Binary class {0: benign, 1: malignant}.</td>\n</tr>\n<tr>\n<td><code>lesion_id</code></td>\n<td>Unique lesion identifier. Present in lesions that were <em>manually tagged</em> as a lesion of interest.</td>\n</tr>\n<tr>\n<td><code>iddx_full</code></td>\n<td>Fully classified lesion diagnosis.</td>\n</tr>\n<tr>\n<td><code>iddx_1</code></td>\n<td>First level lesion diagnosis.</td>\n</tr>\n<tr>\n<td><code>iddx_2</code></td>\n<td>Second level lesion diagnosis.</td>\n</tr>\n<tr>\n<td><code>iddx_3</code></td>\n<td>Third level lesion diagnosis.</td>\n</tr>\n<tr>\n<td><code>iddx_4</code></td>\n<td>Fourth level lesion diagnosis.</td>\n</tr>\n<tr>\n<td><code>iddx_5</code></td>\n<td>Fifth level lesion diagnosis.</td>\n</tr>\n<tr>\n<td><code>mel_mitotic_index</code></td>\n<td>Mitotic index of invasive malignant melanomas.</td>\n</tr>\n<tr>\n<td><code>mel_thick_mm</code></td>\n<td>Thickness in depth of melanoma invasion.</td>\n</tr>\n<tr>\n<td><code>tbp_lv_dnn_lesion_confidence</code></td>\n<td>Lesion confidence score (0-100 scale).+</td>\n</tr>\n</tbody>\n</table>\n\n<br>\n\n<b>Columns in both the <code>train-metadata.csv</code> and <code>test-metadata.csv</code></b>\n\n<table>\n<thead>\n<tr>\n<th>field name</th>\n<th>description</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td><code>isic_id</code></td>\n<td>Unique case identifier.</td>\n</tr>\n<tr>\n<td><code>patient_id</code>&nbsp;</td>\n<td>Unique patient identifier.</td>\n</tr>\n<tr>\n<td><code>age_approx</code></td>\n<td>Approximate age of patient at time of imaging.</td>\n</tr>\n<tr>\n<td><code>sex</code></td>\n<td>Sex of the person.</td>\n</tr>\n<tr>\n<td><code>anatom_site_general</code></td>\n<td>Location of the lesion on the patient's body.</td>\n</tr>\n<tr>\n<td><code>clin_size_long_diam_mm</code></td>\n<td>Maximum diameter of the lesion (mm).+</td>\n</tr>\n<tr>\n<td><code>image_type</code></td>\n<td>Structured field of the ISIC Archive for image type.</td>\n</tr>\n<tr>\n<td><code>tbp_tile_type</code></td>\n<td>Lighting modality of the 3D TBP source image.</td>\n</tr>\n<tr>\n<td><code>tbp_lv_A</code></td>\n<td>A inside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_Aex</code></td>\n<td>A outside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_B</code></td>\n<td>B inside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_Bext</code></td>\n<td>B outside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_C</code></td>\n<td>Chroma inside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_Cext</code></td>\n<td>Chroma outside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_H</code></td>\n<td>Hue inside the lesion; calculated as the angle of A* and B* in L<em>A</em>B* color space. Typical values range from 25 (red) to 75 (brown).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_Hext</code></td>\n<td>Hue outside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_L</code></td>\n<td>L inside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_Lext</code></td>\n<td>L outside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_areaMM2</code></td>\n<td>Area of lesion (mm^2).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_area_perim_ratio</code></td>\n<td>Border jaggedness, the ratio between lesions perimeter and area. Circular lesions will have low values; irregular shaped lesions will have higher values. Values range 0-10.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_color_std_mean</code></td>\n<td>Color irregularity, calculated as the variance of colors within the lesion's boundary.</td>\n</tr>\n<tr>\n<td><code>tbp_lv_deltaA</code></td>\n<td>Average A contrast (inside vs. outside lesion).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_deltaB</code></td>\n<td>Average B contrast (inside vs. outside lesion).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_deltaL</code></td>\n<td>Average L contrast (inside vs. outside lesion).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_deltaLBnorm</code></td>\n<td>Contrast between the lesion and its immediate surrounding skin. Low contrast lesions tend to be faintly visible such as freckles; high contrast lesions tend to be those with darker pigment. Calculated as the average delta L<em>B</em> of the lesion relative to its immediate background in L<em>A</em>B* color space. Typical values range from 5.5 to 25.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_eccentricity</code></td>\n<td>Eccentricity.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_location</code></td>\n<td>Classification of anatomical location, divides arms &amp; legs to upper &amp; lower; torso into thirds.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_location_simple</code></td>\n<td>Classification of anatomical location, simple.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_minorAxisMM</code></td>\n<td>Smallest lesion diameter (mm).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_nevi_confidence</code></td>\n<td>Nevus confidence score (0-100 scale) is a convolutional neural network classifier estimated probability that the lesion is a nevus. The neural network was trained on approximately 57,000 lesions that were classified and labeled by a dermatologist.+,++</td>\n</tr>\n<tr>\n<td><code>tbp_lv_norm_border</code></td>\n<td>Border irregularity (0-10 scale); the normalized average of border jaggedness and asymmetry.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_norm_color</code></td>\n<td>Color variation (0-10 scale); the normalized average of color asymmetry and color irregularity.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_perimeterMM</code></td>\n<td>Perimeter of lesion (mm).+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_radial_color_std_max</code></td>\n<td>Color asymmetry, a measure of asymmetry of the spatial distribution of color within the lesion. This score is calculated by looking at the average standard deviation in L<em>A</em>B* color space within concentric rings originating from the lesion center. Values range 0-10.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_stdL</code></td>\n<td>Standard deviation of L inside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_Lext</code></td>\n<td>Standard deviation of L outside lesion.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_symm_2axis</code></td>\n<td>Border asymmetry; a measure of asymmetry of the lesion's contour about an axis perpendicular to the lesion's most symmetric axis. Lesions with two axes of symmetry will therefore have low scores (more symmetric), while lesions with only one or zero axes of symmetry will have higher scores (less symmetric). This score is calculated by comparing opposite halves of the lesion contour over many degrees of rotation. The angle where the halves are most similar identifies the principal axis of symmetry, while the second axis of symmetry is perpendicular to the principal axis. Border asymmetry is reported as the asymmetry value about this second axis. Values range 0-10.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_symm_2axis_angle</code></td>\n<td>Lesion border asymmetry angle.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_x</code></td>\n<td>X-coordinate of the lesion on 3D TBP.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_y</code></td>\n<td>Y-coordinate of the lesion on 3D TBP.+</td>\n</tr>\n<tr>\n<td><code>tbp_lv_z</code></td>\n<td>Z-coordinate of the lesion on 3D TBP.+</td>\n</tr>\n<tr>\n<td><code>attribution</code></td>\n<td>Image attribution, synonymous with image source.</td>\n</tr>\n<tr>\n<td><code>copyright_license</code></td>\n<td>Copyright license.</td>\n</tr>\n</tbody>\n</table>\n\n---\n\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">SAMPLE SUBMISSION EXAMPLE</b>\n\nFor each image (<b><code>isic_id</code></b>) in the test set, you must predict the probability (<b><code>target</code></b>) that the lesion is malignant. \n\nThe file should contain a header and have the following format:\n\n<pre><code>isic_id,target\nISIC_0015657,0.7\nISIC_0015729,0.9\nISIC_0015740,0.8\netc.</code></pre>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">2.3 <b>EVALUATION</b></h3>\n<hr>\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">PRIMARY EVALUATION - OVERVIEW</b>\n\nSubmissions are evaluated on <b><a href=\"https://en.wikipedia.org/wiki/Partial_Area_Under_the_ROC_Curve\">partial area under the ROC curve (pAUC)</a></b> above 80% true positive rate (TPR) for binary classification of malignant examples. (See the implementation in the <b><a href=\"https://www.kaggle.com/code/metric/isic-pauc-abovetpr\">notebook ISIC pAUC-aboveTPR.</a></b>)\n\nThe receiver operating characteristic (ROC) curve illustrates the diagnostic ability of a given binary classifier system as its discrimination threshold is varied. However, there are regions in the ROC space where the values of TPR are unacceptable in clinical practice. Systems that aid in diagnosing cancers are required to be highly-sensitive, so this metric focuses on the area under the ROC curve AND above 80% TRP. Hence, scores range from [0.0, 0.2].\n\nThe shaded regions in the following example represents the pAUC of two arbitrary algorithms (Ca and Cb) at an arbitrary minimum TPR:\n\n<center><img src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-user-content/o/inbox%2F4972760%2Ff9089439a6256c84a1d98a83910a46a0%2FJiang.png?generation=1717599679915410&alt=media\"></center>\n\n<br>\n\n<b><a href=\"https://en.wikipedia.org/wiki/Partial_Area_Under_the_ROC_Curve#/media/File:Jiang.png\">\"pAUC defined by constraining TPR\" by ProfGigio is licensed under CC-BY-SA-4.0</a></b>\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">SECONDARY EVALUATION - OVERVIEW</b>\n\n<b>TOP-15 RETRIEVAL SENSITIVITY - SECONDARY PRIZE</b>\n\n**Consider a dermatologist conducting a full body skin exam for each patient that visits the clinic.** \n* Each patient undergoes 3D TBP prior to meeting the dermatologist in the examination room. * The dermatologist has just a few minutes to spend with each patient, which is not enough time to view every lesion with their trusted dermatoscope. \n* It would be helpful if, by the time the dermatologist walked into the room, an AI algorithm efficiently recommended an arbitrary number of each patient's most high-risk lesions.\n\n<b>To address this triaging application, one secondary prize will be awarded to the algorithm that is most successful in scoring malignancies within the top-15 highest scored images per patient.</b> \n* In the event of a tie, the algorithm that ranks the detected malignancies highest among those top-15 lesions per patient will win the secondary prize. \n* All primary submissions will be considered for this secondary prize.\n* The scoring algorithm counts the number of positive samples found among the highest 15 scored images per patient. \n* We adjust the count based on the number of malignancies per patient. \n* For example:\n    * If a patient has only one malignancy, and that one is found, that counts as 1. \n    * If a patient has 3 malignancies, and 2 are in the top-15 scores, it counts as 0.667.\n    * Next, we sum these adjusted values and divide by the number of patients with malignancies. \n    * The result is the \"average found malignancies, weighted by patient-malignancies\" to determine the winner. \n* The code for computing this metric across a set of submissions can be found at the following repository: [https://github.com/ISIC-Research/Challenge-2024-Metrics/tree/main](https://github.com/ISIC-Research/Challenge-2024-Metrics/tree/main)\n\n<br>\n\n<b>MODEL EFFICIENCY - SECONDARY PRIZE</b>\n\n<b>For the efficiency prize, we will evaluate submissions on both runtime and diagnostic accuracy. The objective is to minimize the efficiency score.</b>\n* To be eligible for the Efficiency Prize, a submission:\n    * Must be among the submissions selected by a team for the Leaderboard Prize\n    * Must be ranked on the Private Leaderboard higher than the `BENCHMARK.csv` benchmark\n    * All submissions meeting these conditions will be considered for the Efficiency Prize.\n    * A submission may be eligible for both the Leaderboard Prize and the Efficiency Prize.\n* The Efficiency Prize will be awarded to one eligible submission that scores the best according to the following evaluation metric on the private test data. \n* More details may be posted via discussion forum updates.\n* We compute a submission's **efficiency** score by:<br>\n$$\n\\text{Efficiency} = \\frac{\\text{pAUC}}{\\text{BENCHMARK} - \\max\\text{pAUC}} + \\frac{\\text{RuntimeSeconds}}{43200}\n$$\n* where: \n    * `pAUC` is the submission's score on the leaderboard, \n    * `BENCHMARK`, is the score of the \n    * `benchmark.csv` leaderboard, \n    * `maxpAUC` is the highest score on the leaderboard,\n    * `RuntimeSeconds` is the number of seconds it takes for the submission to be evaluated. \n* During the training period of the competition, a leaderboard for the public test data may be updated periodically and posted in a notebook that will be updated daily: **[Efficiency Leaderboard](https://www.kaggle.com/code/inversion/isic-2024-efficiency-lb)**. \n* After the competition ends, we will update the efficiency leaderboard with scores on the *private* test set.\n","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">2.4 <b>IS THIS A CODE COMPETITION?</b></h3>\n<hr>\n\n<br>\n\n<b style=\"font-size: 32px !important; color: red;\">YES</b>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #41C0BA;\" id=\"imports\">3&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #375B6D;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n","metadata":{}},{"cell_type":"code","source":"# print(\"\\n... PIP INSTALLS STARTING ...\\n\")\n# print(\"\\n... PIP INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Competition Specific Import\n# TBD\n\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None); import pandas;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.metrics import roc_curve, auc, roc_auc_score\nimport cv2; print(f\"\\t\\t– CV2 VERSION: {cv2.__version__}\");\n\n# For modelling and dataset\nimport lightgbm as lgb\nfrom imblearn.pipeline import Pipeline\nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.preprocessing import OrdinalEncoder\n\n# Built-In Imports (mostly don't worry about these)\nfrom typing import Iterable, Any, Callable, Generator\nfrom kaggle_datasets import KaggleDatasets\nfrom dataclasses import dataclass\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport subprocess\nimport warnings\nimport requests\nimport textwrap\nimport hashlib\nimport imageio\nimport IPython\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport h5py\nimport json\nimport copy\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport gc\nimport re\nimport os\n\n# Visualization Imports (overkill)\nfrom IPython.core.display import HTML, Markdown\nimport matplotlib.pyplot as plt\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.graph_objects as go\nimport plotly.express as px\nimport plotly\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance, ImageColor; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom colorama import Fore, Style, init; init()\nimport PIL\n\ndef hex_to_rgb(hex_color: str) -> tuple:\n    \"\"\"Convert hex color to RGB tuple.\n\n    Args:\n        hex_color (str): The hex color string, starting with '#'.\n\n    Returns:\n        tuple: A tuple of RGB values.\n    \"\"\"\n    hex_color = hex_color.lstrip('#')\n    return tuple(int(hex_color[i:i+2], 16) for i in (0, 2, 4))\n\ndef clr_print(text: str, color: str = \"#42BFBA\", bold: bool = True) -> None:\n    \"\"\"Print the given text with the specified color and bold formatting.\n\n    Args:\n        text (str): The text to format.\n        color (str): The hex color code to apply. Defaults to \"#752F55\".\n        bold (bool): Whether to apply bold formatting. Defaults to True.\n    \"\"\"\n    _text = text.replace('\\n', '<br>')\n    rgb = hex_to_rgb(color)\n    color_style = f\"color: rgb({rgb[0]}, {rgb[1]}, {rgb[2]});\"\n    bold_style = \"font-weight: bold;\" if bold else \"\"\n    style = f\"{color_style} {bold_style}\"\n    display(HTML(f\"<span style='{style}'>{_text}</span>\"))\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    # tf.random.set_seed(seed)\n    \nseed_it_all()\n\n# Create a Seaborn color palette\nnb_palette = sns.color_palette(palette='tab20')\n\n# Create colors for class labels\nLABELS = [\"Benign\", \"Malignant\"]\nCOLORS = ['#66c2a5', '#fc8d62']\nCLR_MAP_I2C = {i:c for i,c in enumerate(COLORS)}\nCLR_MAP_S2C = {l:c for l,c in zip(LABELS, COLORS)}\nLBL_MAP_I2S = {i:l for i,l in enumerate(LABELS)}\nLBL_MAP_S2I = {v:k for k,v in LBL_MAP_I2S.items()}\n\n# Is this notebook being run on the backend for scoring re-submission\nIS_DEBUG = False if os.getenv('KAGGLE_IS_COMPETITION_RERUN') else True\nprint(f\"IS DEBUG: {IS_DEBUG}\")\n\n# Plot the palette\nclr_print(\"\\n... NOTEBOOK COLOUR PALETTE ...\")\nsns.palplot(nb_palette, size=0.5)\nplt.show()\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:32:14.142640Z","iopub.execute_input":"2024-07-07T22:32:14.143108Z","iopub.status.idle":"2024-07-07T22:32:20.688479Z","shell.execute_reply.started":"2024-07-07T22:32:14.143064Z","shell.execute_reply":"2024-07-07T22:32:20.686322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"setup\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #41C0BA;\" id=\"setup\">4&nbsp;&nbsp;SETUP & HELPER FUNCTIONS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #375B6D;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">4.0 FUNCTIONS FROM <b>OTHER KAGGLERS</b> 🧡</h3>\n<hr><br>\n\n1. Prize Scoring Metric","metadata":{}},{"cell_type":"code","source":"\"\"\"2024 ISIC Challenge primary prize scoring metric\n\nGiven a list of binary labels, an associated list of prediction \nscores ranging from [0,1], this function produces, as a single value, \nthe partial area under the receiver operating characteristic (pAUC) \nabove a given true positive rate (TPR).\nhttps://en.wikipedia.org/wiki/Partial_Area_Under_the_ROC_Curve.\n\n(c) 2024 Nicholas R Kurtansky, MSKCC\n\"\"\"\n\n\nclass ParticipantVisibleError(Exception):\n    pass\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str, min_tpr: float=0.80) -> float:\n    '''\n    2024 ISIC Challenge metric: pAUC\n    \n    Given a solution file and submission file, this function returns the\n    the partial area under the receiver operating characteristic (pAUC) \n    above a given true positive rate (TPR) = 0.80.\n    https://en.wikipedia.org/wiki/Partial_Area_Under_the_ROC_Curve.\n    \n    (c) 2024 Nicholas R Kurtansky, MSKCC\n\n    Args:\n        solution: ground truth pd.DataFrame of 1s and 0s\n        submission: solution dataframe of predictions of scores ranging [0, 1]\n\n    Returns:\n        Float value range [0, max_fpr]\n    '''\n\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    # check submission is numeric\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('Submission target column must be numeric')\n\n    # rescale the target. set 0s to 1s and 1s to 0s (since sklearn only has max_fpr)\n    v_gt = abs(np.asarray(solution.values)-1)\n    \n    # flip the submissions to their compliments\n    v_pred = -1.0*np.asarray(submission.values)\n\n    max_fpr = abs(1-min_tpr)\n\n    # using sklearn.metric functions: (1) roc_curve and (2) auc\n    fpr, tpr, _ = roc_curve(v_gt, v_pred, sample_weight=None)\n    if max_fpr is None or max_fpr == 1:\n        return auc(fpr, tpr)\n    if max_fpr <= 0 or max_fpr > 1:\n        raise ValueError(\"Expected min_tpr in range [0, 1), got: %r\" % min_tpr)\n        \n    # Add a single point at max_fpr by linear interpolation\n    stop = np.searchsorted(fpr, max_fpr, \"right\")\n    x_interp = [fpr[stop - 1], fpr[stop]]\n    y_interp = [tpr[stop - 1], tpr[stop]]\n    tpr = np.append(tpr[:stop], np.interp(max_fpr, x_interp, y_interp))\n    fpr = np.append(fpr[:stop], max_fpr)\n    partial_auc = auc(fpr, tpr)\n\n    #     # Equivalent code that uses sklearn's roc_auc_score\n    #     v_gt = abs(np.asarray(solution.values)-1)\n    #     v_pred = np.array([1.0 - x for x in submission.values])\n    #     max_fpr = abs(1-min_tpr)\n    #     partial_auc_scaled = roc_auc_score(v_gt, v_pred, max_fpr=max_fpr)\n    #     # change scale from [0.5, 1.0] to [0.5 * max_fpr**2, max_fpr]\n    #     # https://math.stackexchange.com/questions/914823/shift-numbers-into-a-different-range\n    #     partial_auc = 0.5 * max_fpr**2 + (max_fpr - 0.5 * max_fpr**2) / (1.0 - 0.5) * (partial_auc_scaled - 0.5)\n    \n    return(partial_auc)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:32:20.691196Z","iopub.execute_input":"2024-07-07T22:32:20.693290Z","iopub.status.idle":"2024-07-07T22:32:20.724318Z","shell.execute_reply.started":"2024-07-07T22:32:20.693201Z","shell.execute_reply":"2024-07-07T22:32:20.720794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">4.1 <b>GENERIC</b> FUNCTIONS</h3>\n<hr><br>\n\nSome functions I bring with me...","metadata":{}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists into a single list.\n\n    Args:\n        nested_list (Iterable): \n            – A list of lists (or iterables) to be flattened.\n\n    Returns:\n        A flattened list containing all items from the input list of lists.\n    \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\n\ndef print_ln(symbol=\"-\", line_len=110, newline_before=False, newline_after=False):\n    \"\"\" Print a horizontal line of a specified length and symbol.\n\n    Args:\n        symbol (str, optional): \n            – The symbol to use for the horizontal line\n        line_len (int, optional): \n            – The length of the horizontal line in characters\n        newline_before (bool, optional): \n            – Whether to print a newline character before the line\n        newline_after (bool, optional): \n            – Whether to print a newline character after the line\n            \n    Returns:\n        None; A divider with pre/post new-lines (optional) is printed\n    \"\"\"\n    if newline_before: print();\n    print(symbol * line_len)\n    if newline_after: print();\n        \n        \ndef display_hr(newline_before=False, newline_after=False):\n    \"\"\" Renders a HTML <hr>\n\n    Args:\n        newline_before (bool, optional): \n            – Whether to print a newline character before the line\n        newline_after (bool, optional): \n            – Whether to print a newline character after the line\n            \n    Returns:\n        None; A divider with pre/post new-lines (optional) is printed\n    \"\"\"\n    if newline_before: print();\n    display(HTML(\"<hr>\"))\n    if newline_after: print();\n\n\ndef wrap_text(text, width=88):\n    \"\"\"Wrap text to a specified width.\n\n    Args:\n        text (str): \n            - The text to wrap.\n        width (int): \n            - The maximum width of a line. Default is 88.\n\n    Returns:\n        str: The wrapped text.\n    \"\"\"\n    return textwrap.fill(text, width)\n\n\ndef wrap_text_by_paragraphs(text, width=88):\n    \"\"\"Wrap text by paragraphs to a specified width.\n\n    Args:\n        text (str): \n            - The text containing multiple paragraphs to wrap.\n        width (int): \n            - The maximum width of a line. Default is 88.\n\n    Returns:\n        str: The wrapped text with preserved paragraph separation.\n    \"\"\"\n    paragraphs = text.split('\\n')  # Assuming paragraphs are separated by newlines\n    wrapped_paragraphs = [textwrap.fill(paragraph, width) for paragraph in paragraphs]\n    return '\\n\\n'.join(wrapped_paragraphs)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:32:20.741832Z","iopub.execute_input":"2024-07-07T22:32:20.742991Z","iopub.status.idle":"2024-07-07T22:32:20.756656Z","shell.execute_reply.started":"2024-07-07T22:32:20.742931Z","shell.execute_reply":"2024-07-07T22:32:20.755310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">4.2 <b>COMPETITION</b> HELPERS</h3>\n<hr><br>\n\n","metadata":{}},{"cell_type":"code","source":"def load_img_from_hdf5(\n    isic_id: str, \n    file_path: str = \"/kaggle/input/isic-2024-challenge/train-image.hdf5\", \n    n_channels: int = 3\n):\n    \"\"\"\n    Load an image from the HDF5 dataset file by specifying an ISIC ID.\n    \n    The ISIC ID is expected to be in the form 'ISIC_#######'.\n    \n    Args:\n        isic_id (str): The ID of the image to load.\n        file_path (str): The path to the HDF5 file.\n        n_channels (int): Number of channels (3 for RGB, 1 for grayscale).\n    \n    Returns:\n        np.ndarray: The loaded image.\n    \n    Raises:\n        KeyError: If the ISIC ID is not found in the HDF5 file.\n        ValueError: If the image data cannot be decoded.\n    \n    Example Usage:\n        img = load_img_from_hdf5('ISIC_0000000')\n    \"\"\"\n    \n    # Handle the case where the isic_id is passed incorrectly\n    if not isic_id.lower().startswith(\"isic\"):\n        isic_id = f\"ISIC_{int(str(isic_id).split('_', 1)[-1]):>07}\"\n        \n    # Open the HDF5 file in read mode\n    with h5py.File(file_path, 'r') as hf:\n        \n        # Retrieve the image data from the HDF5 dataset using the provided ISIC ID\n        try:\n            image_data = hf[isic_id][()]\n        except KeyError:\n            raise KeyError(f\"ISIC ID {isic_id} not found in HDF5 file.\")\n\n        # Convert the binary data to a numpy array\n        image_array = np.frombuffer(image_data, np.uint8)\n\n        # Decode the image from the numpy array\n        if n_channels == 3:\n            # Load the image as a color image (BGR) and convert to RGB\n            image = cv2.cvtColor(cv2.imdecode(image_array, cv2.IMREAD_COLOR), cv2.COLOR_BGR2RGB)\n        else:\n            # Load the image as a grayscale image\n            image = cv2.imdecode(image_array, cv2.IMREAD_GRAYSCALE)\n\n        # If the image failed to load for some reason (problems decoding) ...\n        if image is None:\n            raise ValueError(f\"Could not decode image for ISIC ID: {isic_id}\")\n        \n        return image\n    \nplt.figure(figsize=(6,6))\nplt.title(\"ISIC_0015670\", fontweight=\"bold\")\nplt.imshow(load_img_from_hdf5(\"ISIC_0015670\"))\nplt.show()\n\nMETADATA_COL2DESC = {\n    \"isic_id\": \"Unique identifier for each image case.\",\n    \"target\": \"Binary class label indicating if the lesion is benign (0) or malignant (1).\",\n    \"patient_id\": \"Unique identifier for each patient.\",\n    \"age_approx\": \"Approximate age of the patient at the time of imaging.\",\n    \"sex\": \"Sex of the patient (male or female).\",\n    \"anatom_site_general\": \"General location of the lesion on the patient's body (e.g., upper extremity, posterior torso).\",\n    \"clin_size_long_diam_mm\": \"Maximum diameter of the lesion in millimeters.\",\n    \"image_type\": \"Type of image captured, as defined in the ISIC Archive.\",\n    \"tbp_tile_type\": \"Lighting modality of the 3D Total Body Photography (TBP) source image.\",\n    \"tbp_lv_A\": \"Color channel A inside the lesion; related to the green-red axis in LAB color space.\",\n    \"tbp_lv_Aext\": \"Color channel A outside the lesion; related to the green-red axis in LAB color space.\",\n    \"tbp_lv_B\": \"Color channel B inside the lesion; related to the blue-yellow axis in LAB color space.\",\n    \"tbp_lv_Bext\": \"Color channel B outside the lesion; related to the blue-yellow axis in LAB color space.\",\n    \"tbp_lv_C\": \"Chroma value inside the lesion, indicating color purity.\",\n    \"tbp_lv_Cext\": \"Chroma value outside the lesion, indicating color purity.\",\n    \"tbp_lv_H\": \"Hue value inside the lesion, representing the type of color (e.g., red, brown) in LAB color space.\",\n    \"tbp_lv_Hext\": \"Hue value outside the lesion, representing the type of color (e.g., red, brown) in LAB color space.\",\n    \"tbp_lv_L\": \"Luminance value inside the lesion; related to lightness in LAB color space.\",\n    \"tbp_lv_Lext\": \"Luminance value outside the lesion; related to lightness in LAB color space.\",\n    \"tbp_lv_areaMM2\": \"Area of the lesion in square millimeters.\",\n    \"tbp_lv_area_perim_ratio\": \"Ratio of the lesion's perimeter to its area, indicating border jaggedness.\",\n    \"tbp_lv_color_std_mean\": \"Mean color irregularity within the lesion, calculated as the variance of colors.\",\n    \"tbp_lv_deltaA\": \"Average contrast in color channel A between inside and outside the lesion.\",\n    \"tbp_lv_deltaB\": \"Average contrast in color channel B between inside and outside the lesion.\",\n    \"tbp_lv_deltaL\": \"Average contrast in luminance between inside and outside the lesion.\",\n    \"tbp_lv_deltaLB\": \"Combined contrast between the lesion and its immediate surrounding skin.\",\n    \"tbp_lv_deltaLBnorm\": \"Normalized contrast between the lesion and its immediate surrounding skin in LAB color space.\",\n    \"tbp_lv_eccentricity\": \"Eccentricity of the lesion, indicating how elongated it is.\",\n    \"tbp_lv_location\": \"Detailed anatomical location of the lesion, dividing body parts further (e.g., Left Arm - Upper).\",\n    \"tbp_lv_location_simple\": \"Simplified anatomical location of the lesion (e.g., Left Arm).\",\n    \"tbp_lv_minorAxisMM\": \"Smallest diameter of the lesion in millimeters.\",\n    \"tbp_lv_nevi_confidence\": \"Confidence score (0-100) from a neural network classifier estimating the probability that the lesion is a nevus.\",\n    \"tbp_lv_norm_border\": \"Normalized border irregularity score on a scale of 0-10.\",\n    \"tbp_lv_norm_color\": \"Normalized color variation score on a scale of 0-10.\",\n    \"tbp_lv_perimeterMM\": \"Perimeter of the lesion in millimeters.\",\n    \"tbp_lv_radial_color_std_max\": \"Color asymmetry score within the lesion, based on color variance in concentric rings.\",\n    \"tbp_lv_stdL\": \"Standard deviation of luminance within the lesion.\",\n    \"tbp_lv_stdLExt\": \"Standard deviation of luminance outside the lesion.\",\n    \"tbp_lv_symm_2axis\": \"Measure of asymmetry of the lesion's border about a secondary axis.\",\n    \"tbp_lv_symm_2axis_angle\": \"Angle of the secondary axis of symmetry for the lesion's border.\",\n    \"tbp_lv_x\": \"X-coordinate of the lesion in the 3D TBP model.\",\n    \"tbp_lv_y\": \"Y-coordinate of the lesion in the 3D TBP model.\",\n    \"tbp_lv_z\": \"Z-coordinate of the lesion in the 3D TBP model.\",\n    \"attribution\": \"Source or institution responsible for the image.\",\n    \"copyright_license\": \"Type of copyright license for the image.\",\n    \"lesion_id\": \"Unique identifier for lesions that were manually tagged as lesions of interest.\",\n    \"iddx_full\": \"Full classified diagnosis of the lesion.\",\n    \"iddx_1\": \"First-level diagnosis of the lesion (e.g., Benign, Malignant).\",\n    \"iddx_2\": \"Second-level diagnosis providing more specific details about the lesion.\",\n    \"iddx_3\": \"Third-level diagnosis with further classification details.\",\n    \"iddx_4\": \"Fourth-level diagnosis with additional specificity.\",\n    \"iddx_5\": \"Fifth-level diagnosis, providing the most detailed classification.\",\n    \"mel_mitotic_index\": \"Mitotic index of invasive malignant melanomas, indicating cell division rate.\",\n    \"mel_thick_mm\": \"Thickness in millimeters of melanoma invasion.\",\n    \"tbp_lv_dnn_lesion_confidence\": \"Lesion confidence score (0-100) from a deep neural network classifier.\"\n}\n\n\nMETADATA_COL2NAME = {\n    \"isic_id\": \"Unique Case Identifier\",\n    \"target\": \"Binary Lession Classification\",\n    \"patient_id\": \"Unique Patient Identifier\",\n    \"age_approx\": \"Approximate Age\",\n    \"sex\": \"Sex\",\n    \"anatom_site_general\": \"General Anatomical Location\",\n    \"clin_size_long_diam_mm\": \"Clinical Size (Longest Diameter in mm)\",\n    \"image_type\": \"Image Type\",\n    \"tbp_tile_type\": \"TBP Tile Type\",\n    \"tbp_lv_A\": \"Color Channel A Inside Lesion\",\n    \"tbp_lv_Aext\": \"Color Channel A Outside Lesion\",\n    \"tbp_lv_B\": \"Color Channel B Inside Lesion\",\n    \"tbp_lv_Bext\": \"Color Channel B Outside Lesion\",\n    \"tbp_lv_C\": \"Chroma Inside Lesion\",\n    \"tbp_lv_Cext\": \"Chroma Outside Lesion\",\n    \"tbp_lv_H\": \"Hue Inside Lesion\",\n    \"tbp_lv_Hext\": \"Hue Outside Lesion\",\n    \"tbp_lv_L\": \"Luminance Inside Lesion\",\n    \"tbp_lv_Lext\": \"Luminance Outside Lesion\",\n    \"tbp_lv_areaMM2\": \"Lesion Area (mm²)\",\n    \"tbp_lv_area_perim_ratio\": \"Area-to-Perimeter Ratio\",\n    \"tbp_lv_color_std_mean\": \"Mean Color Irregularity\",\n    \"tbp_lv_deltaA\": \"Delta A (Inside vs. Outside)\",\n    \"tbp_lv_deltaB\": \"Delta B (Inside vs. Outside)\",\n    \"tbp_lv_deltaL\": \"Delta L (Inside vs. Outside)\",\n    \"tbp_lv_deltaLB\": \"Delta LB (Contrast)\",\n    \"tbp_lv_deltaLBnorm\": \"Normalized Delta LB (Contrast)\",\n    \"tbp_lv_eccentricity\": \"Eccentricity\",\n    \"tbp_lv_location\": \"Detailed Anatomical Location\",\n    \"tbp_lv_location_simple\": \"Simplified Anatomical Location\",\n    \"tbp_lv_minorAxisMM\": \"Smallest Diameter (mm)\",\n    \"tbp_lv_nevi_confidence\": \"Nevus Confidence Score\",\n    \"tbp_lv_norm_border\": \"Normalized Border Irregularity\",\n    \"tbp_lv_norm_color\": \"Normalized Color Variation\",\n    \"tbp_lv_perimeterMM\": \"Lesion Perimeter (mm)\",\n    \"tbp_lv_radial_color_std_max\": \"Radial Color Standard Deviation\",\n    \"tbp_lv_stdL\": \"Standard Deviation of Luminance (Inside)\",\n    \"tbp_lv_stdLExt\": \"Standard Deviation of Luminance (Outside)\",\n    \"tbp_lv_symm_2axis\": \"Symmetry (Second Axis)\",\n    \"tbp_lv_symm_2axis_angle\": \"Symmetry Angle (Second Axis)\",\n    \"tbp_lv_x\": \"X-Coordinate\",\n    \"tbp_lv_y\": \"Y-Coordinate\",\n    \"tbp_lv_z\": \"Z-Coordinate\",\n    \"attribution\": \"Image Source\",\n    \"copyright_license\": \"Copyright License\",\n    \"lesion_id\": \"Unique Lesion Identifier\",\n    \"iddx_full\": \"Full Diagnosis\",\n    \"iddx_1\": \"First Level Diagnosis\",\n    \"iddx_2\": \"Second Level Diagnosis\",\n    \"iddx_3\": \"Third Level Diagnosis\",\n    \"iddx_4\": \"Fourth Level Diagnosis\",\n    \"iddx_5\": \"Fifth Level Diagnosis\",\n    \"mel_mitotic_index\": \"Mitotic Index (Melanoma)\",\n    \"mel_thick_mm\": \"Thickness of Melanoma (mm)\",\n    \"tbp_lv_dnn_lesion_confidence\": \"Lesion Confidence Score\"\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:32:20.758825Z","iopub.execute_input":"2024-07-07T22:32:20.759472Z","iopub.status.idle":"2024-07-07T22:32:21.255735Z","shell.execute_reply.started":"2024-07-07T22:32:20.759428Z","shell.execute_reply":"2024-07-07T22:32:21.254277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">4.3 DATASET <b>LOADING</b></h3>\n<hr><br>\n\nLet's load the dataset our own way and improve upon the previously shown code","metadata":{}},{"cell_type":"code","source":"# ROOT PATHS\nWORKING_DIR = \"/kaggle/working\"\nINPUT_DIR = \"/kaggle/input\"\nCOMPETITION_DIR = os.path.join(INPUT_DIR, \"isic-2024-challenge\")\n\n# IMAGE DIRS\nTRAIN_IMAGE_DIR = os.path.join(COMPETITION_DIR, \"train-image\", \"image\")\nTEST_IMAGE_DIR = os.path.join(COMPETITION_DIR, \"test-image\", \"image\")\n\n# FILE PATHS\nTRAIN_METADATA_CSV = os.path.join(COMPETITION_DIR, \"train-metadata.csv\")\nTEST_METADATA_CSV = os.path.join(COMPETITION_DIR, \"test-metadata.csv\")\nTRAIN_IMAGE_HDF5 = os.path.join(COMPETITION_DIR, \"train-image.hdf5\")\nTEST_IMAGE_HDF5 = os.path.join(COMPETITION_DIR, \"test-image.hdf5\")\nSS_CSV_PATH = os.path.join(COMPETITION_DIR, \"sample_submission.csv\")\n\n\n# DEFINE COMPETITION DATAFRAMES\nclr_print(\"\\n\\n... SAMPLE SUBMISSION DATAFRAME ...\\n\\n\")\nss_df = pd.read_csv(SS_CSV_PATH)\ndisplay(ss_df)\n\nclr_print(\"\\n\\n... TRAIN METADATA DATAFRAME ...\\n\\n\")\ntrain_df = pd.read_csv(TRAIN_METADATA_CSV)\ndisplay(train_df)\n\nclr_print(\"\\n\\n... TEST METADATA DATAFRAME ...\\n\\n\")\ntest_df = pd.read_csv(TEST_METADATA_CSV)\ndisplay(test_df)\n\nclr_print(\"\\n\\n... HDF5 (DATASET) PATHS ...\\n\\n\")\nprint(f\"\\t--> {TRAIN_IMAGE_HDF5}\")\nprint(f\"\\t--> {TEST_IMAGE_HDF5}\\n\")\n\nfor _c in train_df.columns:\n    display_hr(True, True)\n    clr_print(f\"COLUMN NAME         : <code>'{_c}'</code>\")\n    clr_print(f\"HUMAN READABLE NAME : <span style='color: black !important;'>'{METADATA_COL2NAME.get(_c)}'</span>\")\n    clr_print(f\"COLUMN DESCRIPTION  : <span style='color: black !important;'>'{METADATA_COL2DESC.get(_c)}'</span>\")\ndisplay_hr(True, True)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:32:21.257511Z","iopub.execute_input":"2024-07-07T22:32:21.257952Z","iopub.status.idle":"2024-07-07T22:32:31.055846Z","shell.execute_reply.started":"2024-07-07T22:32:21.257915Z","shell.execute_reply":"2024-07-07T22:32:31.054694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"eda\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #41C0BA;\" id=\"eda\">5&nbsp;&nbsp;EXPLORATORY DATA ANALYSIS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #375B6D;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n\n<table class=\"table is-hoverable is-bordered\"> <thead> <tr> <th><abbr title=\"Dataset\">Dataset</abbr></th> <th><abbr title=\"Training Images and Input Data\">Training Images and Input Attributes</abbr></th> <th><abbr title=\"Training Supplemental Metadata\">Training Supplement</abbr></th> <th><abbr title=\"Training Ground Truth Data\">Training Ground Truth</abbr></th> <th><abbr title=\"Test Data\">Test Data</abbr></th> <th><abbr title=\"Test Ground Truth Data\">Test Ground Truth</abbr></th> <th><abbr title=\"License\">License</abbr></th> </tr> </thead> <tbody> <tr> <td>SLICE-3D<br><br><b><i>THIS DATASET</i></b></td> <td> <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Training_Input.zip\"> Download (1.2GB) </a> <br> 401,059 JPEG images of skin lesion image crops extracted from 3D TBP with metadata entries of age, sex, general anatomic site, common patient identifier, clinical size, and various data fields from the TBP Lesion Visualizer*. </td> <td> <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Training_Supplement.csv\">Download Supplemental Metadata (40MB)</a> <br> 401,059 metadata entries of attributes which may be useful for training cross-validation. </td> <td> <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Training_GroundTruth.csv\"> Download (7MB) </a> <br> 401,059 entries of gold standard lesion malignancy values. </td> <td rowspan=\"1\"> <!--- <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Test_Input.zip\"> Download (1.5GB) </a> <br> 511,474 JPEG images of skin lesion image crops extracted from 3D TBP with metadata entries of age, sex, general anatomic site, common patient identifier, clinical size, and various data fields from the TBP Lesion Visualizer*. ---> Not Available </td> <td> <div> <span> Not Available </span> </div> </td> <td> <a href=\"https://creativecommons.org/licenses/by-nc/4.0/\"> CC-BY-NC </a> </td> </tr> <tr> <td>SLICE-3D Permissive</td> <td> <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Permissive_Training_Input.zip\"> Download (623MB) </a> <br> 217,477 JPEG images of skin lesion image crops extracted from 3D TBP with metadata entries of age, sex, general anatomic site, common patient identifier, clinical size, and various data from the TBP Lesion Visualizer*. </td> <td> <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Permissive_Training_Supplement.csv\">Download Supplemental Metadata (21MB)</a> <br> 217,477 metadata entries of attributes which may be useful for training cross-validation. </td> <td> <a href=\"https://isic-challenge-data.s3.amazonaws.com/2024/ISIC_2024_Permissive_Training_GroundTruth.csv\"> Download (4MB) </a> <br> 217,477 entries of gold standard lesion malignancy values. </td> <td> Not Available <br> </td> <td> <div> <span> Not Available </span> </div> </td> <td> <a href=\"https://creativecommons.org/licenses/by/4.0/\"> CC-BY </a> </td> </tr> </tbody> </table>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">5.1 DATASET <b>EXPLORATION</b></h3>\n<hr><br>\n\n","metadata":{}},{"cell_type":"code","source":"train_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:32:31.058645Z","iopub.execute_input":"2024-07-07T22:32:31.059009Z","iopub.status.idle":"2024-07-07T22:32:31.809080Z","shell.execute_reply.started":"2024-07-07T22:32:31.058979Z","shell.execute_reply":"2024-07-07T22:32:31.807895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_nan_heatmap(\n    df: pd.DataFrame, \n    figsize: tuple = (17, 8), \n    cmap: str = 'magma_r', \n    title: str = 'NaN Values in DataFrame',\n    x_tick_rotation=60,\n    show_cbar: bool = False, \n    show_yticklabels: bool = False\n) -> None:\n    \"\"\"Create a heatmap to visualize NaN values in a DataFrame.\n\n    Args:\n        df (pd.DataFrame): \n            The input DataFrame to visualize.\n        figsize (tuple[int], optional): \n            Figure size as a tuple of (width, height)\n        cmap (str, optional): \n            Colormap to use for the heatmap\n        title (str, optional): \n            Title for the heatmap.\n        x_tick_rotation (int, optional): \n            Rotation angle for x-axis tick labels.\n        show_cbar (bool, optional): \n            Whether to show the color bar.\n        show_yticklabels (bool, optional): \n            Whether to show y-axis tick labels.\n\n    Returns:\n        None; \n            The function displays the plot using plt.show().\n    \"\"\"\n    \n    # Setup the figure\n    plt.figure(figsize=figsize)\n    \n    # Create the heatmap\n    sns.heatmap(df.isna(), cbar=show_cbar, yticklabels=show_yticklabels, cmap=cmap)\n    \n    # Update the title/labels\n    plt.title(title, fontweight=\"bold\")\n    plt.xlabel('Columns', fontweight=\"bold\")\n    plt.ylabel('Rows', fontweight=\"bold\")\n    \n    \n    # Rotate x-axis labels\n    plt.xticks(rotation=x_tick_rotation, ha='right')\n    \n    # Adjust the bottom margin to prevent label cutoff\n    plt.tight_layout()\n    \n    # Render\n    plt.show()\n\n    # Print NaN counts per column\n    nan_counts = df.isna().sum().sort_values(ascending=False)\n    \n    # clr_print()\n    clr_print(\"NaN counts per column:\")\n    print(nan_counts[nan_counts])\n    clr_print(\"Features with 0 NaN values:\")\n    print(_nan[_nan==0].index.tolist())","metadata":{"execution":{"iopub.status.busy":"2024-07-07T22:36:19.685741Z","iopub.execute_input":"2024-07-07T22:36:19.686268Z","iopub.status.idle":"2024-07-07T22:36:19.698910Z","shell.execute_reply.started":"2024-07-07T22:36:19.686214Z","shell.execute_reply":"2024-07-07T22:36:19.697196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_target_distribution(df: pd.DataFrame, log_y: bool = True) -> None:\n    \"\"\"Plot the distribution of the target variable.\n\n    Args:\n        df (pd.DataFrame): \n            The input dataframe containing the target column.\n        log_y (bool, optional):\n            Whether to log the y-axis (helpful for visualizing large class imbalance)\n\n    Returns:\n        None; \n            This function doesn't return anything, it displays a plot.\n    \"\"\"\n    # Count the occurrences of each target value\n    target_counts = df['target'].value_counts().sort_index()\n    \n    # Calculate percentages\n    total = len(df)\n    percentages = [f\"{count/total:.3%}\" for count in target_counts]\n    \n    # Create the bar plot\n    fig = go.Figure(data=[\n        go.Bar(\n            x=LABELS,  # Assume we have access to this\n            y=target_counts,\n            text=percentages,\n            textposition='auto',\n            marker_color=COLORS  # Assume we have access to this\n        )\n    ])\n    \n    # Customize the layout\n    fig.update_layout(\n        title='<b>DISTRIBUTION OF BENIGN VS MALIGNANT LESIONS',\n        xaxis_title='<b>Lesion Classification</b>', yaxis_title=f'<b>Count {\"<sub>\"+\"(Log Scale)\"+\"</sub>\" if log_y else \"\"}</b>',\n        template='plotly_white', height=600, width=1200,\n    )\n    \n    if log_y:\n        fig.update_layout(yaxis=dict(type='log'))\n    \n    # Add annotation for total count\n    fig.add_annotation(\n        text=f\"<b>TOTAL SAMPLES:  {total:,}</b>\",\n        xref=\"paper\", yref=\"paper\",\n        x=0.98, y=1.05,\n        showarrow=False,\n        font=dict(size=12)\n    )\n    \n    # Show the plot\n    fig.show()\n\n# Call the function\nplot_target_distribution(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:07:52.680034Z","iopub.execute_input":"2024-07-06T01:07:52.680544Z","iopub.status.idle":"2024-07-06T01:07:54.487146Z","shell.execute_reply.started":"2024-07-06T01:07:52.680502Z","shell.execute_reply":"2024-07-06T01:07:54.485961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_target_distribution(\n    df: pd.DataFrame, \n    log_y: bool = True, \n    target_as_str: bool = True,\n    target_col: str = \"target\", \n    color_sequence: list[str] | None = None,\n    template_theme: str = \"plotly_white\"\n) -> None:\n    \"\"\"Plot the distribution of the target variable.\n\n    This function creates a histogram of the target variable distribution,\n    with options for log scale, string labels, and custom color schemes.\n\n    Args:\n        df (pd.DataFrame): \n            The input dataframe containing the target column.\n        log_y (bool, optional): \n            Whether to use log scale for y-axis.\n        target_as_str (bool, optional): \n            Whether to convert target labels to strings.\n        target_col (str, optional): \n            Name of the target column. \n        color_sequence (list[str], optional): \n            Custom color sequence for the bars. \n            Defaults to None, which uses the global COLORS.\n        template_theme (str, optional): \n            Plotly template theme for visuals styling.\n            The available templates are:\n                - 'ggplot2', 'seaborn', 'simple_white', \n                - 'plotly', 'plotly_white', 'plotly_dark', \n                - 'presentation', 'xgridoff', 'ygridoff', 'gridon', \n                - 'none'\n\n    Returns:\n        None; \n            This function displays a plot and doesn't return anything.\n    \"\"\"\n    \n    # Prevent accidental edits to the original dataframe\n    _df = df.copy()\n    \n    # Set default color sequence if not provided\n    if not color_sequence:\n        color_sequence = COLORS\n    \n    # Convert target labels to strings if requested\n    if target_as_str:\n        _df[target_col] = _df[target_col].map(LBL_MAP_I2S)\n    \n    # Create the histogram using Plotly Express\n    fig = px.histogram(\n        _df, x=target_col, color=target_col, \n        color_discrete_sequence=color_sequence, \n        log_y=log_y, height=500, width=1200, template=template_theme,\n        title='<b>DISTRIBUTION OF BENIGN VS MALIGNANT LESIONS',\n    )\n    \n    # Customize the layout\n    fig.update_layout(\n        bargap=0.1,  # Add space between bars\n        xaxis_title='<b>Lesion Classification</b>', \n        yaxis_title=f'<b>Count {\"<sub>(Log Scale)</sub>\" if log_y else \"\"}</b>',\n        showlegend=False  # Hide legend as color already differentiates categories\n    )\n    \n    # Apply log scale to y-axis if requested\n    if log_y:\n        fig.update_layout(yaxis=dict(type='log'))\n    \n    # Add annotation for total sample count\n    fig.add_annotation(\n        text=f\"<b>TOTAL SAMPLES: {len(_df):,}</b>\",\n        xref=\"paper\", yref=\"paper\",\n        x=0.98, y=1.05,\n        showarrow=False,\n        font=dict(size=12)\n    )\n    \n    # Display the plot\n    fig.show()\n\nplot_target_distribution(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:07:54.488954Z","iopub.execute_input":"2024-07-06T01:07:54.489713Z","iopub.status.idle":"2024-07-06T01:07:56.425649Z","shell.execute_reply.started":"2024-07-06T01:07:54.489669Z","shell.execute_reply":"2024-07-06T01:07:56.423603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_categorical_feature_distribution(\n    df: pd.DataFrame, \n    feature_col: str,\n    target_col: str = \"target\",\n    target_as_str: bool = True,\n    log_y: bool = False, \n    color_sequence: list[str] | None = None,\n    template_theme: str = \"plotly_white\",\n    group_by_target: bool = True,\n    stack_bars: bool = False\n) -> None:\n    \"\"\"Plot the distribution of a feature, optionally grouped by the target variable.\n\n    This function creates a histogram of the feature distribution,\n    with options for log scale, custom color schemes, and grouping by target.\n\n    Args:\n        df (pd.DataFrame): \n            The input dataframe containing feature and target columns.\n        feature_col (str): \n            Name of the feature column to plot.\n        target_col (str, optional): \n            Name of the target column.\n        target_as_str (bool, optional): \n            Whether to convert target labels to strings.\n        log_y (bool, optional): \n            Whether to use log scale for y-axis.\n        color_sequence (list[str], optional): \n            Custom color sequence for the bars.\n        template_theme (str, optional): \n            Plotly template theme for visual styling.\n            Available options include: \n                'ggplot2', 'seaborn', 'simple_white', 'plotly',\n                'plotly_white', 'plotly_dark', 'presentation', 'xgridoff', 'ygridoff',\n                'gridon', 'none'.\n        group_by_target (bool, optional): \n            Whether to group bars by target.\n        stack_bars (bool, optional): \n            Whether to stack bars when grouped.\n\n    Returns:\n        None: This function displays a plot and doesn't return anything.\n    \"\"\"\n    # Prevent accidental edits to the original dataframe\n    _df = df.copy().sort_values(by=[feature_col, target_col]).reset_index(drop=True)\n        \n    if target_as_str and group_by_target:\n        _df[target_col] = _df[target_col].map(LBL_MAP_I2S)\n        \n    # Set default color sequence if not provided\n    if not color_sequence:\n        color_sequence = list(nb_palette.as_hex())\n    \n    # Prepare the histogram data\n    if group_by_target:\n        fig = px.histogram(\n            _df, x=feature_col, color=target_col, \n            color_discrete_sequence=COLORS,  # Use target colors for grouping\n            log_y=log_y, height=500, width=1200, template=template_theme,\n            title=f'<b>DISTRIBUTION OF {feature_col.replace(\"_\", \" \").upper()} BY TARGET',\n            barmode='group' if not stack_bars else 'stack'\n        )\n        \n        # Add border to bars using the target colors\n        for i, trace in enumerate(fig.data):\n            trace.marker.line.color = COLORS[i]\n            trace.marker.line.width = 1.5\n    else:\n        fig = px.histogram(\n            _df, x=feature_col, color=feature_col, \n            color_discrete_sequence=color_sequence,\n            log_y=log_y, height=500, width=1200, template=template_theme,\n            title=f'<b>DISTRIBUTION OF {feature_col.replace(\"_\", \" \").upper()}',\n        )\n    \n    # Customize the layout\n    fig.update_layout(\n        bargap=0.1,  # Add space between bars\n        xaxis_title=f'<b>{feature_col.replace(\"_\", \" \").title()}</b>', \n        yaxis_title=f'<b>Count {\"<sub>(Log Scale)</sub>\" if log_y else \"\"}</b>',\n        showlegend=group_by_target  # Show legend only when grouped by target\n    )\n    \n    # Apply log scale to y-axis if requested\n    if log_y:\n        fig.update_layout(yaxis_type='log')\n    \n    # Display the plot\n    fig.show()\n\nplot_categorical_feature_distribution(train_df, 'anatom_site_general', group_by_target=True, stack_bars=False, log_y=True)\n# plot_categorical_feature_distribution(train_df, 'anatom_site_general', group_by_target=False)\n\nplot_categorical_feature_distribution(train_df, \"sex\", group_by_target=True, stack_bars=False, log_y=True)\n# plot_categorical_feature_distribution(train_df, \"sex\", group_by_target=False)\n\n# plot_categorical_feature_distribution(train_df, \"age_approx\", group_by_target=True, stack_bars=False, log_y=True)\nplot_categorical_feature_distribution(train_df, \"age_approx\", group_by_target=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:07:56.428165Z","iopub.execute_input":"2024-07-06T01:07:56.429348Z","iopub.status.idle":"2024-07-06T01:08:00.508819Z","shell.execute_reply.started":"2024-07-06T01:07:56.429307Z","shell.execute_reply":"2024-07-06T01:08:00.505349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_continuous_feature_distribution(\n    df: pd.DataFrame, \n    feature_col: str,\n    plot_style: str = \"histogram\",\n    feature_readable_name: str | None = None,\n    target_col: str = \"target\",\n    target_as_str: bool = True,\n    log_y: bool = False, \n    color_sequence: list[str] | None = None,\n    template_theme: str = \"plotly_white\",\n    group_by_target: bool = True,\n    n_bins: int = 50\n) -> None:\n    \"\"\"Plot the distribution of a continuous feature, optionally grouped by the target variable.\n    \n    This function creates either a histogram or a box plot of the continuous feature distribution,\n    with options for log scale, custom color schemes, and grouping by target.\n    \n    Args:\n        df (pd.DataFrame): \n            The input dataframe containing feature and target columns.\n        feature_col (str): \n            Name of the continuous feature column to plot.\n        plot_style (str, optional):\n            Type of plot to create. Either \"histogram\" or \"box\".\n        feature_readable_name (Optional[str]):\n            An option to replace the column name with a readable name for title/axis.\n        target_col (str, optional): \n            Name of the target column.\n        target_as_str (bool, optional): \n            Whether to convert target labels to strings.\n        log_y (bool, optional): \n            Whether to use log scale for y-axis.\n        color_sequence (Optional[list[str]], optional): \n            Custom color sequence for the plots. \n            Defaults to None, which uses COLORS for target grouping.\n        template_theme (str, optional): \n            Plotly template theme for visual styling.\n        group_by_target (bool, optional): \n            Whether to group the plot by target.\n        n_bins (int, optional):\n            Number of bins for the histogram (only used if plot_style is \"histogram\").\n    Returns:\n        None; \n            This function displays a plot and doesn't return anything.\n    \"\"\"\n    # Prevent accidental edits to the original dataframe\n    _df = df.copy().sort_values(by=[feature_col, target_col]).reset_index(drop=True)\n        \n    if target_as_str:\n        _df[target_col] = _df[target_col].map(LBL_MAP_I2S)\n        \n    # Set default color sequence if not provided\n    if not color_sequence:\n        if group_by_target:\n            color_sequence = COLORS\n        else:\n            color_sequence = list(nb_palette.as_hex())\n    \n    if feature_readable_name is None:\n        feature_readable_name = METADATA_COL2NAME.get(feature_col, feature_col.replace(\"_\", \" \").title())   \n        \n    if plot_style == \"histogram\":\n        if group_by_target:\n            fig = go.Figure()\n            for i, target_value in enumerate(_df[target_col].unique()):\n                subset = _df[_df[target_col] == target_value]\n                fig.add_trace(go.Histogram(\n                    x=subset[feature_col],\n                    name=str(target_value),\n                    marker_color=color_sequence[i % len(color_sequence)],\n                    opacity=0.7,\n                    nbinsx=n_bins\n                ))\n            \n            fig.update_layout(\n                barmode='overlay',\n                title=f\"<b>DISTRIBUTION OF '{feature_readable_name.upper()}' BY TARGET</b>\",\n                height=500, width=1200, template=template_theme\n            )\n        else:\n            fig = px.histogram(\n                _df, x=feature_col,\n                color_discrete_sequence=[color_sequence[0]],\n                log_y=log_y, height=500, width=1200, template=template_theme,\n                title=f\"<b>DISTRIBUTION OF '{feature_readable_name.upper()}'</b>\",\n                nbins=n_bins\n            )\n        \n        # Customize the layout\n        fig.update_layout(\n            xaxis_title=f'<b>{feature_readable_name}</b>', \n            yaxis_title=f'<b>Count {\"<sub>(Log Scale)</sub>\" if log_y else \"\"}</b>',\n            showlegend=group_by_target  # Show legend only when grouped by target\n        )\n        \n    elif plot_style == \"box\":\n        if group_by_target:\n            fig = go.Figure()\n            for i, target_value in enumerate(_df[target_col].unique()):\n                subset = _df[_df[target_col] == target_value]\n                fig.add_trace(go.Box(\n                    y=subset[feature_col],\n                    name=str(target_value),\n                    marker_color=color_sequence[i % len(color_sequence)],\n                    boxpoints='outliers',\n                    boxmean=True\n                ))\n            \n            fig.update_layout(\n                title=f\"<b>DISTRIBUTION OF '{feature_readable_name.upper()}' BY TARGET <sub>(includes likely outliers)</sub></b>\",\n                height=500, width=1200, template=template_theme\n            )\n        else:\n            fig = px.box(\n                _df, y=feature_col,\n                color_discrete_sequence=color_sequence,\n                height=500, width=1200, template=template_theme,\n                title=f\"<b>DISTRIBUTION OF '{feature_readable_name.upper()}'</b>\",\n                points='outliers',\n            )\n        \n        # Customize the layout\n        fig.update_layout(\n            xaxis_title='<b>Target</b>' if group_by_target else '', \n            yaxis_title=f'<b>{feature_readable_name} {\"<sub>(Log Scale)</sub>\" if log_y else \"\"}</b>',\n            showlegend=group_by_target  # Show legend only when grouped by target\n        )\n    \n    else:\n        raise ValueError(\"Invalid plot_style. Choose either 'histogram' or 'box'.\")\n    \n    # Apply log scale to y-axis if requested (only for histogram)\n    if log_y:\n        fig.update_layout(yaxis_type='log')\n    \n    # Display the plot\n    fig.show()\n    \n    \n# Lesion Area\nplot_continuous_feature_distribution(train_df, 'tbp_lv_areaMM2', plot_style=\"histogram\", log_y=True, group_by_target=True, n_bins=100)\n\n# Lesion Perimeter\nplot_continuous_feature_distribution(train_df, 'tbp_lv_perimeterMM', plot_style=\"box\", log_y=True, group_by_target=True, n_bins=100)\n\n# Lesion Diameter\nplot_continuous_feature_distribution(train_df, 'clin_size_long_diam_mm', plot_style=\"box\", log_y=True, group_by_target=True)\n\n# # Border Irregularity\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_norm_border', plot_style=\"histogram\", log_y=True, group_by_target=True, n_bins=100)\n\n# # Lesion Asymmetry\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_symm_2axis', plot_style=\"histogram\", log_y=True, group_by_target=True, n_bins=100)\n\n# # Color Contrast (Delta LB)\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_deltaLBnorm', log_y=True, group_by_target=True, n_bins=100)\n\n# # Color Variation\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_norm_color, log_y=True, group_by_target=True, n_bins=100)\n\n\n# # Nevus Confidence Score\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_nevi_confidence', log_y=True, group_by_target=True, n_bins=100)\n\n# # Hue Inside Lesion\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_H', log_y=True, group_by_target=True, n_bins=100)\n\n# # Luminance Inside Lesion\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_L', log_y=True, group_by_target=True, n_bins=100)\n\n# # Color Standard Deviation Mean\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_color_std_mean', log_y=True, group_by_target=True, n_bins=100)\n\n# # Lesion Eccentricity\n# plot_continuous_feature_distribution(train_df, 'tbp_lv_eccentricity', log_y=True, group_by_target=True, n_bins=100)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:05.936799Z","iopub.execute_input":"2024-07-06T01:08:05.937203Z","iopub.status.idle":"2024-07-06T01:08:13.427571Z","shell.execute_reply.started":"2024-07-06T01:08:05.937172Z","shell.execute_reply":"2024-07-06T01:08:13.426100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_scatter(\n    df: pd.DataFrame,\n    x_col: str,\n    y_col: str,\n    subset_percentage: float = 0.1,\n    color_col: str = \"target\",\n    target_as_str: bool = True,\n    size_col: str | None = None,\n    x_label: str | None = None,\n    y_label: str | None = None,\n    title: str | None = None,\n    color_sequence: list[str] | None = None,\n    template_theme: str = \"plotly_white\",\n    log_x: bool = False,\n    log_y: bool = False,\n    hover_data: list[str] | None = None\n) -> None:\n    \"\"\"Scatter plot to illustrate how two variables impact each other (and others)\n\n    Args:\n        df (pd.DataFrame): The input dataframe containing the data to plot.\n        x_col (str): Name of the column to use for x-axis.\n        y_col (str): Name of the column to use for y-axis.\n        subset_percentage (float, optional): What percentage of the dataset to examine (helps with render)\n        color_col (str, optional): Name of the column to use for color coding points.\n        target_as_str (bool, optional): If the target column is given as a column, turn into str?\n        size_col (str, optional): Name of the column to use for sizing points.\n        x_label (str, optional): Custom label for x-axis. If None, uses x_col.\n        y_label (str, optional): Custom label for y-axis. If None, uses y_col.\n        title (str, optional): Title of the plot. If None, generates a default title.\n        color_sequence (list[str], optional): Custom color sequence for color coding.\n        template_theme (str, optional): Plotly template theme for visual styling.\n        log_x (bool, optional): Whether to use log scale for x-axis.\n        log_y (bool, optional): Whether to use log scale for y-axis.\n        hover_data (list[str], optional): Additional columns to show in hover data.\n\n    Returns:\n        None: This function displays a plot and doesn't return anything.\n    \"\"\"\n    # Prevent accidental edits to the original dataframe\n    _df = df.copy().sort_values(by=[x_col, y_col, color_col]).reset_index(drop=True).sample(frac=subset_percentage)\n        \n    if target_as_str:\n        _df[\"target\"] = _df[\"target\"].map(LBL_MAP_I2S)\n    \n    for _col in x_col, y_col, color_col, size_col:\n        if _col and _df[_col].dtype==\"float64\":\n            _df[_col] = _df[_col].fillna(_df[_col].mean())\n    \n    # Set default color sequence if not provided\n    if not color_sequence:\n        if color_col==\"target\":\n            color_sequence = COLORS\n        else:\n            color_sequence = list(nb_palette.as_hex())\n        \n    # Set up the scatter plot\n    fig = px.scatter(\n        _df,\n        x=x_col,\n        y=y_col,\n        color=color_col,\n        size=size_col,\n        color_discrete_sequence=color_sequence,\n        template=template_theme,\n        log_x=log_x,\n        log_y=log_y,\n        hover_data=hover_data\n    )\n    \n    x_col_str = x_label or METADATA_COL2NAME.get(x_col, x_col.replace('_', ' ').title())\n    y_col_str = y_label or METADATA_COL2NAME.get(y_col, y_col.replace('_', ' ').title())\n    \n    \n    # Customize the layout\n    fig.update_layout(\n        title=title or f\"<b>Scatter Plot: '{y_col_str}' vs '{x_col_str}'</b>\",\n        xaxis_title=f\"<b>{x_col_str}</b>\",\n        yaxis_title=f\"<b>{y_col_str}</b>\",\n        height=600,\n        width=1000\n    )\n\n    # Update axes to show log scale in label if applied\n    if log_x:\n        fig.update_xaxes(title=f\"<b>{x_col_str} <sub>(Log Scale)</sub></b>\")\n    if log_y:\n        fig.update_yaxes(title=f\"<b>{y_col_str} <sub>(Log Scale)</sub></b>\")\n\n    # Display the plot\n    fig.show()\n    \nplot_scatter(train_df, \"tbp_lv_area_perim_ratio\", \"clin_size_long_diam_mm\", color_col=\"anatom_site_general\", size_col=\"tbp_lv_areaMM2\", subset_percentage=0.001)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:13.429501Z","iopub.execute_input":"2024-07-06T01:08:13.429872Z","iopub.status.idle":"2024-07-06T01:08:14.817623Z","shell.execute_reply.started":"2024-07-06T01:08:13.429840Z","shell.execute_reply":"2024-07-06T01:08:14.816520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation** and **Variable Understanding**\n\n---\n","metadata":{}},{"cell_type":"code","source":"# Categorical columns\nCAT_COLS = [\n    # male      265546\n    # female    123996\n    # NaN        11517\n    'sex',  \n    \n    # posterior torso    121902\n    # lower extremity    103028\n    # anterior torso      87770\n    # upper extremity     70557\n    # head/neck           12046\n    # NaN                  5756\n    'anatom_site_general', \n    \n    # TBP tile: close-up    401059\n    'image_type',\n    \n    # 3D: XP       285903\n    # 3D: white    115156\n    'tbp_tile_type', \n\n    # Torso Back Top Third       71112\n    # Torso Front Top Half       63350\n    # Torso Back Middle Third    46185\n    # Left Leg - Lower           27428\n    # Right Leg - Lower          25208\n    # Torso Front Bottom Half    24360\n    # Left Leg - Upper           23673\n    # Right Leg - Upper          23034\n    # Right Arm - Upper          22972\n    # Left Arm - Upper           22816\n    # Head & Neck                12046\n    # Left Arm - Lower           11939\n    # Right Arm - Lower          10636\n    # Unknown                     5756\n    # Torso Back Bottom Third     4596\n    # Left Leg                    1974\n    # Right Leg                   1711\n    # Left Arm                    1593\n    # Right Arm                    601\n    # Torso Front                   60\n    # Torso Back                     9\n    'tbp_lv_location', \n    \n    # Torso Back     121902\n    # Torso Front     87770\n    # Left Leg        53075\n    # Right Leg       49953\n    # Left Arm        36348\n    # Right Arm       34209\n    # Head & Neck     12046\n    # Unknown          5756\n    'tbp_lv_location_simple', \n    \n    # Memorial Sloan Kettering Cancer Center                                             129068\n    # Department of Dermatology, Hospital Clínic de Barcelona                            105724\n    # University Hospital of Basel                                                       65218\n    # Frazer Institute, The University of Queensland, Dermatology Research Centre        51768\n    # ACEMID MIA                                                                         28665\n    # ViDIR Group, Department of Dermatology, Medical University of Vienna               12640\n    # Department of Dermatology, University of Athens, Andreas Syggros Hospital [...]    7976\n    'attribution',\n    \n    # CC-BY       188812\n    # CC-BY-NC    183582\n    # CC-0         28665\n    'copyright_license', \n    \n    # NaN        401006\n    # 0/mm^2         22\n    # <1/mm^2        19\n    # >4/mm^2         4\n    # 1/mm^2          3\n    # 3/mm^2          2\n    # 2/mm^2          2\n    # 4/mm^2          1\n    'mel_mitotic_index'\n]\n\n# Basically these columns are hierarchical.\n# The higher the number the more specific the diagnosis provided.\n# The iddx_full column concatanates non NaN values in the 5 iddx columns with '::' between.\nSTR_COLS = [\n    'iddx_full', 'iddx_1', 'iddx_2', 'iddx_3', 'iddx_4', 'iddx_5'\n]\n\nNUM_COLS = [\n    \n    # 0    400666\n    # 1       393\n    'target',  # int\n\n    # 55.0    58123\n    # 65.0    54946\n    # 60.0    54109\n    # 50.0    47924\n    # 70.0    39775\n    # 40.0    31297\n    # 75.0    30801\n    # 45.0    23580\n    # 80.0    21096\n    # 35.0    11543\n    # 30.0    10400\n    # 85.0     8847\n    # 25.0     3433\n    # NaN      2798\n    # 20.0     1742\n    # 15.0      644\n    # 5.0         1\n    'age_approx',  # Should be an 'int' but the NaN casts it to float\n    \n    # 1758 unique\n    'clin_size_long_diam_mm',  # float\n    \n    # Almost 400,000 unique for all of these\n    'tbp_lv_A',   # float\n    'tbp_lv_Aext',   # float \n    'tbp_lv_B',   # float\n    'tbp_lv_Bext',   # float \n    'tbp_lv_C',   # float \n    'tbp_lv_Cext',   # float \n    'tbp_lv_H',   # float \n    'tbp_lv_Hext',   # float \n    'tbp_lv_L',   # float\n    'tbp_lv_Lext',   # float \n    \n    # 8000 or so unique\n    'tbp_lv_areaMM2',  # float\n    \n    # 160,000 unique\n    'tbp_lv_area_perim_ratio',  # float \n    \n    # 26187 at 0.0 and the rest spread\n    'tbp_lv_color_std_mean',  # float\n    \n    # nearly all unique\n    'tbp_lv_deltaA',  # float\n    'tbp_lv_deltaB',  # float\n    'tbp_lv_deltaL',  # float\n    'tbp_lv_deltaLB',  # float\n    'tbp_lv_deltaLBnorm',  # float\n    'tbp_lv_eccentricity',  # float\n    \n    # 80000 unique\n    'tbp_lv_minorAxisMM',  # float \n    \n    # nearly all unique (heavily stacked near 100%)\n    'tbp_lv_nevi_confidence',  # float \n    \n    # nearly all unique (but a weird bunch at exactly 10.0)\n    'tbp_lv_norm_border',  # float \n    \n    # nearly all unique (but a weird bunch at exactly 0.0 and 10.0)\n    'tbp_lv_norm_color',  # float \n    \n    # 9500 or so unique\n    'tbp_lv_perimeterMM',  # float \n    \n    # nearly all unique (but a huge chunk at 0.0 (40k))\n    'tbp_lv_radial_color_std_max',  # float \n    \n    # nearly all unique for both\n    'tbp_lv_stdL',  # float \n    'tbp_lv_stdLExt',  # float \n    \n    # 75000 unique\n    'tbp_lv_symm_2axis',\n    \n    # 36 unique\n    'tbp_lv_symm_2axis_angle',  # int\n    \n    # nearly all unique\n    'tbp_lv_x',  # float\n    'tbp_lv_y',  # float\n    'tbp_lv_z',  # float\n    \n    # 400996 NaN (nearly all)\n    'mel_thick_mm',  # float\n    \n    # A bunch stacked at 100 (and near 100) but mostly unique\n    'tbp_lv_dnn_lesion_confidence'  # float\n]\n\nID_COLS = ['isic_id', 'patient_id', 'lesion_id']\nCOLS_TO_IGNORE = ID_COLS + STR_COLS + [\"copyright_license\", \"attribution\", \"image_type\", \"mel_thick_mm\", \"mel_mitotic_index\"]\nCOLS_TO_ONE_HOT = [_c for _c in CAT_COLS if _c not in COLS_TO_IGNORE]\nTARGET_COL = \"target\"","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:14.819302Z","iopub.execute_input":"2024-07-06T01:08:14.819743Z","iopub.status.idle":"2024-07-06T01:08:14.839748Z","shell.execute_reply.started":"2024-07-06T01:08:14.819702Z","shell.execute_reply":"2024-07-06T01:08:14.838350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\n\n\ndef handle_missing_values(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"\n    Handle missing values in the dataframe.\n\n    For numerical columns, fill NaNs with the median value.\n    For categorical columns, fill NaNs with the most frequent value (mode).\n\n    Args:\n        df (pd.DataFrame): Input dataframe\n\n    Returns:\n        pd.DataFrame: Dataframe with missing values handled\n    \"\"\"\n    # Identify numerical and categorical columns\n    num_cols = df.select_dtypes(include=['int64', 'float64']).columns\n    cat_cols = df.select_dtypes(include=['object', 'category']).columns\n\n    # Handle missing values in numerical columns\n    num_imputer = SimpleImputer(strategy='median')\n    df[num_cols] = num_imputer.fit_transform(df[num_cols])\n\n    # Handle missing values in categorical columns\n    cat_imputer = SimpleImputer(strategy='most_frequent')\n    df[cat_cols] = cat_imputer.fit_transform(df[cat_cols])\n\n    return df\n\ndef one_hot_encode(df: pd.DataFrame, columns_to_encode: list[str]) -> pd.DataFrame:\n    \"\"\"Perform one-hot encoding on specified categorical columns.\n\n    Args:\n        df (pd.DataFrame): Input dataframe\n        columns_to_encode (list[str]): List of column names to one-hot encode\n\n    Returns:\n        pd.DataFrame: Dataframe with specified columns one-hot encoded\n    \"\"\"\n    # Filter columns that are actually present in the dataframe\n    columns_to_encode = [col for col in columns_to_encode if col in df.columns]\n\n    # Perform one-hot encoding\n    encoder = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    encoded_cols = encoder.fit_transform(df[columns_to_encode])\n    \n    # Create new column names for encoded features\n    new_columns = encoder.get_feature_names_out(columns_to_encode)\n    \n    # Create a new dataframe with encoded features\n    encoded_df = pd.DataFrame(encoded_cols, columns=new_columns, index=df.index)\n    \n    # Concatenate the original dataframe with the encoded features\n    result_df = pd.concat([df.drop(columns_to_encode, axis=1), encoded_df], axis=1)\n    \n    return result_df\n\ndef scale_numerical_features(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"Scale numerical features using StandardScaler.\n\n    Args:\n        df (pd.DataFrame): Input dataframe\n\n    Returns:\n        pd.DataFrame: Dataframe with scaled numerical features\n    \"\"\"\n    # Identify numerical columns (excluding the target column)\n    num_cols = df.select_dtypes(include=['int64', 'float64']).columns\n    num_cols = [col for col in num_cols if col != TARGET_COL]\n\n    # Scale numerical features\n    scaler = StandardScaler()\n    df[num_cols] = scaler.fit_transform(df[num_cols])\n\n    return df\n\ndef handle_age_approx(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"Handle the 'age_approx' column by rounding it to the nearest integer.\n\n    Args:\n        df (pd.DataFrame): Input dataframe\n\n    Returns:\n        pd.DataFrame: Dataframe with 'age_approx' rounded to nearest integer\n    \"\"\"\n    if 'age_approx' in df.columns:\n        df['age_approx'] = df['age_approx'].round().astype('Int64')\n    return df\n\ndef preprocess_df(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"\n    Preprocess the input dataframe for machine learning tasks.\n\n    This function performs the following steps:\n    1. Handle missing values\n    2. One-hot encode categorical variables\n    3. Scale numerical features\n    4. Handle the 'age_approx' column\n    5. Drop unnecessary columns\n\n    Args:\n        df (pd.DataFrame): Input dataframe\n\n    Returns:\n        pd.DataFrame: Preprocessed dataframe ready for machine learning tasks\n    \"\"\"\n    # Just in case\n    _df = df.copy()\n    \n    # Handle missing values\n    clr_print(\"Handling missing values\")\n    _df = handle_missing_values(_df)\n\n    # One-hot encode categorical variables\n    clr_print(\"One hot encoding categorical variables\")\n    columns_to_encode = [col for col in COLS_TO_ONE_HOT if col in _df.columns]\n    _df = one_hot_encode(_df, columns_to_encode)\n    \n    # Scale numerical features\n    clr_print(\"Scaling numerical features\")\n    _df = scale_numerical_features(_df)\n\n    # Drop unnecessary columns\n    clr_print(\"Dropping unnecessary remaining columns\")\n    columns_to_drop = [col for col in COLS_TO_IGNORE+COLS_TO_ONE_HOT if col in _df.columns]\n    return _df.drop(columns=columns_to_drop)\n\npreprocessed_train_df = preprocess_df(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:14.841335Z","iopub.execute_input":"2024-07-06T01:08:14.841679Z","iopub.status.idle":"2024-07-06T01:08:23.721600Z","shell.execute_reply.started":"2024-07-06T01:08:14.841649Z","shell.execute_reply":"2024-07-06T01:08:23.720435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_target_correlation(\n    df: pd.DataFrame,\n    target_col: str = 'target',\n    n_top_features: int = 30,\n    color_sequence: list[str] | None = None,\n    template_theme: str = \"plotly_white\"\n) -> None:\n    \"\"\"Create a correlation plot showing the top correlated features with the target variable.\n\n    Args:\n        df (pd.DataFrame): \n            The input dataframe containing feature and target columns.\n        target_col (str, optional): \n            Name of the target column.\n        n_top_features (int, optional): \n            Number of top correlated features to display.\n        color_sequence (list[str], optional): \n            Custom color sequence for the plot.\n            Defaults to None, which uses a blue to red color scale.\n        template_theme (str, optional): \n            Plotly template theme for visual styling.\n\n    Returns:\n        None: This function displays a plot and doesn't return anything.\n    \"\"\"\n    # Calculate correlations\n    correlations = df.corr()[target_col]\n    \n    # Sort by absolute correlation value\n    correlations_abs = correlations.abs().sort_values(ascending=False)\n    \n    # Select top correlated features (excluding the target itself)\n    top_correlations = correlations[correlations_abs.index[1:n_top_features+1]]\n    \n    # Prepare data for plotting\n    feature_names = top_correlations.index\n    correlation_values = top_correlations.values\n    \n    # Set up color scale\n    if color_sequence is None:\n        color_sequence = ['#0d0887', '#46039f', '#7201a8', '#9c179e', '#bd3786', '#d8576b', '#ed7953', '#fb9f3a', '#fdca26', '#f0f921']\n    \n    # Create the bar plot\n    fig = go.Figure()\n    fig.add_trace(go.Bar(\n        y=feature_names,\n        x=correlation_values,\n        orientation='h',\n        marker=dict(\n            color=correlation_values,\n            colorscale=color_sequence,\n            colorbar=dict(title=\"Correlation\"),\n        )\n    ))\n    \n    # Customize the layout\n    fig.update_layout(\n        title=f\"<b>Top {n_top_features} Features Correlated with {target_col.capitalize()}</b>\",\n        xaxis_title=\"<b>Correlation Coefficient</b>\",\n        yaxis_title=\"<b>Feature</b>\",\n        height=800,\n        width=1200,\n        template=template_theme,\n    )\n    \n    # Add vertical line at x=0 for reference\n    fig.add_shape(\n        type=\"line\",\n        x0=0, y0=-0.5,\n        x1=0, y1=len(feature_names) - 0.5,\n        line=dict(color=\"black\", width=1, dash=\"dash\")\n    )\n    \n    # Display the plot\n    fig.show()\n    \nplot_target_correlation(preprocessed_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:23.722900Z","iopub.execute_input":"2024-07-06T01:08:23.723294Z","iopub.status.idle":"2024-07-06T01:08:29.911832Z","shell.execute_reply.started":"2024-07-06T01:08:23.723262Z","shell.execute_reply":"2024-07-06T01:08:29.910650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compare_area_to_patient_mean(df: pd.DataFrame, area_col: str = \"tbp_lv_areaMM2\", patient_col: str = \"patient_id\") -> pd.DataFrame:\n    \"\"\"\n    Create a new column that quantifies how much bigger a particular area is compared to the mean for the patient.\n\n    Args:\n        df (pd.DataFrame): The input dataframe.\n        area_col (str, optional): The name of the column containing the area measurements. Defaults to \"tbp_lv_areaMM2\".\n        patient_col (str, optional): The name of the column containing patient IDs. Defaults to \"patient_id\".\n\n    Returns:\n        pd.DataFrame: The input dataframe with a new column added.\n    \"\"\"\n    # Copy\n    _df = df.copy()\n    \n    # Calculate the mean area for each patient\n    patient_mean_area = _df.groupby(patient_col)[area_col].transform('mean')\n    \n    # Calculate the ratio of each area to the patient's mean area\n    _df[f'{area_col}_ratio_to_patient_mean'] = _df[area_col] / patient_mean_area\n    \n    # Calculate the percentage difference from the patient's mean area\n    _df[f'{area_col}_pct_diff_from_patient_mean'] = (_df[area_col] - patient_mean_area) / patient_mean_area * 100\n    \n    return _df\n\n# Possible new features could be created...\ncompare_area_to_patient_mean(train_df)[[\"tbp_lv_areaMM2_pct_diff_from_patient_mean\", \"target\"]].corr()","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:29.913800Z","iopub.execute_input":"2024-07-06T01:08:29.914194Z","iopub.status.idle":"2024-07-06T01:08:30.087304Z","shell.execute_reply.started":"2024-07-06T01:08:29.914151Z","shell.execute_reply":"2024-07-06T01:08:30.086189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #375B6D; background-color: #ffffff;\">5.2 IMAGE <b>EXPLORATION</b></h3>\n<hr><br>\n\n","metadata":{}},{"cell_type":"code","source":"def plot_img_from_id(\n    isic_id: str,\n    hdf5_file_path: str = \"/kaggle/input/isic-2024-challenge/train-image.hdf5\",\n    figsize: tuple = (6, 6)\n):\n    \"\"\"Plot an image from the HDF5 dataset file by specifying an ISIC ID.\n    \n    Args:\n        isic_id (str): The ID of the image to plot.\n        hdf5_file_path (str): The path to the HDF5 file.\n        figsize (tuple): The size of the figure (width, height) in inches.\n    \n    Returns:\n        None; Plots the image\n    \n    Example Usage:\n        plot_img_from_id('ISIC_0000000')\n    \"\"\"\n    # Load the image using the existing function\n    img = load_img_from_hdf5(isic_id, hdf5_file_path)\n    \n    # Create a new figure with the specified size\n    plt.figure(figsize=figsize)\n    \n    # Set the title to the ISIC ID\n    plt.title(f\"ISIC ID: {isic_id}\", fontweight=\"bold\")\n    \n    # Display the image\n    plt.imshow(img)\n    \n    # Remove axis and update plot layout\n    plt.axis('off')\n    plt.tight_layout()\n    \n    # Show the plot\n    plt.show()\n    \nplot_img_from_id(train_df.sample(1).isic_id.values[0])","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:30.088680Z","iopub.execute_input":"2024-07-06T01:08:30.089037Z","iopub.status.idle":"2024-07-06T01:08:30.333891Z","shell.execute_reply.started":"2024-07-06T01:08:30.089009Z","shell.execute_reply":"2024-07-06T01:08:30.332793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_colored_border(img: np.ndarray, color: tuple[int] | str, border_width: int = 3) -> np.ndarray:\n    \"\"\"Add a colored border to an image.\n    \n    Args:\n        img (np.ndarray): Input image as a numpy array.\n        color (tuple[int] | str): Border color in RGB format OR a HEX string.\n        border_width (int, optional): Width of the border in pixels.\n    \n    Returns:\n        np.ndarray: Image with added border.\n    \"\"\"\n    \n    # Ensure color is RGB\n    if isinstance(color, str) and color.startswith(\"#\"):\n        color = ImageColor.getcolor(color, \"RGB\")\n        \n    # Draw border\n    bordered_img = cv2.copyMakeBorder(\n        src=img, \n        top=border_width, \n        bottom=border_width, \n        left=border_width, \n        right=border_width, \n        borderType=cv2.BORDER_CONSTANT, \n        value=color\n    )\n    return bordered_img\n\n\ndef plot_patient_images(\n    patient_isic_ids: list[str],\n    labels: list[int] | None = None,\n    patient_id: str | None = None,\n    hdf5_file_path: str = \"/kaggle/input/isic-2024-challenge/train-image.hdf5\",\n    max_images: int = 60,\n    images_per_row: int = 10,\n    fig_width: int = 20,\n):\n    \"\"\"Plot multiple images for a patient in a tiled layout with colored borders\n    \n    Args:\n        patient_isic_ids (list[str]): \n            List of ISIC IDs for the patient's images.\n        labels (list[int], optional): \n            The list of labels which will be used to color code malignant images.\n                - Malignant images will be outlined with red (COLORS[1])\n                - Benign images will be outlined with green (COLORS[0])\n        patient_id (str, optional): \n            The patient id the isic_ids belong to.\n        hdf5_file_path (str, optional): \n            The path to the HDF5 file.\n        max_images (int, optional): \n            Maximum number of images to display.\n        images_per_row (int, optional): \n            Number of images to display in each row.\n        fig_width (int, optional): \n            The size of the figure width\n    \n    Returns:\n        None; \n            plots the tiled images\n    \"\"\"\n    # Limit the number of images to plot\n    total_num_images = len(patient_isic_ids)\n    patient_isic_ids = patient_isic_ids[:max_images]\n    num_of_images_to_plot = len(patient_isic_ids)\n    \n    # Calculate the number of rows needed and figsize\n    num_rows = math.ceil(num_of_images_to_plot / images_per_row)\n    figsize = (fig_width, int(2.666*num_rows))\n    \n    # Create the figure\n    fig = plt.figure(figsize=figsize)\n    plt.suptitle(f\"IMAGES FOR PATIENT: {patient_id}  (showing {num_of_images_to_plot} out of {total_num_images} images)\", fontsize=16, fontweight=\"bold\")\n    \n    # Process images in batches\n    for i, isic_id in enumerate(patient_isic_ids):\n        # Calculate the subplot position\n        position = i + 1\n        # Create a new subplot\n        plt.subplot(num_rows, images_per_row, position)\n        plt.title(f\"{isic_id}{' - '+LBL_MAP_I2S[labels[i]] if labels is not None else ''}\", fontsize=8 if labels is None else 7)\n        \n        # Load the image\n        img = load_img_from_hdf5(isic_id, hdf5_file_path)\n        \n        # Add colored border\n        if labels is not None:\n            img = add_colored_border(img, COLORS[labels[i]])\n        \n        # Display the image\n        plt.imshow(img)\n        plt.axis('off')\n    \n    fig.tight_layout(rect=[0, 0.03, 1, 0.97])\n    plt.show()\n    \n\nDEMO_PATIENT_DF = train_df[train_df.patient_id==train_df[train_df.target==1][\"patient_id\"].sample(1).values[0]].sort_values(by=[\"target\", \"isic_id\"], ascending=False).reset_index(drop=True)\n\ndisplay_hr(True, True)\nclr_print(f\"# OF MALIGNANT TILES:  <code>{DEMO_PATIENT_DF.target.sum()}</code>\")\ndisplay_hr(True, True)\ndisplay(DEMO_PATIENT_DF)\ndisplay_hr(True, True)\n\nplot_patient_images(\n    patient_isic_ids=DEMO_PATIENT_DF.isic_id.to_list(),\n    labels=DEMO_PATIENT_DF.target.to_list(),\n    patient_id=DEMO_PATIENT_DF.patient_id[0]\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:30.335837Z","iopub.execute_input":"2024-07-06T01:08:30.336955Z","iopub.status.idle":"2024-07-06T01:08:49.118982Z","shell.execute_reply.started":"2024-07-06T01:08:30.336895Z","shell.execute_reply":"2024-07-06T01:08:49.117743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plot batches of images so we can examine images representative of a particular feature**","metadata":{}},{"cell_type":"code","source":"def plot_image_batch(\n    isic_ids: list[str],\n    labels: list[int] | None = None,\n    batch_description: str | None = None,\n    hdf5_file_path: str = \"/kaggle/input/isic-2024-challenge/train-image.hdf5\",\n    max_images: int = 24,\n    images_per_row: int = 8,\n    fig_width: int = 20,\n):\n    \"\"\"Plot multiple images as a batch.\n    \n    Args:\n        isic_ids (list[str]): \n            List of ISIC IDs for the batch.\n        labels (list[int], optional): \n            The list of labels which will be used to color code malignant images.\n                - Malignant images will be outlined with red (COLORS[1])\n                - Benign images will be outlined with green (COLORS[0])\n        batch_description (str, optional):\n            The description for what we are trying to visualize within the batch\n        hdf5_file_path (str, optional): \n            The path to the HDF5 file.\n        max_images (int, optional): \n            Maximum number of images to display.\n        images_per_row (int, optional): \n            Number of images to display in each row.\n        fig_width (int, optional): \n            The size of the figure width\n    \n    Returns:\n        None; \n            plots the tiled images\n    \"\"\"\n    # Limit the number of images to plot\n    total_num_images = len(isic_ids)\n    isic_ids = isic_ids[:max_images]\n    num_of_images_to_plot = len(isic_ids)\n    \n    # Calculate the number of rows needed and figsize\n    num_rows = math.ceil(num_of_images_to_plot / images_per_row)\n    figsize = (fig_width, int(3*num_rows))\n    \n    # Create the figure\n    fig = plt.figure(figsize=figsize)\n    plt.suptitle(f\"IMAGE BATCH - {batch_description or '<UNK>'}\", fontsize=14, fontweight=\"bold\")\n    \n    # Process images in batches\n    for i, isic_id in enumerate(isic_ids):\n        # Calculate the subplot position\n        position = i + 1\n        \n        # Create a new subplot\n        plt.subplot(num_rows, images_per_row, position)\n        plt.title(f\"{isic_id}{' - '+LBL_MAP_I2S[labels[i]] if labels is not None else ''}\", fontsize=8 if labels is None else 7)\n        \n        # Load the image\n        img = load_img_from_hdf5(isic_id, hdf5_file_path)\n        \n        # Add colored border\n        if labels is not None:\n            img = add_colored_border(img, COLORS[labels[i]])\n        \n        # Display the image\n        plt.imshow(img)\n        plt.axis('off')\n    \n    fig.tight_layout(rect=[0, 0.03, 1, 0.97])\n    plt.show()\n    \n    \ndef get_balanced_df(df: pd.DataFrame, balance_col: str, shuffle: bool = True) -> pd.DataFrame:\n    \"\"\"Balances the DataFrame by sampling an equal number of rows from each category.\n    \n    Args:\n        df (pd.DataFrame): The input DataFrame to balance.\n        balance_col (str): The column name to balance by.\n        shuffle (bool): Whether to shuffle the resulting dataframe. Defaults to True.\n    \n    Returns:\n        pd.DataFrame: A balanced DataFrame with an equal number of rows from each category.\n    \"\"\"\n    # Create safe edit copy\n    _df = df.copy()\n    \n    # Get value counts and determine the size of the smallest category\n    min_col_count = _df[balance_col].value_counts().min()\n    \n    # Sample from each category and concatenate all balanced samples\n    balanced_df = pd.concat([\n        _df[_df[balance_col] == category].sample(n=min_col_count, replace=False) \n        for category in df[balance_col].unique()\n    ], ignore_index=True)\n    \n    \n    # Shuffle the combined DataFrame if required\n    if shuffle:\n        balanced_df = balanced_df.sample(frac=1).reset_index(drop=True)\n    \n    return balanced_df\n\n\ndef process_batches(\n    df: pd.DataFrame, \n    label_split: str, \n    feature_col: str\n) -> dict[str, dict[str, list[Any]]]:\n    \"\"\"Processes batches from the training DataFrame based on label split criteria.\n\n    Args:\n        df (pd.DataFrame): \n            The input DataFrame.\n        label_split (str): \n            The criteria for splitting labels (\"malignant\", \"benign\", \"half\", or \"random\").\n        feature_col (str): \n            The column name to group by for batch processing.\n\n    Returns:\n        dict[str, dict[str, list[Any]]]: \n            A dictionary with feature strings as keys and dictionaries of ISIC IDs and labels as values.\n    \"\"\"\n    batches = {}\n    for feature_str, _df in train_df.groupby(feature_col):\n        if label_split == \"malignant\":\n            _df = _df[_df[\"target\"] == 1]\n        elif label_split == \"benign\":\n            _df = _df[_df[\"target\"] == 0]\n        elif label_split == \"half\":\n            _df = get_balanced_df(_df, \"target\")\n        batches[feature_str] = {\"isic_ids\": _df[\"isic_id\"].to_list(), \"labels\": _df[\"target\"].tolist()}\n    return batches\n\n\ndef plot_batches(\n    feature_col: str, \n    label_split: str,\n    df: pd.DataFrame | None = None,\n    batches: dict[str, dict[str, list[Any]]] | None = None, \n) -> None:\n    \"\"\"Displays and plots image batches with descriptions.\n\n    Args:\n        feature_col (str): \n            The column name to describe the batches.\n        label_split (str): \n            The label split style used.\n        df (pd.DataFrame, optional): \n            The input DataFrame.\n        batches (dict[str, dict[str, list[Any]]], optional): \n            The dictionary containing batch data to display and plot.\n            \n    Raises:\n        ValueError:\n            If the input data is not correctly passed.\n    \n    Returns:\n        None; Plots...\n    \"\"\"\n    \n    if df is None and batches is None:\n        raise ValueError(\"\\n... One of `df` or `batches` must be provided as an input ...\\n\")\n    elif batches is None:\n        batches = process_batches(df, label_split, feature_col)\n    \n    display_hr(True, True)\n    clr_print(METADATA_COL2DESC[feature_col])\n    for feature_str, feature_batch in batches.items():\n        display_hr(True, True)\n        plot_image_batch(\n            **feature_batch, \n            batch_description=f\"{METADATA_COL2NAME[feature_col]} - '{feature_str}' - SPLIT STYLE={label_split}\"\n        )\n        display_hr(True, True)\n        \n\nLABEL_SPLIT = \"half\"  # malignant, benign, half, random\nDEMO_FEAT_COL = \"anatom_site_general\"\n\nDEMO_BATCHES = process_batches(train_df, LABEL_SPLIT, DEMO_FEAT_COL)\nplot_batches(DEMO_FEAT_COL, LABEL_SPLIT, batches=DEMO_BATCHES)\n\nplot_batches(\"tbp_tile_type\", LABEL_SPLIT, df=train_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:08:49.120840Z","iopub.execute_input":"2024-07-06T01:08:49.121541Z","iopub.status.idle":"2024-07-06T01:09:26.858786Z","shell.execute_reply.started":"2024-07-06T01:08:49.121502Z","shell.execute_reply":"2024-07-06T01:09:26.857455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**BASIC MODEL**\n\nTBD - for now we will just submit based on area","metadata":{}},{"cell_type":"code","source":"# # train_df[train_df.target==1][\"tbp_lv_areaMM2\"].describe()\n# tbp_lv_areaMM2_thresh = 10.0\n\n# random_ss_df = train_df[[\"isic_id\", \"target\"]].copy()\n# random_ss_df[\"target\"] = [random.random() for i in range(len(random_ss_df))]\n# train_ss_df = train_df[[\"isic_id\", \"target\"]].copy()\n# train_ss_df[\"target\"] = train_df[\"tbp_lv_areaMM2\"]>tbp_lv_areaMM2_thresh\n\n# clr_print(\"AREA THRESH PREDICTION SCORE ON TRAINING DATA:\")\n# print(score(train_df[[\"isic_id\", \"target\"]].copy(), train_ss_df, row_id_column_name=\"isic_id\"))\n\n# clr_print(\"<br>RANDOM PREDICTION SCORE ON TRAINING DATA:\")\n# print(score(train_df[[\"isic_id\", \"target\"]].copy(), random_ss_df, row_id_column_name=\"isic_id\"))\n\n# ss_df[\"target\"] = test_df[\"tbp_lv_areaMM2\"]>tbp_lv_areaMM2_thresh\n# ss_df.to_csv(\"submission.csv\", index=False)\n# ss_df","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:09:26.870400Z","iopub.status.idle":"2024-07-06T01:09:26.870868Z","shell.execute_reply.started":"2024-07-06T01:09:26.870648Z","shell.execute_reply":"2024-07-06T01:09:26.870669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_additional_features(df: pd.DataFrame) -> tuple[pd.DataFrame, list[str]]:\n    \"\"\"Create additional features based on domain knowledge and existing features.\n\n    Args:\n        df (pd.DataFrame): Input dataframe containing original and engineered features.\n\n    Returns:\n        tuple[pd.DataFrame, list[str]]: A tuple containing:\n            - The dataframe with additional features added\n            - A list of new numerical column names\n    \"\"\"\n    additional_num_cols = []\n\n    # 1. Color Variance Ratio\n    #    - This feature compares the color variance within the lesion to the color variance of the surrounding skin.\n    df[\"color_variance_ratio\"] = df[\"tbp_lv_color_std_mean\"] / df[\"tbp_lv_stdLExt\"]\n    \n\n    # 2. Border Color Interaction\n    #    - This interaction term combines border irregularity with color variation.\n    df[\"border_color_interaction\"] = df[\"tbp_lv_norm_border\"] * df[\"tbp_lv_norm_color\"]\n\n    # 3. Size Color Contrast Ratio\n    #    - This ratio might help identify large lesions with low contrast, or small lesions with high contrast.\n    df[\"size_color_contrast_ratio\"] = df[\"clin_size_long_diam_mm\"] / df[\"tbp_lv_deltaLBnorm\"]\n\n    # 4. Age Normalized Nevi Confidence\n    #    - This feature adjusts the nevus confidence score by age.\n    df[\"age_normalized_nevi_confidence\"] = df[\"tbp_lv_nevi_confidence\"] / df[\"age_approx\"]\n\n    # 5. Color Asymmetry Index\n    #    - This combines color asymmetry with border asymmetry.\n    df[\"color_asymmetry_index\"] = df[\"tbp_lv_radial_color_std_max\"] * df[\"tbp_lv_symm_2axis\"]\n\n    # 6. 3D Volume Approximation\n    #    - This attempts to approximate the volume of the lesion in 3D space.\n    df[\"3d_volume_approximation\"] = df[\"tbp_lv_areaMM2\"] * np.sqrt(df[\"tbp_lv_x\"]**2 + df[\"tbp_lv_y\"]**2 + df[\"tbp_lv_z\"]**2)\n\n    # 7. Color Range\n    #     - This feature captures the total range of color difference across all color dimensions.\n    df[\"color_range\"] = (df[\"tbp_lv_L\"] - df[\"tbp_lv_Lext\"]).abs() + (df[\"tbp_lv_A\"] - df[\"tbp_lv_Aext\"]).abs() + (df[\"tbp_lv_B\"] - df[\"tbp_lv_Bext\"]).abs()\n\n    # 8. Shape Color Consistency\n    #    - This interaction term might help identify lesions that are both elongated and have inconsistent coloration.\n    df[\"shape_color_consistency\"] = df[\"tbp_lv_eccentricity\"] * df[\"tbp_lv_color_std_mean\"]\n\n    # 9. Border Length Ratio\n    #    - This ratio compares the actual perimeter to the perimeter of a perfect circle with the same area.\n    df[\"border_length_ratio\"] = df[\"tbp_lv_perimeterMM\"] / (2 * np.pi * np.sqrt(df[\"tbp_lv_areaMM2\"] / np.pi))\n\n    # 10. Age Size Symmetry Index\n    #    - This composite feature combines age, size, and asymmetry.\n    df[\"age_size_symmetry_index\"] = df[\"age_approx\"] * df[\"clin_size_long_diam_mm\"] * df[\"tbp_lv_symm_2axis\"]\n    \n    additional_num_cols += [\n        \"color_variance_ratio\", \"border_color_interaction\", \"size_color_contrast_ratio\", \n        \"age_normalized_nevi_confidence\", \"color_asymmetry_index\", \"3d_volume_approximation\", \n        \"color_range\", \"shape_color_consistency\", \"border_length_ratio\", \"age_size_symmetry_index\"\n    ]\n    return df, additional_num_cols\n\n\ndef feature_engineering(df: pd.DataFrame) -> tuple[pd.DataFrame, list[str], list[str]]:\n    \"\"\"Perform comprehensive feature engineering on the input dataframe for skin cancer detection.\n    \n    Code originally from here --> https://www.kaggle.com/code/vyacheslavbolotin/ensemble-lgbm-cat-with-new-features\n    I have made a function out of the above code and done my best to explain things in more detail.\n\n    This function creates new features based on existing ones, potentially improving\n    the model's ability to detect skin cancer. It includes various geometric, color,\n    and composite features that may be indicative of malignancy.\n\n    Args:\n        df (pd.DataFrame): Input dataframe containing original features.\n\n    Returns:\n        Tuple[pd.DataFrame, list[str], list[str]]: A tuple containing:\n            - The dataframe with new features added\n            - A list of new numerical column names\n            - A list of new categorical column names\n    \"\"\"\n    \n    ### [Original feature engineering code from notebook] ###\n    \n    # Geometric features\n    #   - This ratio can help identify irregular growth patterns typical in melanomas\n    df[\"lesion_size_ratio\"] = df[\"tbp_lv_minorAxisMM\"] / df[\"clin_size_long_diam_mm\"]\n    #   - A measure of compactness; melanomas often have more irregular shapes\n    df[\"lesion_shape_index\"] = df[\"tbp_lv_areaMM2\"] / (df[\"tbp_lv_perimeterMM\"] ** 2)\n    #   - Another measure of shape irregularity; higher values may indicate more complex borders\n    df[\"perimeter_to_area_ratio\"] = df[\"tbp_lv_perimeterMM\"] / df[\"tbp_lv_areaMM2\"]\n\n    # Color-based features\n    #   - Contrast between lesion and surrounding skin can be indicative of malignancy\n    df[\"hue_contrast\"] = (df[\"tbp_lv_H\"] - df[\"tbp_lv_Hext\"]).abs()\n    df[\"luminance_contrast\"] = (df[\"tbp_lv_L\"] - df[\"tbp_lv_Lext\"]).abs()\n    #   - Overall color difference in 3D color space; larger differences may suggest malignancy\n    df[\"lesion_color_difference\"] = np.sqrt(df[\"tbp_lv_deltaA\"] ** 2 + df[\"tbp_lv_deltaB\"] ** 2 + df[\"tbp_lv_deltaL\"] ** 2)\n    #   - Measure of color consistency; benign lesions often have more uniform color\n    df[\"color_uniformity\"] = df[\"tbp_lv_color_std_mean\"] / df[\"tbp_lv_radial_color_std_max\"]\n\n    # Composite features\n    #   - Combines border irregularity and asymmetry, both indicators of potential malignancy\n    df[\"border_complexity\"] = df[\"tbp_lv_norm_border\"] + df[\"tbp_lv_symm_2axis\"]\n    #   - 3D position might correlate with certain high-risk body areas\n    df[\"3d_position_distance\"] = np.sqrt(df[\"tbp_lv_x\"] ** 2 + df[\"tbp_lv_y\"] ** 2 + df[\"tbp_lv_z\"] ** 2) \n    #   - Combines color and brightness differences, potentially highlighting atypical lesions\n    df[\"lesion_visibility_score\"] = df[\"tbp_lv_deltaLBnorm\"] + df[\"tbp_lv_norm_color\"]\n    #   - More specific anatomical location might correlate with cancer risk\n    df[\"combined_anatomical_site\"] = df[\"anatom_site_general\"] + \"_\" + df[\"tbp_lv_location\"]\n    #   - Interaction between symmetry and border regularity; asymmetric lesions with irregular borders are more suspicious\n    df[\"symmetry_border_consistency\"] = df[\"tbp_lv_symm_2axis\"] * df[\"tbp_lv_norm_border\"]\n    #   - Measure of color consistency within the lesion compared to surrounding skin\n    df[\"color_consistency\"] = df[\"tbp_lv_stdL\"] / df[\"tbp_lv_Lext\"]\n    #   - Larger lesions in older patients might be more concerning\n    df[\"size_age_interaction\"] = df[\"clin_size_long_diam_mm\"] * df[\"age_approx\"]\n    #   - Interaction between hue and color variation might highlight atypical pigmentation\n    df[\"hue_color_std_interaction\"] = df[\"tbp_lv_H\"] * df[\"tbp_lv_color_std_mean\"]\n    #   -  Composite score combining border, color, and shape irregularities\n    df[\"lesion_severity_index\"] = (df[\"tbp_lv_norm_border\"] + df[\"tbp_lv_norm_color\"] + df[\"tbp_lv_eccentricity\"]) / 3\n    #   - Overall measure of shape complexity\n    df[\"shape_complexity_index\"] = df[\"border_complexity\"] + df[\"lesion_shape_index\"]\n    #   - Comprehensive measure of color contrast\n    df[\"color_contrast_index\"] = df[\"tbp_lv_deltaA\"] + df[\"tbp_lv_deltaB\"] + df[\"tbp_lv_deltaL\"] + df[\"tbp_lv_deltaLBnorm\"]\n    #   - Log transformation can help handle skewed size distributions\n    df[\"log_lesion_area\"] = np.log(df[\"tbp_lv_areaMM2\"] + 1)\n    #   - Size relative to patient age; rapid growth might be more concerning in younger patients\n    df[\"normalized_lesion_size\"] = df[\"clin_size_long_diam_mm\"] / df[\"age_approx\"]\n    #   - Average hue might indicate overall pigmentation level\n    df[\"mean_hue_difference\"] = (df[\"tbp_lv_H\"] + df[\"tbp_lv_Hext\"]) / 2\n    #   - Standard deviation of color contrast across different color dimensions\n    df[\"std_dev_contrast\"] = np.sqrt((df[\"tbp_lv_deltaA\"] ** 2 + df[\"tbp_lv_deltaB\"] ** 2 + df[\"tbp_lv_deltaL\"] ** 2) / 3)\n    #   - Composite index combining color variation, shape, and symmetry\n    df[\"color_shape_composite_index\"] = (df[\"tbp_lv_color_std_mean\"] + df[\"tbp_lv_area_perim_ratio\"] + df[\"tbp_lv_symm_2axis\"]) / 3\n    #   - Orientation of the lesion in 3D space might correlate with certain types of growths\n    df[\"3d_lesion_orientation\"] = np.arctan2(df[\"tbp_lv_y\"], df[\"tbp_lv_x\"])\n    #   - Average color difference across different color dimensions\n    df[\"overall_color_difference\"] = (df[\"tbp_lv_deltaA\"] + df[\"tbp_lv_deltaB\"] + df[\"tbp_lv_deltaL\"]) / 3\n    #   - Interaction between symmetry and perimeter; large asymmetric lesions might be more concerning\n    df[\"symmetry_perimeter_interaction\"] = df[\"tbp_lv_symm_2axis\"] * df[\"tbp_lv_perimeterMM\"]\n    #    - Comprehensive index combining multiple aspects of lesion characteristics\n    df[\"comprehensive_lesion_index\"] = (df[\"tbp_lv_area_perim_ratio\"] + df[\"tbp_lv_eccentricity\"] + df[\"tbp_lv_norm_color\"] + df[\"tbp_lv_symm_2axis\"]) / 4\n    \n    new_num_cols = [\n        \"lesion_size_ratio\", \"lesion_shape_index\", \"hue_contrast\",\n        \"luminance_contrast\", \"lesion_color_difference\", \"border_complexity\",\n        \"color_uniformity\", \"3d_position_distance\", \"perimeter_to_area_ratio\",\n        \"lesion_visibility_score\", \"symmetry_border_consistency\", \"color_consistency\",\n        \"size_age_interaction\", \"hue_color_std_interaction\", \"lesion_severity_index\", \n        \"shape_complexity_index\", \"color_contrast_index\", \"log_lesion_area\",\n        \"normalized_lesion_size\", \"mean_hue_difference\", \"std_dev_contrast\",\n        \"color_shape_composite_index\", \"3d_lesion_orientation\", \"overall_color_difference\",\n        \"symmetry_perimeter_interaction\", \"comprehensive_lesion_index\",\n    ]\n    new_cat_cols = [\"combined_anatomical_site\"]\n    \n    # Call our new function for additional features\n    df, additional_num_cols = create_additional_features(df)\n\n    # Update new_num_cols to include additional features\n    \n    new_num_cols.extend(additional_num_cols)\n\n    return df, new_num_cols, new_cat_cols\n\n\ndef get_feature_columns(new_num_cols: list[str], new_cat_cols: list[str]) -> tuple[list[str], list[str]]:\n    \"\"\"Combine base feature columns with newly engineered features.\n\n    Args:\n        new_num_cols (list[str]): List of new numerical column names.\n        new_cat_cols (list[str]): List of new categorical column names.\n\n    Returns:\n        Tuple[list[str], list[str]]: A tuple containing:\n            - List of all numerical column names\n            - List of all categorical column names\n    \"\"\"\n    base_num_cols = [\n        'age_approx', 'clin_size_long_diam_mm', 'tbp_lv_A', 'tbp_lv_Aext', 'tbp_lv_B', 'tbp_lv_Bext', \n        'tbp_lv_C', 'tbp_lv_Cext', 'tbp_lv_H', 'tbp_lv_Hext', 'tbp_lv_L', \n        'tbp_lv_Lext', 'tbp_lv_areaMM2', 'tbp_lv_area_perim_ratio', 'tbp_lv_color_std_mean', \n        'tbp_lv_deltaA', 'tbp_lv_deltaB', 'tbp_lv_deltaL', 'tbp_lv_deltaLB',\n        'tbp_lv_deltaLBnorm', 'tbp_lv_eccentricity', 'tbp_lv_minorAxisMM',\n        'tbp_lv_nevi_confidence', 'tbp_lv_norm_border', 'tbp_lv_norm_color',\n        'tbp_lv_perimeterMM', 'tbp_lv_radial_color_std_max', 'tbp_lv_stdL',\n        'tbp_lv_stdLExt', 'tbp_lv_symm_2axis', 'tbp_lv_symm_2axis_angle',\n        'tbp_lv_x', 'tbp_lv_y', 'tbp_lv_z',\n    ]\n    base_cat_cols = [\"sex\", \"tbp_tile_type\", \"tbp_lv_location\", \"tbp_lv_location_simple\", \"anatom_site_general\"]\n    \n    num_cols = base_num_cols + new_num_cols\n    cat_cols = base_cat_cols + new_cat_cols\n    \n    return num_cols, cat_cols\n\n\ndef handle_missing_values(df: pd.DataFrame, num_cols: list[str]) -> pd.DataFrame:\n    \"\"\"Handle missing and infinite values in numerical columns using robust imputation.\n\n    This function replaces infinity values with NaN, then uses median imputation\n    for missing values. It also includes a fallback to mean imputation if median\n    fails due to all-NaN slices.\n\n    Args:\n        df (pd.DataFrame): Input dataframe.\n        num_cols (list[str]): List of numerical column names.\n\n    Returns:\n        pd.DataFrame: Dataframe with missing and infinite values handled.\n    \"\"\"\n    df = df.copy()  # Create a copy to avoid modifying the original dataframe\n\n    for col in num_cols:\n        # Replace infinity values with NaN\n        df[col] = df[col].replace([np.inf, -np.inf], np.nan)\n\n        # Check if the column has any non-NaN values\n        if df[col].notna().any():\n            # Use median imputation\n            imputer = SimpleImputer(strategy='median')\n            try:\n                df[col] = imputer.fit_transform(df[[col]])\n            except ValueError:\n                # If median imputation fails, fall back to mean imputation\n                print(f\"Warning: Median imputation failed for column {col}. Using mean imputation instead.\")\n                imputer = SimpleImputer(strategy='mean')\n                df[col] = imputer.fit_transform(df[[col]])\n        else:\n            # If all values are NaN, fill with 0 or another appropriate value\n            print(f\"Warning: All values in column {col} are NaN. Filling with 0.\")\n            df[col] = 0\n\n    return df\n\n\ndef encode_categorical_features(df: pd.DataFrame, cat_cols: list[str]) -> tuple[pd.DataFrame, OrdinalEncoder]:\n    \"\"\"Encode categorical features using OrdinalEncoder.\n\n    Args:\n        df (pd.DataFrame): Input dataframe.\n        cat_cols (list[str]): List of categorical column names.\n\n    Returns:\n        tuple[pd.DataFrame, OrdinalEncoder]: A tuple containing:\n            - Dataframe with encoded categorical features\n            - Fitted OrdinalEncoder object\n    \"\"\"\n    category_encoder = OrdinalEncoder(\n        categories='auto',\n        dtype=int,\n        handle_unknown='use_encoded_value',\n        unknown_value=-2,\n        encoded_missing_value=-1,\n    )\n    \n    df[cat_cols] = category_encoder.fit_transform(df[cat_cols])\n    return df, category_encoder\n\n\ndef create_group_kfolds(\n    df: pd.DataFrame,\n    n_splits: int = 5,\n    target_col: str = \"target\",\n    group_col: str = \"patient_id\",\n    random_state: int | None = None\n) -> pd.DataFrame:\n    \"\"\"Create fold assignments for GroupKFold cross-validation.\n\n    This function adds a 'fold' column to the input dataframe, assigning each row\n    to a specific fold while ensuring that all data from the same group (e.g., patient)\n    stays in the same fold.\n\n    Args:\n        df (pd.DataFrame): \n            Input dataframe.\n        n_splits (int, optional): \n            Number of folds.\n        target_col (str, optional): \n            Name of the target column.\n        group_col (str, optional):\n            Name of the column to use for grouping.\n        random_state (int, optional): \n            Random state for reproducibility. \n            If None, the splits will not be shuffled.\n\n    Returns:\n        pd.DataFrame: The input dataframe with an additional 'fold' column.\n\n    Note:\n        If random_state is provided, it will use KFold with shuffle=True instead of GroupKFold,\n        as GroupKFold does not support shuffling.\n    \"\"\"\n    df = df.copy()  # Create a copy to avoid modifying the original dataframe\n    df[\"fold\"] = -1  # Initialize fold column with -1\n\n    if random_state is not None:\n        # Use KFold with shuffling if random_state is provided\n        kf = KFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n        split_method = kf.split(df)\n    else:\n        # Use GroupKFold without shuffling\n        gkf = GroupKFold(n_splits=n_splits)\n        split_method = gkf.split(df, df[target_col], groups=df[group_col])\n\n    # Assign folds\n    for fold, (_, val_idx) in enumerate(split_method):\n        df.loc[val_idx, \"fold\"] = fold\n\n    return df\n\n\n# Get updated dataframes and num/cat cols\ntrain_df, new_num_cols, new_cat_cols = feature_engineering(train_df.copy())\ntest_df, _, _ = feature_engineering(test_df.copy())\n\n# update the feature columns\nnum_cols, cat_cols = get_feature_columns(new_num_cols, new_cat_cols)\n\n# Handle missing values (ordinal)\ntrain_df = handle_missing_values(train_df, num_cols)\ntest_df = handle_missing_values(test_df, num_cols)\n\n# Encode categorical features\ntrain_df, category_encoder = encode_categorical_features(train_df, cat_cols)\ntest_df[cat_cols] = category_encoder.transform(test_df[cat_cols])\n\n# Combine all columns\ntrain_cols = num_cols + cat_cols\n\n# Add fold identification to train dataframe\ntrain_df = create_group_kfolds(train_df, n_splits=5)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-07-06T01:22:01.708207Z","iopub.execute_input":"2024-07-06T01:22:01.708637Z","iopub.status.idle":"2024-07-06T01:22:11.068325Z","shell.execute_reply.started":"2024-07-06T01:22:01.708603Z","shell.execute_reply":"2024-07-06T01:22:11.066732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_lgb_params(random_seed: int | None = None) -> dict[str, Any]:\n    \"\"\"Generate LightGBM parameters using a structured random search approach.\n\n    This function generates a set of LightGBM parameters by randomly selecting values\n    from predefined ranges for each hyperparameter. It provides a balance between\n    exploration of the parameter space and control over the ranges.\n\n    Args:\n        random_seed (int, optional): \n            Seed for random number generator. If None, uses system time.\n\n    Returns:\n        dict[str, Any]: A dictionary of LightGBM parameters.\n    \"\"\"\n    if random_seed is not None:\n        random.seed(random_seed)\n\n    # Define parameter ranges (instead of specific values)\n    param_ranges = {\n        'n_estimators': (1400, 2400),\n        'learning_rate': (0.001, 0.003),\n        'num_leaves': (16, 40),\n        'min_data_in_leaf': (16, 60),\n        'pos_bagging_fraction': (0.74, 0.78),\n        'neg_bagging_fraction': (0.04, 0.08),\n        'feature_fraction': (0.5, 0.78),\n        'lambda_l1': (0.1, 0.4),\n        'lambda_l2': (0.7, 3.0)\n    }\n\n    # Generate random values for each parameter\n    params = {\n        'n_estimators': int(random.uniform(*param_ranges['n_estimators'])),\n        'learning_rate': random.uniform(*param_ranges['learning_rate']),\n        'num_leaves': int(random.uniform(*param_ranges['num_leaves'])),\n        'min_data_in_leaf': int(random.uniform(*param_ranges['min_data_in_leaf'])),\n        'pos_bagging_fraction': random.uniform(*param_ranges['pos_bagging_fraction']),\n        'neg_bagging_fraction': random.uniform(*param_ranges['neg_bagging_fraction']),\n        'feature_fraction': random.uniform(*param_ranges['feature_fraction']),\n        'lambda_l1': random.uniform(*param_ranges['lambda_l1']),\n        'lambda_l2': random.uniform(*param_ranges['lambda_l2'])\n    }\n\n    # Add fixed parameters\n    fixed_params = {\n        'objective': 'binary',\n        'random_state': 42,\n        'bagging_freq': 1,\n        'verbosity': -1,\n        # 'is_unbalance': True\n    }\n    params.update(fixed_params)\n\n    return params\n\n\ndef print_lgb_params(params: dict[str, Any]) -> None:\n    \"\"\"Print the generated LightGBM parameters in a readable format.\n\n    Args:\n        params (dict[str, Any]): \n            Dictionary of LightGBM parameters.\n    \"\"\"\n    clr_print(\"\\nGenerated LightGBM Parameters:\")\n    for key, value in params.items():\n        print(f\"\\t{repr(key):<22} --> {value}\")\n        \nlgb_params = generate_lgb_params(random_seed=4)\nprint_lgb_params(lgb_params)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T02:21:12.733343Z","iopub.execute_input":"2024-07-06T02:21:12.733738Z","iopub.status.idle":"2024-07-06T02:21:12.776039Z","shell.execute_reply.started":"2024-07-06T02:21:12.733706Z","shell.execute_reply":"2024-07-06T02:21:12.775084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We will try to use this loss directly next time...\ndef focal_loss(\n    y_true: np.ndarray, \n    y_pred: np.ndarray, \n    gamma: float = 2.0, \n    alpha: float = 0.25\n) -> tuple[np.ndarray, np.ndarray]:\n    \"\"\"Compute Focal Loss for LightGBM.\n\n    Args:\n        y_true (np.ndarray): True labels.\n        y_pred (np.ndarray): Predicted probabilities.\n        gamma (float): Focusing parameter.\n        alpha (float): Balancing parameter.\n\n    Returns:\n        tuple[np.ndarray, np.ndarray]: \n            Gradient and Hessian.\n    \"\"\"\n    eps = 1e-7\n    y_pred = np.clip(y_pred, eps, 1 - eps)\n    \n    pt = np.where(y_true == 1, y_pred, 1 - y_pred)\n    \n    alpha_t = np.where(y_true == 1, alpha, 1 - alpha)\n    \n    focal_weight = alpha_t * np.power(1 - pt, gamma)\n    \n    gradient = focal_weight * (y_pred - y_true)\n    hessian = focal_weight * (1 - pt) * pt\n    \n    return gradient, hessian\n\n\ndef perform_lightgbm_cv(\n    df: pd.DataFrame,\n    train_cols: list[str],\n    target_col: str,\n    fold_col: str,\n    lgb_params: dict[str, Any],\n    n_folds: int = 5\n) -> tuple[list[float], list[lgb.LGBMRegressor]]:\n    \"\"\"Perform cross-validation using LightGBM and compute scores.\n\n    Args:\n        df (pd.DataFrame): \n            The full training dataframe.\n        train_cols (list[str]): \n            List of column names to use for training.\n        target_col (str): \n            Name of the target column.\n        fold_col (str): \n            Name of the column containing fold assignments.\n        lgb_params (dict[str, Any]): \n            Parameters for LightGBM model.\n        n_folds (int, optional): \n            Number of folds for cross-validation.\n\n    Returns:\n        tuple[list[float], list[lgb.LGBMRegressor]]: \n            A tuple containing:\n                - List of scores for each fold\n                - List of trained LightGBM models\n    \"\"\"\n    # Initialize\n    scores, models, _df = [], [], df.copy()    \n    for fold in range(n_folds):\n        # Split data into train and validation sets\n        _train_df = _df[_df[fold_col] != fold].reset_index(drop=True)\n        _val_df = _df[_df[fold_col] == fold].reset_index(drop=True)\n\n        # Initialize and train the model\n        model = lgb.LGBMRegressor(**lgb_params)\n        model.fit(_train_df[train_cols], _train_df[target_col])\n\n        # Make predictions\n        preds = model.predict(_val_df[train_cols])\n\n        # Prepare DataFrames for scoring\n        true_df = _val_df[['isic_id', target_col]].copy()\n        pred_df = pd.DataFrame({'isic_id': _val_df['isic_id'], 'prediction': preds})\n\n        # Compute score with 'score' function\n        _score = score(true_df, pred_df, \"isic_id\")\n        clr_print(f\"Fold: {fold} - Score: {_score:.5f}\")\n\n        scores.append(_score)\n        models.append(model)\n\n    print(f\"\\nMean Score: {sum(scores) / len(scores):.5f}\")\n    print(f\"Std Dev of Score: {pd.Series(scores).std():.5f}\")\n    return scores, models\n\n\ndef perform_lightgbm_cv_with_partial_smote(\n    df: pd.DataFrame,\n    train_cols: list[str],\n    target_col: str,\n    fold_col: str,\n    lgb_params: dict[str, Any],\n    n_folds: int = 5,\n    smote_ratio: float = 0.025,\n    random_state: int = 42\n) -> tuple[list[float], list[lgb.LGBMRegressor]]:\n    \"\"\"Perform cross-validation using LightGBM with partial SMOTE and class weights.\n\n    Args:\n        df (pd.DataFrame): The full training dataframe.\n        train_cols (list[str]): List of column names to use for training.\n        target_col (str): Name of the target column.\n        fold_col (str): Name of the column containing fold assignments.\n        lgb_params (dict[str, Any]): Parameters for LightGBM model.\n        n_folds (int, optional): Number of folds for cross-validation.\n        smote_ratio (float, optional): Desired ratio of minority to majority class after SMOTE.\n        random_state (int, optional): Random state for reproducibility.\n\n    Returns:\n        tuple[list[float], list[lgb.LGBMRegressor]]: A tuple containing:\n            - List of scores for each fold\n            - List of trained LightGBM models\n    \"\"\"\n    scores, models = [], []\n    _df = df.copy()\n\n    # Calculate class weights\n    class_weights = {0: 1, 1: len(df[df[target_col] == 0]) / len(df[df[target_col] == 1])}\n    lgb_params['class_weight'] = class_weights\n\n    # Initialize SMOTE with the desired ratio\n    smote = SMOTE(sampling_strategy=smote_ratio, random_state=random_state)\n    \n    for fold in range(n_folds):\n        train_df = _df[_df[fold_col] != fold].reset_index(drop=True)\n        val_df = _df[_df[fold_col] == fold].reset_index(drop=True)\n\n        X_train, y_train = train_df[train_cols], train_df[target_col]\n\n        # Apply partial SMOTE\n        X_resampled, y_resampled = smote.fit_resample(X_train, y_train)\n\n        X_resampled_df = pd.DataFrame(X_resampled, columns=train_cols)\n\n        model = lgb.LGBMRegressor(**lgb_params)\n        model.fit(X_resampled_df, y_resampled)\n\n        X_val = val_df[train_cols]\n        preds = model.predict(X_val)\n\n        true_df = val_df[['isic_id', target_col]].copy()\n        pred_df = pd.DataFrame({'isic_id': val_df['isic_id'], 'prediction': preds})\n\n        _score = score(true_df, pred_df, \"isic_id\")\n        print(f\"Fold: {fold} - Score: {_score:.5f}\")\n\n        scores.append(_score)\n        models.append(model)\n\n    print(f\"\\nMean Score: {np.mean(scores):.5f}\")\n    print(f\"Std Dev of Score: {np.std(scores):.5f}\")\n\n    return scores, models\n\n\n## ... Mean of ~0.15 on run w/ original parameters... ###\nscores, models = perform_lightgbm_cv(\n    df=train_df,\n    train_cols=train_cols,\n    target_col=\"target\",\n    fold_col=\"fold\",\n    lgb_params=lgb_params,\n)\n\n# ### ... Mean of  on run w/ same parameters as below... ###\n# scores, models = perform_lightgbm_cv_with_partial_smote(\n#     df=train_df,\n#     train_cols=train_cols,\n#     target_col=\"target\",\n#     fold_col=\"fold\",\n#     lgb_params=lgb_params,\n# )","metadata":{"execution":{"iopub.status.busy":"2024-07-06T02:21:25.961180Z","iopub.execute_input":"2024-07-06T02:21:25.961625Z","iopub.status.idle":"2024-07-06T02:31:02.075459Z","shell.execute_reply.started":"2024-07-06T02:21:25.961586Z","shell.execute_reply":"2024-07-06T02:31:02.074166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_feature_importance(\n    models: list[lgb.LGBMRegressor], \n    top_n: int = 30,\n    template_theme: str = \"plotly_white\",\n) -> None:\n    \"\"\"Create an interactive bar chart of feature importances with text labels.\n\n    This function calculates the mean feature importance across multiple \n    LightGBM models and plots the top N most important features with text labels.\n\n    Args:\n        models (list[lgb.LGBMRegressor]): \n            List of trained LightGBM models.\n        top_n (int, optional): \n            Number of top features to display. \n        template_theme (str, optional):\n            The plotly theme to use for plotly chart. .\n\n    Returns:\n        None; \n            Displays the Plotly chart.\n    \"\"\"\n    # Calculate mean feature importance\n    importances = np.mean([model.feature_importances_ for model in models], axis=0)\n    feature_names = models[0].feature_name_\n\n    # Create DataFrame and sort by importance\n    df_imp = pd.DataFrame({\"feature\": feature_names, \"importance\": importances})\n    df_imp = df_imp.sort_values(\"importance\", ascending=True).tail(top_n)\n    \n    #     for i, _c in enumerate(df_imp[\"feature\"].values):\n    #         if _c in [\"color_variance_ratio\", \"border_color_interaction\", \"size_color_contrast_ratio\", \n    #                   \"age_normalized_nevi_confidence\", \"color_asymmetry_index\", \"3d_volume_approximation\", \n    #                   \"color_range\", \"shape_color_consistency\", \"border_length_ratio\", \"age_size_symmetry_index\"]:\n    #             print(i, _c)\n    \n    # Create Plotly bar chart\n    fig = go.Figure(go.Bar(\n        y=df_imp[\"feature\"],\n        x=df_imp[\"importance\"],\n        orientation='h',\n        marker=dict(\n            color=df_imp[\"importance\"],\n            colorscale='Magma',\n            colorbar=dict(title=\"Importance\")\n        ),\n        text=df_imp[\"importance\"].apply(lambda x: f\"{x:.1f}\"),  # Format importance values\n        textposition='outside',  # Position text outside of bars\n        textfont=dict(size=10),  # Adjust text size as needed\n    ))\n\n    # Update layout for better readability\n    fig.update_layout(\n        title={\n            'text': f'<b>Top {top_n} Feature Importances</b>',\n            'y':0.95, 'x':0.5, 'xanchor': 'center', 'yanchor': 'top'\n        },\n        xaxis_title=\"<b>Importance</b>\",\n        yaxis_title=\"<b>Features</b>\",\n        height=1500,\n        width=1200,\n        yaxis={'categoryorder':'total ascending'},\n        template=template_theme,\n        margin=dict(l=200, r=200),  # Increase left and right margins for text labels\n        xaxis=dict(range=[0, df_imp[\"importance\"].max() * 1.1])  # Extend x-axis range for text labels\n    )\n\n    # Show the plot\n    fig.show()\n\n    \nplot_feature_importance(models, top_n=20)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T02:34:42.412274Z","iopub.execute_input":"2024-07-06T02:34:42.412757Z","iopub.status.idle":"2024-07-06T02:34:42.478369Z","shell.execute_reply.started":"2024-07-06T02:34:42.412723Z","shell.execute_reply":"2024-07-06T02:34:42.477151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_with_ensemble(\n    models: list[lgb.LGBMRegressor],\n    df: pd.DataFrame,\n    train_cols: list[str],\n    aggregation_method: str = 'mean'\n) -> pd.DataFrame:\n    \"\"\"\n    Make predictions on data using an ensemble of LightGBM models.\n\n    Args:\n        models (list[lgb.LGBMRegressor]): \n            List of trained LightGBM models.\n        df (pd.DataFrame): \n            Dataframe to predict on.\n        train_cols (list[str]): \n            List of column names used for training.\n        aggregation_method (str, optional): \n            Method to aggregate predictions ('mean' or 'median'). \n    \n    Raises:\n        ValueError:\n            If the aggregation method provided is invalid.\n    \n    Returns:\n        pd.DataFrame: \n            DataFrame with 'isic_id' and aggregated 'target' columns.\n    \"\"\"\n    _df = df.copy()\n    \n    # Make predictions with each model\n    all_predictions = [model.predict(_df[train_cols]) for model in models]\n    \n    # Aggregate predictions\n    if aggregation_method == 'mean':\n        final_predictions = np.mean(all_predictions, axis=0)\n    elif aggregation_method == 'median':\n        final_predictions = np.median(all_predictions, axis=0)\n    else:\n        raise ValueError(\"Invalid aggregation method. Choose 'mean' or 'median'.\")\n    \n    # Create result DataFrame\n    result_df = pd.DataFrame({\n        'isic_id': _df['isic_id'],\n        'target': final_predictions\n    })\n    \n    return result_df\n\ntest_predictions = predict_with_ensemble(\n    models=models,  \n    df=test_df,\n    train_cols=train_cols,\n    aggregation_method='mean'\n)\ntest_predictions.to_csv('submission.csv', index=False)\ndisplay(test_predictions)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T02:35:35.320861Z","iopub.execute_input":"2024-07-06T02:35:35.321349Z","iopub.status.idle":"2024-07-06T02:35:35.337581Z","shell.execute_reply.started":"2024-07-06T02:35:35.321312Z","shell.execute_reply":"2024-07-06T02:35:35.335861Z"},"trusted":true},"execution_count":null,"outputs":[]}]}