{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install fastai -Uqq > /dev/null","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:23:37.309782Z","iopub.execute_input":"2023-07-22T03:23:37.310136Z","iopub.status.idle":"2023-07-22T03:23:55.938839Z","shell.execute_reply.started":"2023-07-22T03:23:37.310105Z","shell.execute_reply":"2023-07-22T03:23:55.937403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the dependancies","metadata":{}},{"cell_type":"code","source":"from fastai.basics import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\n\nimport pydicom\nimport seaborn as sns\n\nimport numpy as np\nimport pandas as pd\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:24:02.109098Z","iopub.execute_input":"2023-07-22T03:24:02.109642Z","iopub.status.idle":"2023-07-22T03:24:19.239497Z","shell.execute_reply.started":"2023-07-22T03:24:02.109596Z","shell.execute_reply":"2023-07-22T03:24:19.238429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import data source","metadata":{}},{"cell_type":"code","source":"melanoma_source = Path(\"../input/siim-isic-melanoma-classification\")\nmelanoma_source.ls()","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:24:24.132509Z","iopub.execute_input":"2023-07-22T03:24:24.133586Z","iopub.status.idle":"2023-07-22T03:24:24.148534Z","shell.execute_reply.started":"2023-07-22T03:24:24.133535Z","shell.execute_reply":"2023-07-22T03:24:24.147536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = melanoma_source/'train'\ntrain_files = get_dicom_files(train)\nlen(train_files)","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:24:43.074817Z","iopub.execute_input":"2023-07-22T03:24:43.075346Z","iopub.status.idle":"2023-07-22T03:25:32.622687Z","shell.execute_reply.started":"2023-07-22T03:24:43.075293Z","shell.execute_reply":"2023-07-22T03:25:32.621614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lets see what information is contained within each DICOM file","metadata":{}},{"cell_type":"code","source":"patient = 7\nmelanoma_sample = train_files[patient].dcmread()\nmelanoma_sample","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:25:42.884464Z","iopub.execute_input":"2023-07-22T03:25:42.884939Z","iopub.status.idle":"2023-07-22T03:25:42.954230Z","shell.execute_reply.started":"2023-07-22T03:25:42.884901Z","shell.execute_reply":"2023-07-22T03:25:42.953320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"melanoma_sample.pixel_array.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:26:02.208323Z","iopub.execute_input":"2023-07-22T03:26:02.209092Z","iopub.status.idle":"2023-07-22T03:26:02.698199Z","shell.execute_reply.started":"2023-07-22T03:26:02.209052Z","shell.execute_reply":"2023-07-22T03:26:02.697248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Find out image shape distribution","metadata":{}},{"cell_type":"code","source":"# Load dataset\ndata_dir = melanoma_source/'train'\n","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:31:50.827685Z","iopub.execute_input":"2023-07-22T03:31:50.828256Z","iopub.status.idle":"2023-07-22T03:31:50.833677Z","shell.execute_reply.started":"2023-07-22T03:31:50.828216Z","shell.execute_reply":"2023-07-22T03:31:50.832546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({\n    'image':(\n        *get_dicom_files(data_dir),\n    )\n})\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:32:07.171949Z","iopub.execute_input":"2023-07-22T03:32:07.172393Z","iopub.status.idle":"2023-07-22T03:32:14.282303Z","shell.execute_reply.started":"2023-07-22T03:32:07.172357Z","shell.execute_reply":"2023-07-22T03:32:14.281126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the parent folder name of each image path\nimage_paths = df[\"image\"].tolist()\nparent_folder_names = []\nfor image_path in image_paths:\n    parent_folder_name = os.path.dirname(image_path)\n    label_value = parent_folder_name.split(\"/\")[-1]\n    parent_folder_names.append(label_value)\n\n# Add the \"label\" column to the DataFrame\ndf[\"label\"] = parent_folder_names\n\n# Print the DataFrame\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:32:34.213876Z","iopub.execute_input":"2023-07-22T03:32:34.214345Z","iopub.status.idle":"2023-07-22T03:32:34.460090Z","shell.execute_reply.started":"2023-07-22T03:32:34.214304Z","shell.execute_reply":"2023-07-22T03:32:34.459028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['size'] = df.apply(\n    lambda x: x.image.dcmread().pixel_array.shape, 'columns'\n)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:40:32.205109Z","iopub.execute_input":"2023-07-22T06:40:32.205489Z","iopub.status.idle":"2023-07-22T06:40:32.611362Z","shell.execute_reply.started":"2023-07-22T06:40:32.205457Z","shell.execute_reply":"2023-07-22T06:40:32.610092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('img_meta.csv', index=False) #save it for future use","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.value_counts('label')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w, h, c = list(zip(*df['size'].values)) # Unzip list into width and height using zip() and *\n\nsizesdata = pd.DataFrame({\n    'width':w, 'height':h, \n    'label':df['label'].values\n})\n\nfig = px.scatter(\n    sizesdata, x = w, y = h, \n    color = 'label', \n    labels={'x':'Width', 'y':'Height', 'label':'Diagnosis'}, \n    height = 550, width = 700, \n    title = 'Distribution of Image Width and Height', \n    marginal_x = 'histogram', marginal_y = 'histogram'\n)\nfig.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_one_patient(file):\n    \"\"\" function to view patient image and choosen tags within the head of the DICOM\"\"\"\n    pat = file.dcmread()\n    print(f'patient Name: {pat.PatientName}')\n    print(f'Patient ID: {pat.PatientID}')\n    print(f'Patient age: {pat.PatientAge}')\n    print(f'Patient Sex: {pat.PatientSex}')\n    print(f'Body part: {pat.BodyPartExamined}')\n    trans = Transform(Resize(256))\n    dicom_create = PILDicom.create(file)\n    dicom_transform = trans(dicom_create)\n    return show_image(dicom_transform)\n\npatient = 7\nmelanoma_sample = train_files[patient]\nshow_one_patient(melanoma_sample)","metadata":{"execution":{"iopub.status.busy":"2023-07-22T03:26:08.628865Z","iopub.execute_input":"2023-07-22T03:26:08.629347Z","iopub.status.idle":"2023-07-22T03:26:09.962754Z","shell.execute_reply.started":"2023-07-22T03:26:08.629291Z","shell.execute_reply":"2023-07-22T03:26:09.961825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These images are stored in YBR_FULL_422 color space and this is stated in the following tag:\n\n(0028, 0004) Photometric Interpretation CS: 'YBR_FULL_422'\n\nTo view the images as they are intended the color space needs to be converted from YBR_FULL_422 to RGB. Pydicom provides a means of converting from one color space to another by using convert_color_space where it takes the (pixel array, current color space, desired color space) as attributes. This is done by acessing the pixel_array and then converting to the desired color space","metadata":{}},{"cell_type":"code","source":"from pydicom.pixel_data_handlers.util import convert_color_space","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:24:00.428476Z","iopub.execute_input":"2023-07-19T13:24:00.429289Z","iopub.status.idle":"2023-07-19T13:24:00.434337Z","shell.execute_reply.started":"2023-07-19T13:24:00.429252Z","shell.execute_reply":"2023-07-19T13:24:00.433274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"melanoma_sample_dimg = train_files[patient].dcmread()\narr = melanoma_sample_dimg.pixel_array\nconvert = convert_color_space(arr, 'YBR_FULL_422', 'RGB')\nshow_image(convert)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:26:03.552111Z","iopub.execute_input":"2023-07-19T13:26:03.552568Z","iopub.status.idle":"2023-07-19T13:26:09.043952Z","shell.execute_reply.started":"2023-07-19T13:26:03.552513Z","shell.execute_reply":"2023-07-19T13:26:09.042968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"markdown","source":"### Load .csv","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(melanoma_source/'train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:28:10.802386Z","iopub.execute_input":"2023-07-19T13:28:10.803445Z","iopub.status.idle":"2023-07-19T13:28:10.901219Z","shell.execute_reply.started":"2023-07-19T13:28:10.803404Z","shell.execute_reply":"2023-07-19T13:28:10.900215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot 3 comparisons\ndef plot_comparison3(df, feature, feature1, feature2):\n    \"Plot 3 comparisons from a dataframe\"\n    fig, (ax1, ax2, ax3) = plt.subplots(1,3, figsize = (16, 4))\n#     s1 = sns.countplot(df[feature], ax=ax1)\n#     s1.set_title(feature)\n    s2 = sns.countplot(df[feature1], ax=ax2)\n    s2.set_title(feature1)\n    s3 = sns.countplot(df[feature2], ax=ax3)\n    s3.set_title(feature2)\n    plt.show()\n\nplot_comparison3(df, 'sex', 'age_approx', 'benign_malignant')","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:32:46.696930Z","iopub.execute_input":"2023-07-19T13:32:46.697416Z","iopub.status.idle":"2023-07-19T13:32:48.725468Z","shell.execute_reply.started":"2023-07-19T13:32:46.697371Z","shell.execute_reply":"2023-07-19T13:32:48.724008Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Remove features","metadata":{}},{"cell_type":"code","source":"eda_df = df[['sex','age_approx','anatom_site_general_challenge','diagnosis','target']]\neda_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:34:13.970647Z","iopub.execute_input":"2023-07-19T13:34:13.971669Z","iopub.status.idle":"2023-07-19T13:34:13.990724Z","shell.execute_reply.started":"2023-07-19T13:34:13.971627Z","shell.execute_reply":"2023-07-19T13:34:13.989594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process NaN","metadata":{}},{"cell_type":"code","source":"sex_count = eda_df['sex'].isna().sum() \nage_count = eda_df['age_approx'].isna().sum()\nanatom_count = eda_df['anatom_site_general_challenge'].isna().sum()\nprint(f'Nan values in sex column: {sex_count},\\\n      age column: {age_count},\\\n      anatom count: {anatom_count}')","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:36:07.453806Z","iopub.execute_input":"2023-07-19T13:36:07.454265Z","iopub.status.idle":"2023-07-19T13:36:07.470787Z","shell.execute_reply.started":"2023-07-19T13:36:07.454231Z","shell.execute_reply":"2023-07-19T13:36:07.469604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_drop = eda_df.dropna()\nlen(df_drop)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:36:31.530863Z","iopub.execute_input":"2023-07-19T13:36:31.531815Z","iopub.status.idle":"2023-07-19T13:36:31.569943Z","shell.execute_reply.started":"2023-07-19T13:36:31.531776Z","shell.execute_reply":"2023-07-19T13:36:31.568653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\nedaa_df = eda_df.apply(\n    lambda col: le.fit_transform(col.astype(str)),\n    axis=0, result_type='expand'\n)\nedaa_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:38:22.314755Z","iopub.execute_input":"2023-07-19T13:38:22.315244Z","iopub.status.idle":"2023-07-19T13:38:22.420495Z","shell.execute_reply.started":"2023-07-19T13:38:22.315210Z","shell.execute_reply":"2023-07-19T13:38:22.419568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\nsns.set_context(\"paper\")\nsns.pairplot(eda_df, hue=\"target\", height=5, aspect=2, palette='gist_rainbow_r')","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:39:34.172860Z","iopub.execute_input":"2023-07-19T13:39:34.173318Z","iopub.status.idle":"2023-07-19T13:39:35.941931Z","shell.execute_reply.started":"2023-07-19T13:39:34.173285Z","shell.execute_reply":"2023-07-19T13:39:35.940668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting the data ready for training","metadata":{}},{"cell_type":"code","source":"train_files[patient]","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:41:09.836949Z","iopub.execute_input":"2023-07-19T13:41:09.837348Z","iopub.status.idle":"2023-07-19T13:41:09.843963Z","shell.execute_reply.started":"2023-07-19T13:41:09.837316Z","shell.execute_reply":"2023-07-19T13:41:09.843001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_x = lambda x: melanoma_source/'train'/f'{x[0]}.dcm'","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:44:34.662202Z","iopub.execute_input":"2023-07-19T13:44:34.662667Z","iopub.status.idle":"2023-07-19T13:44:34.668247Z","shell.execute_reply.started":"2023-07-19T13:44:34.662631Z","shell.execute_reply":"2023-07-19T13:44:34.666607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_y = ColReader('target')","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:44:37.018071Z","iopub.execute_input":"2023-07-19T13:44:37.018472Z","iopub.status.idle":"2023-07-19T13:44:37.023458Z","shell.execute_reply.started":"2023-07-19T13:44:37.018438Z","shell.execute_reply":"2023-07-19T13:44:37.022251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Getting some quick `batch_tfms`","metadata":{}},{"cell_type":"code","source":"batch_tfms = aug_transforms(flip_vert=True, max_lighting=0.1, max_zoom=1.05, max_warp=0.)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:44:41.196884Z","iopub.execute_input":"2023-07-19T13:44:41.197944Z","iopub.status.idle":"2023-07-19T13:44:41.207310Z","shell.execute_reply.started":"2023-07-19T13:44:41.197897Z","shell.execute_reply":"2023-07-19T13:44:41.206245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blocks = (ImageBlock(cls=PILDicom), CategoryBlock)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:46:17.223773Z","iopub.execute_input":"2023-07-19T13:46:17.224208Z","iopub.status.idle":"2023-07-19T13:46:17.233082Z","shell.execute_reply.started":"2023-07-19T13:46:17.224174Z","shell.execute_reply":"2023-07-19T13:46:17.231977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"melanoma = DataBlock(\n    blocks = blocks,\n    get_x = get_x,\n    splitter = RandomSplitter(),\n    item_tfms = Resize(128),\n    get_y = ColReader('target'),\n    batch_tfms = batch_tfms\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:46:42.739884Z","iopub.execute_input":"2023-07-19T13:46:42.740282Z","iopub.status.idle":"2023-07-19T13:46:42.750084Z","shell.execute_reply.started":"2023-07-19T13:46:42.740248Z","shell.execute_reply":"2023-07-19T13:46:42.748795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = melanoma.dataloaders(df, bs=32)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:59:39.062369Z","iopub.execute_input":"2023-07-19T13:59:39.062820Z","iopub.status.idle":"2023-07-19T13:59:39.375817Z","shell.execute_reply.started":"2023-07-19T13:59:39.062785Z","shell.execute_reply":"2023-07-19T13:59:39.374707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = dls.cuda()\ndls.show_batch(max_n=12, nrows=2, ncols=6)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T13:59:44.766514Z","iopub.execute_input":"2023-07-19T13:59:44.766944Z","iopub.status.idle":"2023-07-19T14:00:00.339467Z","shell.execute_reply.started":"2023-07-19T13:59:44.766909Z","shell.execute_reply":"2023-07-19T14:00:00.337938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc = RocAuc()\ndls.c","metadata":{"execution":{"iopub.status.busy":"2023-07-19T14:00:11.195875Z","iopub.execute_input":"2023-07-19T14:00:11.196271Z","iopub.status.idle":"2023-07-19T14:00:11.203526Z","shell.execute_reply.started":"2023-07-19T14:00:11.196236Z","shell.execute_reply":"2023-07-19T14:00:11.202512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xresnet18_deeper(n_out = dls.c)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T14:00:16.139682Z","iopub.execute_input":"2023-07-19T14:00:16.140065Z","iopub.status.idle":"2023-07-19T14:00:16.343117Z","shell.execute_reply.started":"2023-07-19T14:00:16.140033Z","shell.execute_reply":"2023-07-19T14:00:16.342046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(77)\nlearn = Learner(\n    dls, model, \n    opt_func = ranger,\n    loss_func = LabelSmoothingCrossEntropy(),\n    metrics = [accuracy],\n    cbs = ShowGraphCallback()\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T14:00:21.260525Z","iopub.execute_input":"2023-07-19T14:00:21.260936Z","iopub.status.idle":"2023-07-19T14:00:21.269581Z","shell.execute_reply.started":"2023-07-19T14:00:21.260903Z","shell.execute_reply":"2023-07-19T14:00:21.268563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\nlearn.freeze()\nlearn.fit_one_cycle(1, 5e-2)","metadata":{"execution":{"iopub.status.busy":"2023-07-19T21:41:14.585354Z","iopub.execute_input":"2023-07-19T21:41:14.586487Z","iopub.status.idle":"2023-07-19T21:41:15.034985Z","shell.execute_reply.started":"2023-07-19T21:41:14.586440Z","shell.execute_reply":"2023-07-19T21:41:15.033631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}