{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":184483960,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"We will now start with Daat handing:\n**Steps**\n<li>.Data preperation\n<li>Model building\n<li>Evluation and optimisation\n\n    \n    \n**Steps for Data preperation**\n    <li>Data Review and Cleaning\n     <li>Data Normalization and Resizing\n     <li>Data Augmentatio\n\n**Starting with Data review and cleaning :**\n<li>Review Dataset: Ensure all required columns (image file paths, labels, coordinates) are present and correctly formatted.\n<li>Handle Missing Data: Check for and handle any missing or corrupt data.\n    \n    We will use code from train_label_coordinates_data dataset from RSNA_img_display notebook. (we have imported this in the input panel on right)\n\n","metadata":{}},{"cell_type":"markdown","source":"> **Configure notebook to use GP**\n\nEnsure that tensorflow uses GPU\n\n> Check if tensorflow is having access to GPU","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nif (len(tf.config.list_physical_devices('GPU'))==1):\n    print(\"TensorFlow has access to the GPU.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T06:07:07.282127Z","iopub.execute_input":"2024-06-20T06:07:07.282502Z","iopub.status.idle":"2024-06-20T06:07:07.287615Z","shell.execute_reply.started":"2024-06-20T06:07:07.282458Z","shell.execute_reply":"2024-06-20T06:07:07.286787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Ensure that Tensorflow uses GPU","metadata":{}},{"cell_type":"code","source":"gpus=tf.config.list_physical_devices('GPU')\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n            print(\"GPU is available and configured sucesssfully\")\n    except RuntimeError as e:\n        print(e)\nelse:\n    print('No GPUS found!')","metadata":{"execution":{"iopub.status.busy":"2024-06-20T06:07:10.612322Z","iopub.execute_input":"2024-06-20T06:07:10.613174Z","iopub.status.idle":"2024-06-20T06:07:10.619767Z","shell.execute_reply.started":"2024-06-20T06:07:10.613130Z","shell.execute_reply":"2024-06-20T06:07:10.618821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Run a testcode to see if Tensorflow is using GPU:","metadata":{}},{"cell_type":"code","source":"with tf.device('/GPU:0'):\n    a = tf.constant([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]])\n    b = tf.constant([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]])\n    c = tf.matmul(a, b, transpose_b=True)  # Correct the matrix dimensions for matmul\n    print(c.numpy())","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:10.066053Z","iopub.execute_input":"2024-06-20T05:30:10.066417Z","iopub.status.idle":"2024-06-20T05:30:10.484197Z","shell.execute_reply.started":"2024-06-20T05:30:10.066372Z","shell.execute_reply":"2024-06-20T05:30:10.483219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Import librarries and files:","metadata":{}},{"cell_type":"code","source":"#Now define a the input directoy path:\nstart_dir_path=\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n#import relevant libraries:\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt #For ploting and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\nimport os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. \nimport ipywidgets as widgets\nfrom IPython.display import display\nimport logging\nfrom multiprocessing import Pool\n\n# Assign file path to simple variables:\n# CSV fiels provided and thier paths:\ntrain_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\nsample_csv_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv'\ntrain_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv'\ntrain_label_coordinates_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntest_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv'\n\n# Image folder paths:\ntrain_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images'\ntest_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'\n\n\n# Read csv files:\nsample_csv=pd.read_csv(sample_csv_path)\ntrain_data=pd.read_csv(train_data_path)\ntrain_series_description=pd.read_csv(train_series_descriptions_path)\ntest_series_description=pd.read_csv(test_series_descriptions_path)\ntrain_label_coordinates_data=pd.read_csv(train_label_coordinates_data_path)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:10.486631Z","iopub.execute_input":"2024-06-20T05:30:10.486926Z","iopub.status.idle":"2024-06-20T05:30:11.797435Z","shell.execute_reply.started":"2024-06-20T05:30:10.486900Z","shell.execute_reply":"2024-06-20T05:30:11.796291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Trun f warnings!\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:11.798875Z","iopub.execute_input":"2024-06-20T05:30:11.799258Z","iopub.status.idle":"2024-06-20T05:30:11.804082Z","shell.execute_reply.started":"2024-06-20T05:30:11.799226Z","shell.execute_reply":"2024-06-20T05:30:11.803042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Data Preperation**","metadata":{}},{"cell_type":"code","source":"# Add a new column to store the categories in the train_labels_coordinates_data dataset.\ntrain_label_coordinates_data[\"category\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_label_coordinates_data.iterrows():\n    r = row['study_id']\n    col = (row['condition'].lower().replace(' ', '_')) + '_' + (row['level'].lower().replace('/', '_'))\n    \n    # Ensure the column exists in train_data to avoid KeyError\n    if col in train_data.columns:\n        # Check if there is a matching study_id and the column exists\n        value = train_data.loc[train_data['study_id'] == r, col].values\n        if len(value) > 0:\n            train_label_coordinates_data.at[idx, \"category\"] = value[0]\n        else:\n            train_label_coordinates_data.at[idx, \"category\"] = None\n    else:\n        train_label_coordinates_data.at[idx, \"category\"] = None\n\n# Display the first 5 rows of the updated DataFrame\ntrain_label_coordinates_data.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:11.805456Z","iopub.execute_input":"2024-06-20T05:30:11.806277Z","iopub.status.idle":"2024-06-20T05:30:33.811794Z","shell.execute_reply.started":"2024-06-20T05:30:11.806242Z","shell.execute_reply":"2024-06-20T05:30:33.810803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we need to create a list of image file paths :\n# For this, let us first create a new df containing only the first three columns. All we need to do is to drop rest of the columns.\nfirst_three_cols=train_label_coordinates_data.iloc[:,:3]\n\n\n# Now all we have to do is to convert this df into a new df that contains combined values of the each row.\njoin_row_as_str=lambda row: train_images_path+\"/\"+\"/\".join(row.astype(str))+\".dcm\"\ntrain_label_images_path=first_three_cols.apply(join_row_as_str,axis=1)\n\n#check for the integrity of the new df\nprint(f'The number of rows in train_label_images_path and train_label_coordinates_data are equal? {train_label_images_path.shape[0]==train_label_coordinates_data.shape[0]}')\n\n#Display the shape of the updated dataset:\ntrain_label_coordinates_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:33.813194Z","iopub.execute_input":"2024-06-20T05:30:33.813578Z","iopub.status.idle":"2024-06-20T05:30:36.976015Z","shell.execute_reply.started":"2024-06-20T05:30:33.813546Z","shell.execute_reply":"2024-06-20T05:30:36.975084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add img file patth as column in train_label_coordinates_data\n# Add a new column to store the categories\ntrain_label_coordinates_data[\"img_file_path\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_label_coordinates_data.iterrows():\n    file_path = train_label_images_path.loc[idx]\n    train_label_coordinates_data.at[idx, \"img_file_path\"] = file_path\n    \n# Display the first row of the updated DataFrame\ntrain_label_coordinates_data.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:36.977418Z","iopub.execute_input":"2024-06-20T05:30:36.977779Z","iopub.status.idle":"2024-06-20T05:30:41.091691Z","shell.execute_reply.started":"2024-06-20T05:30:36.977747Z","shell.execute_reply":"2024-06-20T05:30:41.090775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Review and cleaning:**","metadata":{}},{"cell_type":"markdown","source":"#  **DATA REVVIEW**\n","metadata":{}},{"cell_type":"code","source":"# Run functions already written in rsna_funnctions script - this will impport several functions:\n\n%run '/kaggle/usr/lib/rsna_functons/rsna_functons.py'","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.092667Z","iopub.execute_input":"2024-06-20T05:30:41.092941Z","iopub.status.idle":"2024-06-20T05:30:41.144912Z","shell.execute_reply.started":"2024-06-20T05:30:41.092918Z","shell.execute_reply":"2024-06-20T05:30:41.144124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_dicom_images(1,train_label_coordinates_data)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.148879Z","iopub.execute_input":"2024-06-20T05:30:41.149372Z","iopub.status.idle":"2024-06-20T05:30:41.434122Z","shell.execute_reply.started":"2024-06-20T05:30:41.149347Z","shell.execute_reply":"2024-06-20T05:30:41.433208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_image_slider(train_label_coordinates_data)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.435219Z","iopub.execute_input":"2024-06-20T05:30:41.435501Z","iopub.status.idle":"2024-06-20T05:30:41.750781Z","shell.execute_reply.started":"2024-06-20T05:30:41.435477Z","shell.execute_reply":"2024-06-20T05:30:41.749950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# call the function print_dir_str()\nprint_dir_str(start_dir_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.752042Z","iopub.execute_input":"2024-06-20T05:30:41.752422Z","iopub.status.idle":"2024-06-20T05:30:41.760370Z","shell.execute_reply.started":"2024-06-20T05:30:41.752371Z","shell.execute_reply":"2024-06-20T05:30:41.759350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coordinates_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-20T06:07:17.025100Z","iopub.execute_input":"2024-06-20T06:07:17.025994Z","iopub.status.idle":"2024-06-20T06:07:17.040336Z","shell.execute_reply.started":"2024-06-20T06:07:17.025959Z","shell.execute_reply":"2024-06-20T06:07:17.039281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coordinates_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.776799Z","iopub.execute_input":"2024-06-20T05:30:41.777090Z","iopub.status.idle":"2024-06-20T05:30:41.786203Z","shell.execute_reply.started":"2024-06-20T05:30:41.777067Z","shell.execute_reply":"2024-06-20T05:30:41.785178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Change the name of dataset from current 'train_label_coordinates_data' to \"train_data\" for convinience","metadata":{}},{"cell_type":"code","source":"train_data = train_label_coordinates_data\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.787417Z","iopub.execute_input":"2024-06-20T05:30:41.787723Z","iopub.status.idle":"2024-06-20T05:30:41.795338Z","shell.execute_reply.started":"2024-06-20T05:30:41.787693Z","shell.execute_reply":"2024-06-20T05:30:41.794558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Check for missing or corrupt data:","metadata":{}},{"cell_type":"code","source":"# Check for missing values\ndisplay(train_data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.796365Z","iopub.execute_input":"2024-06-20T05:30:41.796665Z","iopub.status.idle":"2024-06-20T05:30:41.827296Z","shell.execute_reply.started":"2024-06-20T05:30:41.796642Z","shell.execute_reply":"2024-06-20T05:30:41.826442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Visualise the missing categories to get some idea what is missing :","metadata":{}},{"cell_type":"code","source":"# Display rows where the \"category\" column has missing values\nmissing_category_entries = train_data[train_data['category'].isnull()]\n\ndisplay(missing_category_entries.head(2))\nprint(f\"Number of rows with missing 'category' entries: {len(missing_category_entries)}\\n----\\n\")\n\n# Get unique study IDs where 'category' is null\nunique_study_ids_missing_category_entries = missing_category_entries['study_id'].unique()\n\n# Count the number of unique study IDs\nnum_unique_study_ids_missing_category_entries = len(unique_study_ids_missing_category_entries)\n\n# Print the number of unique study IDs with missing category entries\nprint(f\"Number of unique study IDs with missing 'category' entries: {num_unique_study_ids_missing_category_entries}\")\ndisplay(unique_study_ids_missing_category_entries)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.828492Z","iopub.execute_input":"2024-06-20T05:30:41.828802Z","iopub.status.idle":"2024-06-20T05:30:41.855032Z","shell.execute_reply.started":"2024-06-20T05:30:41.828777Z","shell.execute_reply":"2024-06-20T05:30:41.853982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Cleaning data**","metadata":{}},{"cell_type":"markdown","source":"> > We will loose only 7 patients if we drop those withb missing category enteries.So its worth dropping them.","metadata":{}},{"cell_type":"code","source":"# Drop rows with missing values without modifiying original df (hence inplace=False) and store this in new df names train:\ntrain= train_data.dropna(inplace=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.856151Z","iopub.execute_input":"2024-06-20T05:30:41.856480Z","iopub.status.idle":"2024-06-20T05:30:41.889059Z","shell.execute_reply.started":"2024-06-20T05:30:41.856455Z","shell.execute_reply":"2024-06-20T05:30:41.888145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Visualise the results.  ","metadata":{}},{"cell_type":"code","source":"#Confirm if there are still any missing valyes in new df:\nnum_of_missing_values=train.isnull().sum()\nprint(num_of_missing_values)\n\n#Visualise the original and new df shapes and compare them :\nprint(f'\\n----\\nThe original dataframe (train_data) had {train_data.shape[0]} rows and {train_data.shape[1]} columns.')\nprint(f'The new dataframe (train) has {train.shape[0]} rows and {train.shape[1]} columns\\n----\\n')\nprint(f'We succesfulyl dropped {train_data.shape[0] - train.shape[0]} rows and {num_unique_study_ids_missing_category_entries} patients.')","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.890152Z","iopub.execute_input":"2024-06-20T05:30:41.890478Z","iopub.status.idle":"2024-06-20T05:30:41.916156Z","shell.execute_reply.started":"2024-06-20T05:30:41.890454Z","shell.execute_reply":"2024-06-20T05:30:41.915155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Data Normalization and Resizing**\n\nWe'll use libraries like OpenCV or PIL for image processing. Please refer to rsna_function utility script where we have defined preprocesss_image function normalise and resize images:\n> > Further notes:\n> > >  **Normalization:** (Scale pixel values to the range [0, 1] using image = image / 255.0):\n> > > > Reason for normalizing the images: \n> > > >  Many deep learning models benefit from normalized input, which can help in stabilizing and speeding up the training process. Since the images are grayscale, normalizing pixel values to [0, 1] is straightforward and beneficial.\n\n> > > **Resizing** (Resize images to a consistent size, such as 224x224 pixels, using OpenCV or similar libraries):\n> > > > Reason for resizing the images.\n> > > > Uniform image sizes are essential for CNNs, which require fixed input dimensions. Resizing to a consistent size, like 224x224, is typical for many pre-trained models and ensures compatibility.","metadata":{}},{"cell_type":"code","source":"# CALL PREPROCESS_IMAGES FUNCTION\n#We will parallel processing of python to process images faster: \nhelp(preprocess_images)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.917507Z","iopub.execute_input":"2024-06-20T05:30:41.917861Z","iopub.status.idle":"2024-06-20T05:30:41.925179Z","shell.execute_reply.started":"2024-06-20T05:30:41.917821Z","shell.execute_reply":"2024-06-20T05:30:41.924206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Call preprocess_images_parallel()\nprocessed_images=preprocess_images(train,\"img_file_path\",(224,224),-1)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:30:41.926246Z","iopub.execute_input":"2024-06-20T05:30:41.926586Z","iopub.status.idle":"2024-06-20T05:45:46.906617Z","shell.execute_reply.started":"2024-06-20T05:30:41.926556Z","shell.execute_reply":"2024-06-20T05:45:46.905688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check if processed_images is created properly.\nprint(len(processed_images))","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:46.907852Z","iopub.execute_input":"2024-06-20T05:45:46.908157Z","iopub.status.idle":"2024-06-20T05:45:46.912840Z","shell.execute_reply.started":"2024-06-20T05:45:46.908132Z","shell.execute_reply":"2024-06-20T05:45:46.911976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets add processed images as new column in the our dataset.\n\n# Create a temporary DataFrame to store the processed images\nprocessed_train = train.copy()\n#check if the len of the processed_images and the number of rows in processed_train is equal or not.\nprint(f'Is number of procssed images in processed_images is = number of rows in processed_train\\ndataframe?\\n-:{len(processed_images)==processed_train.shape[0]}.')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:46.913850Z","iopub.execute_input":"2024-06-20T05:45:46.914088Z","iopub.status.idle":"2024-06-20T05:45:46.937332Z","shell.execute_reply.started":"2024-06-20T05:45:46.914067Z","shell.execute_reply":"2024-06-20T05:45:46.936439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# So all is good, add new column to our df.\nprocessed_train['processed_images'] = processed_images","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:46.938468Z","iopub.execute_input":"2024-06-20T05:45:46.938731Z","iopub.status.idle":"2024-06-20T05:45:46.958875Z","shell.execute_reply.started":"2024-06-20T05:45:46.938709Z","shell.execute_reply":"2024-06-20T05:45:46.957984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check the updated dataset:\ndisplay(processed_train.head(3))\nprint(processed_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:46.959967Z","iopub.execute_input":"2024-06-20T05:45:46.960291Z","iopub.status.idle":"2024-06-20T05:45:47.007260Z","shell.execute_reply.started":"2024-06-20T05:45:46.960261Z","shell.execute_reply":"2024-06-20T05:45:47.006446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Fit label encoder to your training labels\nle = LabelEncoder()\nle.fit(processed_train['category'])\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.008334Z","iopub.execute_input":"2024-06-20T05:45:47.008635Z","iopub.status.idle":"2024-06-20T05:45:47.161949Z","shell.execute_reply.started":"2024-06-20T05:45:47.008612Z","shell.execute_reply":"2024-06-20T05:45:47.161052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transform processed_train labels using fitted labelEncoder:\nprocessed_train['category'] = le.transform(processed_train['category'])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.163317Z","iopub.execute_input":"2024-06-20T05:45:47.163634Z","iopub.status.idle":"2024-06-20T05:45:47.179150Z","shell.execute_reply.started":"2024-06-20T05:45:47.163610Z","shell.execute_reply":"2024-06-20T05:45:47.178325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check the updated dataset:\ndisplay(processed_train.head(3))\nprint(processed_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.184354Z","iopub.execute_input":"2024-06-20T05:45:47.184685Z","iopub.status.idle":"2024-06-20T05:45:47.224558Z","shell.execute_reply.started":"2024-06-20T05:45:47.184650Z","shell.execute_reply":"2024-06-20T05:45:47.223824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Further steps:\n\nThe data is now clean, we have added processed image column whihc contian serializsed imaged, we have LabelEncoded the label class and confiremd the integriy of our data.\n\nThis processed data is called processed_train.\n\n> Next steps:\n\nWe hsall split the trainig dataset into train and val subsets. While splitting the dataset, it is impportant to ensure that the distrubution of the label class is more or less simillar in both val and train subsets- this will become more important if our datasr has calss imbalnce. \n\nSo we shall first check for class imbalnce in main dataset-:\n> Checking for class imbalance in label / categories.","metadata":{}},{"cell_type":"code","source":"#check the laeblecnoding : Remeebr we named Labelenvoder as le before fitting it to outr classes-:\nle_map=dict(zip(sorted(le.classes_),le.transform(le.classes_)))\nprint(le_map)\n# See how mny. ategories (labels) are there and what count of each category ?\ncategory_counts=processed_train['category'].value_counts()\nprint(category_counts)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.225556Z","iopub.execute_input":"2024-06-20T05:45:47.225809Z","iopub.status.idle":"2024-06-20T05:45:47.236509Z","shell.execute_reply.started":"2024-06-20T05:45:47.225787Z","shell.execute_reply":"2024-06-20T05:45:47.235565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Visualise the distrubution of the various categories (labels)-:\n","metadata":{}},{"cell_type":"code","source":"# Depict this in a graph\ncategories = [keys for keys in le_map]\nprint(categories)\n\n# Create a bar plot\nbars = plt.bar(categories, category_counts, color=[\"green\", \"orange\", \"red\"])\n\n# Set the title and labels\nplt.title(\"Categories distribution\")\nplt.xlabel(\"Category\")\nplt.ylabel(\"Frequency\")\n\n# Add labels to each bar\nfor bar, (cat, label) in zip(bars, le_map.items()):\n    bar.set_label(f'{cat}: {label}')\n\n# Add the legend\nplt.legend(title=\"Categories\")\n\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.238008Z","iopub.execute_input":"2024-06-20T05:45:47.238334Z","iopub.status.idle":"2024-06-20T05:45:47.527225Z","shell.execute_reply.started":"2024-06-20T05:45:47.238311Z","shell.execute_reply":"2024-06-20T05:45:47.526278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clearly there is severe class balance. So we will split trian val set keeping this mind-:\nfrom sklearn.model_selection import train_test_split\n\n# Split the data\ntrain_df, val_df = train_test_split(processed_train, test_size=0.2, random_state=42, stratify=processed_train['category'])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.528412Z","iopub.execute_input":"2024-06-20T05:45:47.528694Z","iopub.status.idle":"2024-06-20T05:45:47.646306Z","shell.execute_reply.started":"2024-06-20T05:45:47.528671Z","shell.execute_reply":"2024-06-20T05:45:47.645360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DCheck if the train and val df are in good shape:\n\nprint(\"Showing train_df\\n-----------n\")\ndisplay(train_df.head())\nprint(train_df.shape)\n\nprint(\"Showing val_df\\n-----------n\")\ndisplay(val_df.head())\nprint(val_df.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:47.647549Z","iopub.execute_input":"2024-06-20T05:45:47.648751Z","iopub.status.idle":"2024-06-20T05:45:47.758162Z","shell.execute_reply.started":"2024-06-20T05:45:47.648722Z","shell.execute_reply":"2024-06-20T05:45:47.757283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make sure working directory is clearn: Runs only once.\ni=-1\n\nif i==1:\n    print(f\"The directoy has been already cleaned once.Please ensure you are not cleaning it again by mistake.\")\n    i=int(input(f\"Do you want to clean the dir again?\\nPlease enter 0 if you want to clean the dir again and 1 if not.\"))\n    if i==0:\n        remove_files(\"/kaggle/working/\")\n        i=i+1\nelse:\n    remove_files(\"/kaggle/working/\")\n    i=i+1","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:57:31.139321Z","iopub.execute_input":"2024-06-20T05:57:31.139967Z","iopub.status.idle":"2024-06-20T05:57:34.207804Z","shell.execute_reply.started":"2024-06-20T05:57:31.139935Z","shell.execute_reply":"2024-06-20T05:57:34.206817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the split datasets\n\nval_df.to_pickle('/kaggle/working/val_data.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:59:01.223751Z","iopub.execute_input":"2024-06-20T05:59:01.224479Z","iopub.status.idle":"2024-06-20T05:59:10.079329Z","shell.execute_reply.started":"2024-06-20T05:59:01.224441Z","shell.execute_reply":"2024-06-20T05:59:10.078524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if proper pickling done or not for val_df\nval_check = pd.read_pickle('/kaggle/working/val_data.pkl')\n\nif val_df.equals(val_check):\n    print(\"The val_df has been properly pickled.\")\nelse:\n    print(\"There was an error in pickling the val_df.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T06:07:49.016933Z","iopub.execute_input":"2024-06-20T06:07:49.017300Z","iopub.status.idle":"2024-06-20T06:08:02.567411Z","shell.execute_reply.started":"2024-06-20T06:07:49.017271Z","shell.execute_reply":"2024-06-20T06:08:02.566440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Rest your RAM !\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T06:08:57.779824Z","iopub.execute_input":"2024-06-20T06:08:57.780201Z","iopub.status.idle":"2024-06-20T06:08:57.785549Z","shell.execute_reply.started":"2024-06-20T06:08:57.780173Z","shell.execute_reply":"2024-06-20T06:08:57.784163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_pickle('/kaggle/working/train_data.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-06-20T06:17:06.886576Z","iopub.execute_input":"2024-06-20T06:17:06.887011Z","iopub.status.idle":"2024-06-20T06:18:13.694821Z","shell.execute_reply.started":"2024-06-20T06:17:06.886978Z","shell.execute_reply":"2024-06-20T06:18:13.693955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if proper pickling done or not for train_df\ntrain_check = pd.read_pickle('/kaggle/working/train_data.pkl')\n\nif train_df.equals(train_check):\n    print(\"The val_df has been properly pickled.\")\nelse:\n    print(\"There was an error in pickling the val_df.\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final messge:\nprint(f\"The training and validation data subsets are prepared and reaady for model building.\\n------\\nPlease go to RSNA_model notebook and ensure the train_df and val_df are uploaded in input dir of this notebook before proceeding to next steps.\")\nprint(\"Goodluck and Goodbye!\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T05:45:48.472681Z","iopub.status.idle":"2024-06-20T05:45:48.473008Z","shell.execute_reply.started":"2024-06-20T05:45:48.472848Z","shell.execute_reply":"2024-06-20T05:45:48.472862Z"},"trusted":true},"execution_count":null,"outputs":[]}]}