{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Now define a the input directoy path:\nstart_dir_path=\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n#First import relevant libraries:\nimport os as os\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt #For ploting and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\nimport os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. \nimport ipywidgets as widgets\nfrom IPython.display import display\nimport pydicom\nimport pickle\nimport logging\nfrom multiprocessing import Pool\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Assign file path to simple variables:\n# CSV fiels provided and thier paths:\ntrain_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\nsample_csv_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv'\ntrain_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv'\ntrain_label_coordinates_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntest_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv'\n\n# Image folder paths:\ntrain_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images'\ntest_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-20T04:19:48.269390Z","iopub.execute_input":"2024-06-20T04:19:48.269907Z","iopub.status.idle":"2024-06-20T04:19:48.280266Z","shell.execute_reply.started":"2024-06-20T04:19:48.269867Z","shell.execute_reply":"2024-06-20T04:19:48.279099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets check the directory structure. We will take help of os library:\n\n## Define a function to pring the files and filder overview:\ndef print_dir_str(dir_path):\n    \"\"\"A function to print the files and folders in the root input folder\"\"\"\n    \n    #creates a tupple containing root(main folder), dirs(subfolder) and files.\n    directories=os.walk(dir_path) \n    \n    #make a list of of all files and folder in input root folder.\n    start_dir_contents=os.listdir(dir_path) \n    print(f\"Total number of files and folder in root directory are = {len(start_dir_contents)}\\n---------------------------------------------------\\n\")\n    \n    #loop through tupple to print file/folder path:\n    for root,dirs,files in directories: \n        print('\\n[Root dir]\\n')\n        print(f\"{root}\")\n        print('\\n[Folders]\\n')\n        for directory in dirs:\n            print (f'{\" \"*4}{os.path.join(dir_path,directory)}')\n        print('\\n[Files]\\n')\n        for file in files:\n            print(f'{\" \"*8}{os.path.join(dir_path,file)}')\n        return(start_dir_contents)\n        break","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.282747Z","iopub.execute_input":"2024-06-20T04:19:48.283176Z","iopub.status.idle":"2024-06-20T04:19:48.296907Z","shell.execute_reply.started":"2024-06-20T04:19:48.283144Z","shell.execute_reply":"2024-06-20T04:19:48.295450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read dir structure in stepwsie interactie interphase -:\n> Function name: read-dir_stepwise()\nFunction Description\nThe read_dir_stepwise function walks through a directory tree starting from the given directory path (dir_path). For each directory encountered, it interacts with the user to determine if the details of the directory (including subfolders and files) should be printed.\n\nParameters\ndir_path (str): The path of the directory to traverse.\nUsage\nCall the function with the path of the directory you want to traverse. For example:","metadata":{}},{"cell_type":"code","source":"import os\n\ndef read_dir_stepwise(dir_path):\n    \"\"\"\n    A function to traverse a directory stepwise and interactively.\n    \n    Parameters:\n    dir_path (str): The path of the directory to traverse.\n\n    This function walks through the directory tree starting from the given directory path.\n    For each directory (root), it will ask the user if they want to print the folder details.\n    If the user enters 'y', it will print the root directory, subfolders, and files within it.\n    If the user enters 'n', it will skip to the next directory.\n    The function will continue until all directories have been processed or the user aborts the process.\n\n    Usage:\n    read_dir_stepwise('/path/to/directory')\n\n    Example:\n    read_dir_stepwise('/kaggle/input/dataset')\n    \"\"\"\n    for root, folders, files in os.walk(dir_path):\n        user_input = input('Do you want to print this folder? If yes, enter y; otherwise, enter n: ').strip().lower()\n        if user_input == 'y':\n            if len(root) > 0:\n                print(f\"\\nRoot directory: {root}\")\n            else:\n                print(f\"There is no parent folder.\")\n                \n            if len(folders) > 0:\n                print(f\"Subfolders: {folders}\")\n            else:\n                print(f\"There are no subfolders in the root folder.\")\n                \n            if len(files) > 0:\n                print(f\"\\n....There are {len(files)} files in the root folder: {files}\")\n            else:\n                print(f\"There are {len(folders)} folders in the root folder but there are no files.\")\n        elif user_input == 'n':\n            print(\"Skipping to the next folder.\")\n        else:\n            print(\"The process was aborted by the user.\")\n            break\n\n# Example usage\n# read_dir_stepwise('/kaggle/working')\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import os\n\ndef read_dir_stepwise(dir_path):\n    \"\"\"\n    A function to traverse a directory stepwise and interactively.\n    \n    Parameters:\n    dir_path (str): The path of the directory to traverse.\n\n    This function walks through the directory tree starting from the given directory path.\n    For each directory (root), it will ask the user if they want to print the folder details.\n    If the user enters 'y', it will print the root directory, subfolders, and files within it.\n    If the user enters 'n', it will skip to the next directory.\n    The function will continue until all directories have been processed or the user aborts the process.\n\n    Usage:\n    read_dir_stepwise('/path/to/directory')\n\n    Example:\n    read_dir_stepwise('/kaggle/input/dataset')\n    \"\"\"\n    for root, folders, files in os.walk(dir_path):\n        user_input = input('Do you want to print this folder? If yes, enter y; otherwise, enter n: ').strip().lower()\n        if user_input == 'y':\n            if len(root) > 0:\n                print(f\"\\nRoot directory: {root}\")\n            else:\n                print(f\"There is no parent folder.\")\n                \n            if len(folders) > 0:\n                print(f\"Subfolders: {folders}\")\n            else:\n                print(f\"There are no subfolders in the root folder.\")\n                \n            if len(files) > 0:\n                print(f\"\\n....There are {len(files)} files in the root folder: {files}\")\n            else:\n                print(f\"There are {len(folders)} folders in the root folder but there are no files.\")\n        elif user_input == 'n':\n            print(\"Skipping to the next folder.\")\n        else:\n            print(\"The process was aborted by the user.\")\n            break\n\n# Example usage\n# read_dir_stepwise('/kaggle/working')\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Display messages when the script runs succesfully :\nprint(\"Function print_dir_str() sucesfully imported.\\nThis function prints the files and folders in the root input folder.\\n---------\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.299082Z","iopub.execute_input":"2024-06-20T04:19:48.299537Z","iopub.status.idle":"2024-06-20T04:19:48.311580Z","shell.execute_reply.started":"2024-06-20T04:19:48.299500Z","shell.execute_reply":"2024-06-20T04:19:48.310131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define a function to display DICOM images for a given file path\ndef display_dicom_images(idx,df):\n    \"\"\"A function to display aDICOM images for a given index.\n    The function takes idx , a integer whihc represent index of the image.\n    the second arg is df which the dataset from wchich the image file path is derived.\n    The images are annoted by a transparent circle to marke the x and y coordinates marking \n    the epicentre of the site being analysed.\n    Finally this function also creates a widget object Index-slider\"\"\"\n    \n    # Define the list of DICOM file paths\n    img_file_paths = df['img_file_path'].tolist()\n    x_coordinates= df['x'].tolist()\n    y_coordinates= df['y'].tolist()\n    file_path = img_file_paths[idx]\n    \n    # Import DICOM object\n    dicom_obj = pydicom.dcmread(file_path)\n    # Read DICOM object as image\n    img = dicom_obj.pixel_array\n    # Display the image using matplotlib.pyplot, use grayscale color map to display image\n    plt.imshow(img, cmap='gray')\n    plt.title(f'{df.loc[idx, \"condition\"]} at {df.loc[idx, \"level\"]}\\n-------------\\n{df.loc[idx, \"category\"]}')\n    plt.axis('off')\n    \n    #overlay annotation on images:\n      # Overlay annotation (colored circle)\n    x = x_coordinates[idx]\n    y = y_coordinates[idx]\n    #circle = plt.patches.Circle((x, y), radius=20, color='yellow', fill=False, linestyle='dashed')\n    plt.plot(x_coordinates[idx],y_coordinates[idx],'o', markersize=20, linestyle=\"solid\",alpha=0.5)  # 'o' for circle, alpha for transparency\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.313216Z","iopub.execute_input":"2024-06-20T04:19:48.313655Z","iopub.status.idle":"2024-06-20T04:19:48.325017Z","shell.execute_reply.started":"2024-06-20T04:19:48.313611Z","shell.execute_reply":"2024-06-20T04:19:48.323654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Function disaply_dicom_images(idx,df) sucessfully loaded.\\nThis function to display a DICOM images for a given index (idx).\\nThe function takes idx ,a integer which represents index of the image.\\nThe second arg is df which is the dataset from wchich the image file path is derived.\\nThe images are annoted by a transparent circle to marke the x and y coordinates marking the epicentre of the site being analysed\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.328658Z","iopub.execute_input":"2024-06-20T04:19:48.329246Z","iopub.status.idle":"2024-06-20T04:19:48.338776Z","shell.execute_reply.started":"2024-06-20T04:19:48.329198Z","shell.execute_reply":"2024-06-20T04:19:48.337702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_image_slider(df):\n    \"\"\"Creates a widget for scrolling through DICOM images.\"\"\"\n    # Create a widget for scrolling through images\n    index_slider = widgets.IntSlider(value=0, min=0, max=len(df['img_file_path']) - 1, step=1, description='Image Index')\n    \n    # Use the widget to interactively scroll through images\n    widgets.interact(display_dicom_images, idx=index_slider, df=widgets.fixed(df))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.340452Z","iopub.execute_input":"2024-06-20T04:19:48.340877Z","iopub.status.idle":"2024-06-20T04:19:48.349913Z","shell.execute_reply.started":"2024-06-20T04:19:48.340834Z","shell.execute_reply":"2024-06-20T04:19:48.348700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n----------\\nFunction create_image_slider(df) sucessfully imported.\\nThis function creates a widget for scrolling through DICOM images\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.351929Z","iopub.execute_input":"2024-06-20T04:19:48.352448Z","iopub.status.idle":"2024-06-20T04:19:48.365266Z","shell.execute_reply.started":"2024-06-20T04:19:48.352407Z","shell.execute_reply":"2024-06-20T04:19:48.364131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_images(df, image_column, target_size, num_of_rows=1000):\n    \"\"\"\n    Function to resize and normalize the images in a dataset, and serialize them using pickle.\n\n    Args:\n    df (DataFrame): The dataset containing the path to images.\n    image_column (str): The name of the column in df which contains the image file paths.\n    target_size (tuple): The desired target size of the image (usually (224, 224)).\n    num_of_rows (int): An integer that defines how many rows are processed at a time. \n                       By default, it is set to 1000. To process the entire dataset, set this to -1.\n\n    Returns:\n    list: A list of serialized processed images that can be appended to any df as a new column.\n\n    Steps Involved:\n    1. Loop through the specified number of rows in the dataframe.\n    2. Read each DICOM file using pydicom.\n    3. Extract the image data from the DICOM file.\n    4. Convert the image to a format compatible with OpenCV.\n    5. Resize the image using OpenCV.\n    6. Normalize pixel values.\n    7. Serialize the processed image using pickle.\n    8. Append the serialized image to a list.\n    9. Return the list of serialized processed images.\n\n    Example usage:\n    processed_images = preprocess_images(train, 'image_file_path', (224, 224))\n\n    To save the processed images to a DataFrame:\n    temp_train = train.copy()\n    temp_train.loc[:999, 'processed_images'] = processed_images\n\n    To save the DataFrame to a pickle file:\n    temp_train.to_pickle('processed_train_data.pkl')\n    \"\"\"\n\n    processed_images = []\n    \n    if num_of_rows == -1:\n        num_of_rows = len(df)\n    \n    for image_path in tqdm(df[:num_of_rows][image_column]):\n        try:\n            # Read the DICOM file using pydicom\n            dicom = pydicom.dcmread(image_path)\n            \n            # Extract the image data from the DICOM file\n            image = dicom.pixel_array\n            \n            # Convert the image to a format compatible with OpenCV\n            image = cv2.convertScaleAbs(image)\n            \n            # Resize the image using OpenCV\n            image = cv2.resize(image, target_size)\n            \n            # Normalize pixel values\n            image = image / 255.0\n            \n            # Serialize the processed image\n            serialized_image = pickle.dumps(image)\n            processed_images.append(serialized_image)\n        except Exception as e:\n            print(f\"Error processing file {image_path}: {e}\")\n    \n    print(f\"Successfully processed images in {len(df[:num_of_rows])} rows\")\n    return processed_images\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.366682Z","iopub.execute_input":"2024-06-20T04:19:48.367053Z","iopub.status.idle":"2024-06-20T04:19:48.379850Z","shell.execute_reply.started":"2024-06-20T04:19:48.366998Z","shell.execute_reply":"2024-06-20T04:19:48.378486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef preprocess_images_in_batches(df, image_column, target_size, num_of_rows=10000):\n    \"\"\"\n    Function to resize and normalize the images in a dataset, and serialize them using pickle.\n\n    Args:\n    df (DataFrame): The dataset containing the path to images.\n    image_column (str): The name of the column in df which contains the image file paths.\n    target_size (tuple): The desired target size of the image (usually (224, 224)).\n    num_of_rows (int): An integer that defines how many rows are processed at a time. \n                       By default, it is set to 10,000. To process the entire dataset, set this to -1.\n\n    Returns:\n    list: A list of serialized processed images that can be appended to any df as a new column.\n\n    Steps Involved:\n    1. Loop through the specified number of rows in the dataframe.\n    2. Read each DICOM file using pydicom.\n    3. Extract the image data from the DICOM file.\n    4. Convert the image to a format compatible with OpenCV.\n    5. Resize the image using OpenCV.\n    6. Normalize pixel values.\n    7. Serialize the processed image using pickle.\n    8. Append the serialized image to a list.\n    9. Return the list of serialized processed images.\n\n    Example usage:\n    processed_images = preprocess_images(train, 'image_file_path', (224, 224))\n\n    To save the processed images to a DataFrame:\n    temp_train = train.copy()\n    temp_train.loc[:999, 'processed_images'] = processed_images\n\n    To save the DataFrame to a pickle file:\n    temp_train.to_pickle('processed_train_data.pkl')\n    \"\"\"\n    processed_images = []\n    total_rows = len(df) if num_of_rows == -1 else min(num_of_rows, len(df))\n    \n    for start in range(0, total_rows, 10000):\n        end = min(start + 10000, total_rows)\n        batch_df = df.iloc[start:end]\n        \n        for image_path in tqdm(batch_df[image_column], desc=f\"Processing batch {start // 10000 + 1}\"):\n            try:\n                # Read the DICOM file using pydicom\n                dicom = pydicom.dcmread(image_path)\n                \n                # Extract the image data from the DICOM file\n                image = dicom.pixel_array\n                \n                # Convert the image to a format compatible with OpenCV\n                image = cv2.convertScaleAbs(image)\n                \n                # Resize the image using OpenCV\n                image = cv2.resize(image, target_size)\n                \n                # Normalize pixel values\n                image = image / 255.0\n                \n                # Serialize the processed image\n                serialized_image = pickle.dumps(image)\n                processed_images.append(serialized_image)\n            except Exception as e:\n                print(f\"Error processing file {image_path}: {e}\")\n        \n        print(f\"Successfully processed images in rows {start} to {end-1}\")\n        if end < total_rows:\n            input(\"Press Enter to continue processing the next batch...\")\n\n    print(f\"Successfully processed all {total_rows} images\")\n    return processed_images\n\n# Example usage\n# processed_images = preprocess_images(train, 'image_file_path', (224, 224))\n# temp_train = train.copy()\n# temp_train['processed_images'] = processed_images\n# temp_train.to_pickle('processed_train_data.pkl')\n# print(\"Processed images have been successfully added to the DataFrame and saved to 'processed_train_data.pkl'\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.381562Z","iopub.execute_input":"2024-06-20T04:19:48.381944Z","iopub.status.idle":"2024-06-20T04:19:48.399248Z","shell.execute_reply.started":"2024-06-20T04:19:48.381912Z","shell.execute_reply":"2024-06-20T04:19:48.397928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Function to process images using parallel processing power pf pythong:\ndef preprocess_image(image_path, target_size):\n    try:\n        # Read the DICOM file using pydicom\n        dicom = pydicom.dcmread(image_path)\n        \n        # Extract the image data from the DICOM file\n        image = dicom.pixel_array\n        \n        # Convert the image to a format compatible with OpenCV\n        image = cv2.convertScaleAbs(image)\n        \n        # Resize the image using OpenCV\n        image = cv2.resize(image, target_size)\n        \n        # Normalize pixel values\n        image = image / 255.0\n        \n        # Serialize the processed image\n        serialized_image = pickle.dumps(image)\n        return serialized_image\n    except Exception as e:\n        print(f\"Error processing file {image_path}: {e}\")\n        return None\n\ndef preprocess_images_in_parallel(df, image_column, target_size, num_of_rows=10000):\n    processed_images = []\n    total_rows = len(df) if num_of_rows == -1 else min(num_of_rows, len(df))\n    \n    with Pool() as pool:\n        image_paths = df[:total_rows][image_column].tolist()\n        processed_images = pool.starmap(preprocess_image, [(path, target_size) for path in image_paths])\n    \n    print(f\"Successfully processed all {total_rows} images\")\n    return processed_images\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.400711Z","iopub.execute_input":"2024-06-20T04:19:48.401149Z","iopub.status.idle":"2024-06-20T04:19:48.415451Z","shell.execute_reply.started":"2024-06-20T04:19:48.401111Z","shell.execute_reply":"2024-06-20T04:19:48.414092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_to_pickle(df, image_column, target_size=(224, 224), num_of_rows=1000):\n    \"\"\"\n    Function to preprocess images and store the DataFrame with serialized images as a pickle file.\n\n    Args:\n    df (DataFrame): The dataset containing the path to images.\n    image_column (str): The name of the column in df which contains the image file paths.\n    target_size (tuple): The desired target size of the image (usually (224, 224)).\n    num_of_rows (int): An integer that defines how many rows are processed at a time. \n                       By default, it is set to 1000. To process the entire dataset, set this to -1.\n\n    Steps Involved:\n    1. Preprocess the images using the preprocess_images function.\n    2. Create a temporary DataFrame to store the processed images.\n    3. Save the DataFrame with processed images to a pickle file.\n\n    Example usage:\n    save_to_pickle(train, 'image_file_path', (224, 224))\n    \"\"\"\n    # Preprocess the images\n    processed_images = preprocess_images(df, image_column, target_size, num_of_rows)\n\n    # Create a temporary DataFrame to store the processed images\n    temp_df = df.copy()\n    temp_df.loc[:num_of_rows-1, 'processed_images'] = processed_images\n\n    # Save the DataFrame with processed images to a pickle file\n    temp_df.to_pickle('processed_train_data.pkl')\n\n    print(\"Processed images have been successfully added to the DataFrame and saved to 'processed_train_data.pkl'\")","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.417135Z","iopub.execute_input":"2024-06-20T04:19:48.417506Z","iopub.status.idle":"2024-06-20T04:19:48.431207Z","shell.execute_reply.started":"2024-06-20T04:19:48.417476Z","shell.execute_reply":"2024-06-20T04:19:48.429854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_and_deserialize_example(pickle_file_path, index):\n    \"\"\"\n    Function to load a DataFrame from a pickle file and deserialize an image.\n\n    Args:\n    pickle_file_path (str): The path to the pickle file containing the DataFrame with serialized images.\n    index (int): The index of the image to deserialize.\n\n    Returns:\n    numpy.ndarray: The deserialized image.\n\n    Steps Involved:\n    1. Load the DataFrame from the pickle file.\n    2. Deserialize the image at the specified index using pickle.\n\n    Example usage:\n    example_image = load_and_deserialize_example('processed_train_data.pkl', 0)\n    print(example_image.shape)\n    \"\"\"\n    loaded_df = pd.read_pickle(pickle_file_path)\n    example_image = pickle.loads(loaded_df.loc[index, 'processed_images'])\n    print(example_image.shape)\n    plt.plot(example_image)\n    plot.show(example_image)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.432707Z","iopub.execute_input":"2024-06-20T04:19:48.433138Z","iopub.status.idle":"2024-06-20T04:19:48.445603Z","shell.execute_reply.started":"2024-06-20T04:19:48.433092Z","shell.execute_reply":"2024-06-20T04:19:48.444422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Due to data size (over 50,000 images) it may be possible that our function may get stuck.\n# We can multiprocessing power of multicore CPU of kaggle. for this we will now write process_image and process_images_parallel() that will allow processing in paralle making best use of multiple cores of CPU.#\n# Multiprocessing has been used to import pool (see above)\n# Logging has already been imported.\n\n# Setting up logging\nlogging.basicConfig(filename='preprocess_errors.log', level=logging.ERROR)\n\ndef preprocess_image(image_path, target_size):\n    \"\"\"\n    Preprocess a single image by reading a DICOM file, resizing, normalizing, and serializing it.\n    \n    Args:\n    - image_path (str): Path to the DICOM image file.\n    - target_size (tuple): Desired size to resize the image (height, width).\n    \n    Returns:\n    - bytes: Serialized processed image.\n    \"\"\"\n    try:\n        # Read the DICOM file using pydicom\n        dicom = pydicom.dcmread(image_path)\n        \n        # Extract the image data from the DICOM file\n        image = dicom.pixel_array\n        \n        # Convert the image to a format compatible with OpenCV\n        image = cv2.convertScaleAbs(image)\n        \n        # Resize the image using OpenCV\n        image = cv2.resize(image, target_size)\n        \n        # Normalize pixel values\n        image = image / 255.0\n        \n        # Serialize the processed image\n        serialized_image = pickle.dumps(image)\n        return serialized_image\n    except Exception as e:\n        logging.error(f\"Error processing file {image_path}: {e}\")\n        return None\n\ndef preprocess_images_parallel(df, image_column, target_size, num_of_rows=1000):\n    \"\"\"\n    Preprocess images in parallel by reading, resizing, normalizing, and serializing them.\n    \n    Args:\n    - df (pd.DataFrame): DataFrame containing image file paths.\n    - image_column (str): Column name in df that contains the image file paths.\n    - target_size (tuple): Desired size to resize the images (height, width).\n    - num_of_rows (int): Number of rows to process at a time. Process entire dataset if set to -1.\n    \n    Returns:\n    - list: List of serialized processed images.\n    \"\"\"\n    if num_of_rows == -1:\n        num_of_rows = len(df)\n    \n    image_paths = df[:num_of_rows][image_column].tolist()\n    with Pool() as pool:\n        processed_images = pool.starmap(preprocess_image, [(path, target_size) for path in image_paths])\n    \n    return processed_images\n\n# Documentation Example:\n\"\"\"\nExample Usage:\n\n# Load dataset\ndata = pd.read_csv('path_to_train_label_coordinates_data.csv')\n\n# Preprocess images in parallel\nprocessed_images = preprocess_images_parallel(data, 'image_path', (224, 224))\n\n# Add processed images to DataFrame\ndata['processed_image'] = processed_images\n\n# Save the DataFrame to a pickle file\ndata.to_pickle('processed_train_data.pkl')\n\"\"\"\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.447144Z","iopub.execute_input":"2024-06-20T04:19:48.447596Z","iopub.status.idle":"2024-06-20T04:19:48.468889Z","shell.execute_reply.started":"2024-06-20T04:19:48.447553Z","shell.execute_reply":"2024-06-20T04:19:48.467247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_augmentation_generator(df, image_column, batch_size=32, target_size=(224, 224)):\n    \"\"\"\n    Generator function to yield batches of augmented images.\n\n    Args:\n    - df (pd.DataFrame): DataFrame containing the serialized images.\n    - image_column (str): Column name in df that contains the serialized images.\n    - batch_size (int): Size of the batches to yield.\n    - target_size (tuple): Desired size of the image (height, width).\n    \n    Yields:\n    - tuple: Batch of augmented images and corresponding labels.\n    \"\"\"\n    num_samples = len(df)\n    \n    while True:\n        for offset in range(0, num_samples, batch_size):\n            batch_df = df.iloc[offset:offset + batch_size]\n            batch_images = []\n            \n            for serialized_image in batch_df[image_column]:\n                image = pickle.loads(serialized_image)\n                batch_images.append(image)\n            \n            batch_images = np.array(batch_images)\n            batch_images = batch_images.reshape((-1, target_size[0], target_size[1], 1))\n            \n            augmented_images = datagen.flow(batch_images, batch_size=batch_size, shuffle=False).next()\n            \n            yield augmented_images\n\n# Example usage in a model training context\n\"\"\"\n# Assuming you have a DataFrame `processed_train` with a 'processed_image' column\n# and the corresponding labels in a 'label' column\n\n# Split the data into training and validation sets\nfrom sklearn.model_selection import train_test_split\ntrain_df, val_df = train_test_split(processed_train, test_size=0.2, random_state=42)\n\n# Create the generators\ntrain_generator = image_augmentation_generator(train_df, 'processed_image', batch_size=32)\nval_generator = image_augmentation_generator(val_df, 'processed_image', batch_size=32)\n\n# Assuming a Keras model `model`\n# model.fit(train_generator, validation_data=val_generator, steps_per_epoch=len(train_df)//32, validation_steps=len(val_df)//32, epochs=10)\n\"\"\"\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.473725Z","iopub.execute_input":"2024-06-20T04:19:48.474227Z","iopub.status.idle":"2024-06-20T04:19:48.489722Z","shell.execute_reply.started":"2024-06-20T04:19:48.474182Z","shell.execute_reply":"2024-06-20T04:19:48.488510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef remove_files(dir, n=0):\n    \"\"\"\n    A function specifically designed to delete files from a directory.\n    The function provides active feedback to the user.\n    It takes one argument:\n    - dir: The path of the directory in which files to be deleted are present.\n    \n    The function, when called, will ask the user to enter how many files they want to remove.\n    \"\"\"\n    dir_list = os.listdir(dir)\n    print(f'There are {len(dir_list)} items in this directory.')\n    \n    # Filter the list to include only files\n    file_list = [f for f in dir_list if os.path.isfile(os.path.join(dir, f))]\n    print(f'There are {len(file_list)} files in this directory.')\n\n    # If n is not provided or is zero, ask the user for input\n    if n == 0:\n        n = int(input(f'How many files do you want to remove? '))\n    \n    if n < 0 or n > len(file_list):\n        raise ValueError(f'The number of files to remove cannot be less than 0 or greater than the number of files in the directory ({len(file_list)}).')\n    \n    for i in range(n):\n        file_path = os.path.join(dir, file_list[i])\n        os.remove(file_path)\n        print(f'Removed {file_list[i]} successfully.')\n    \n    print(f'Removed {n} files from {dir}.\\n---\\nRemaining {len(file_list) - n} files')\n\n\n#example\n#remove_files('/kaggle/working')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.491558Z","iopub.execute_input":"2024-06-20T04:19:48.492085Z","iopub.status.idle":"2024-06-20T04:19:48.503498Z","shell.execute_reply.started":"2024-06-20T04:19:48.492022Z","shell.execute_reply":"2024-06-20T04:19:48.502102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print messages\nprint(\"\"\"\nFunctions preprocess_images(), load_and_deserialize_example(), and save_to_pickle() successfully loaded.\n\nFunction preprocess_images():\n- Purpose: Resizes and normalizes images in a DataFrame, then serializes them using pickle.\n- Arguments:\n  - df (DataFrame): The dataset containing the paths to images.\n  - image_column (str): The name of the column in df which contains the image file paths.\n  - target_size (tuple): The desired target size of the image (e.g., (224, 224)).\n  - num_of_rows (int): Number of rows to process at a time. Defaults to 1000. Set to -1 to process the entire dataset.\n- Returns: A list of serialized processed images.\n\nExample usage:\nprocessed_images = preprocess_images(train, 'image_file_path', (224, 224))\nTo save the processed images to a DataFrame:\ntemp_train = train.copy()\ntemp_train.loc[:999, 'processed_images'] = processed_images\nTo save the DataFrame to a pickle file:\ntemp_train.to_pickle('processed_train_data.pkl')\n\nFunction load_and_deserialize_example():\n- Purpose: Loads a DataFrame from a pickle file and deserializes an image.\n- Arguments:\n  - pickle_file_path (str): The path to the pickle file containing the DataFrame with serialized images.\n  - index (int): The index of the image to deserialize.\n- Returns: A deserialized image (numpy.ndarray).\n\nExample usage:\nexample_image = load_and_deserialize_example('processed_train_data.pkl', 0)\nprint(example_image.shape)\n\nFunction save_to_pickle():\n- Purpose: Preprocesses images and stores the DataFrame with serialized images as a pickle file.\n- Arguments:\n  - df (DataFrame): The dataset containing the path to images.\n  - image_column (str): The name of the column in df which contains the image file paths.\n  - target_size (tuple): The desired target size of the image (usually (224, 224)).\n  - num_of_rows (int): An integer that defines how many rows are processed at a time. \n                       By default, it is set to 1000. To process the entire dataset, set this to -1.\n\nSteps Involved:\n1. Preprocess the images using the preprocess_images function.\n2. Create a temporary DataFrame to store the processed images.\n3. Save the DataFrame with processed images to a pickle file.\n\nExample usage:\nsave_to_pickle(train, 'image_file_path', (224, 224))\n\"\"\")\n\nprint('\\n----------\\nFunction process_image() and process_images_prallel() imported succefully.\\n---\\nDue to data size (over 50,000 images) it may be possible that our function may get stuck.\\nWe can multiprocessing power of multicore CPU of kaggle.\\nFor this we will now write process_image and process_images_parallel() that will allow processing in paralle making best use of multiple cores of CPU.#Multiprocessing has been used to import pool (see above).\\n A logging function also added.')","metadata":{"execution":{"iopub.status.busy":"2024-06-20T04:19:48.505438Z","iopub.execute_input":"2024-06-20T04:19:48.505949Z","iopub.status.idle":"2024-06-20T04:19:48.520399Z","shell.execute_reply.started":"2024-06-20T04:19:48.505908Z","shell.execute_reply":"2024-06-20T04:19:48.519068Z"},"trusted":true},"execution_count":null,"outputs":[]}]}