{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":992,"sourceType":"modelInstanceVersion","modelInstanceId":846,"modelId":101}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ncounter = 0  # Sayaç başlat\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        counter += 1\n        if counter == 15:  # 5 dosya yazdırdıktan sonra dur\n            break\n    if counter == 15:  # İç döngü kırıldığında dış döngüyü de kır\n        break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:24.661177Z","iopub.execute_input":"2024-12-17T16:24:24.661834Z","iopub.status.idle":"2024-12-17T16:24:25.985138Z","shell.execute_reply.started":"2024-12-17T16:24:24.661801Z","shell.execute_reply":"2024-12-17T16:24:25.984197Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We are preparing this notebook to prepare the dataset before the training era. We were inspired from these notebooks: https://www.kaggle.com/code/itsuki9180/rsna2024-lsdc-making-dataset/notebook https://www.kaggle.com/code/itsuki9180/rsna2024-lsdc-making-dataset I think creating a new row_id for test and train cvs files is best approach to use dicom files in a easy way. Its too complicated pulling rows from different csv files for one dicom image. We didnt decided yet but we probably convert these dicom images to png and use this notebook output as a input for our model notebook.","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\nimport matplotlib.pyplot as plt\nimport os\nimport time\nimport numpy as np\nimport glob\nimport json\nimport collections\nimport torch\nimport torch.nn as nn\n\nimport pydicom as dicom\nimport matplotlib.patches as patches\n\nfrom matplotlib import animation, rc\nimport pandas as pd\n\nimport pydicom as dicom # dicom\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:25.986590Z","iopub.execute_input":"2024-12-17T16:24:25.986895Z","iopub.status.idle":"2024-12-17T16:24:29.421992Z","shell.execute_reply.started":"2024-12-17T16:24:25.986866Z","shell.execute_reply":"2024-12-17T16:24:29.421087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We added the libraries that we need. You can see the pythorch libraries. We might delete these becasue we are going to use a different notebook for training. We added just in case we might need it. ","metadata":{}},{"cell_type":"code","source":"# read data\ntrain_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\n\ntrain  = pd.read_csv(train_path + 'train.csv')\nlabel = pd.read_csv(train_path + 'train_label_coordinates.csv')\ntrain_desc  = pd.read_csv(train_path + 'train_series_descriptions.csv')\ntest_desc   = pd.read_csv(train_path + 'test_series_descriptions.csv')\nsub         = pd.read_csv(train_path + 'sample_submission.csv')\nlen(test_desc) #number of test_description.csv rows ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:29.423096Z","iopub.execute_input":"2024-12-17T16:24:29.423476Z","iopub.status.idle":"2024-12-17T16:24:29.568273Z","shell.execute_reply.started":"2024-12-17T16:24:29.423450Z","shell.execute_reply":"2024-12-17T16:24:29.567436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:29.570104Z","iopub.execute_input":"2024-12-17T16:24:29.570366Z","iopub.status.idle":"2024-12-17T16:24:29.582820Z","shell.execute_reply.started":"2024-12-17T16:24:29.570341Z","shell.execute_reply":"2024-12-17T16:24:29.582018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:29.583863Z","iopub.execute_input":"2024-12-17T16:24:29.584201Z","iopub.status.idle":"2024-12-17T16:24:29.607546Z","shell.execute_reply.started":"2024-12-17T16:24:29.584164Z","shell.execute_reply":"2024-12-17T16:24:29.606783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:29.608947Z","iopub.execute_input":"2024-12-17T16:24:29.609319Z","iopub.status.idle":"2024-12-17T16:24:29.622465Z","shell.execute_reply.started":"2024-12-17T16:24:29.609280Z","shell.execute_reply":"2024-12-17T16:24:29.621606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to generate image paths based on directory structure\ndef generate_image_paths(df, data_dir):\n    image_paths = []\n    for study_id, series_id in zip(df['study_id'], df['series_id']):\n        study_dir = os.path.join(data_dir, str(study_id))\n        series_dir = os.path.join(study_dir, str(series_id))\n        images = os.listdir(series_dir)\n        image_paths.extend([os.path.join(series_dir, img) for img in images])\n    return image_paths\n\n# Generate image paths for train and test data\ntrain_image_paths = generate_image_paths(train_desc, f'{train_path}/train_images')\ntest_image_paths = generate_image_paths(test_desc, f'{train_path}/test_images')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:24:29.623292Z","iopub.execute_input":"2024-12-17T16:24:29.623509Z","iopub.status.idle":"2024-12-17T16:25:21.316244Z","shell.execute_reply.started":"2024-12-17T16:24:29.623487Z","shell.execute_reply":"2024-12-17T16:25:21.315304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_desc)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:21.317481Z","iopub.execute_input":"2024-12-17T16:25:21.317846Z","iopub.status.idle":"2024-12-17T16:25:21.323753Z","shell.execute_reply.started":"2024-12-17T16:25:21.317807Z","shell.execute_reply":"2024-12-17T16:25:21.322879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:21.324963Z","iopub.execute_input":"2024-12-17T16:25:21.325467Z","iopub.status.idle":"2024-12-17T16:25:21.334611Z","shell.execute_reply.started":"2024-12-17T16:25:21.325439Z","shell.execute_reply":"2024-12-17T16:25:21.333751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to reshape a single row of the DataFrame\ndef reshape_row(row):\n    data = {'study_id': [], 'condition': [], 'level': [], 'severity': []}\n    \n    for column, value in row.items():\n        if column not in ['study_id', 'series_id', 'instance_number', 'x', 'y', 'series_description']:\n            parts = column.split('_')\n            condition = ' '.join([word.capitalize() for word in parts[:-2]])\n            level = parts[-2].capitalize() + '/' + parts[-1].capitalize()\n            data['study_id'].append(row['study_id'])\n            data['condition'].append(condition)\n            data['level'].append(level)\n            data['severity'].append(value)\n    \n    return pd.DataFrame(data)\n\n# Reshape the DataFrame for all rows\nnew_train_df = pd.concat([reshape_row(row) for _, row in train.iterrows()], ignore_index=True)\n\n# Display the first few rows of the reshaped dataframe\nnew_train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:21.340815Z","iopub.execute_input":"2024-12-17T16:25:21.341081Z","iopub.status.idle":"2024-12-17T16:25:22.426585Z","shell.execute_reply.started":"2024-12-17T16:25:21.341057Z","shell.execute_reply":"2024-12-17T16:25:22.425722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print columns in a neat way\nprint(\"\\nColumns in new_train_df:\")\nprint(\",\".join(new_train_df.columns))\n\nprint(\"\\nColumns in label:\")\nprint(\",\".join(label.columns))\n\nprint(\"\\nColumns in test_desc:\")\nprint(\",\".join(test_desc.columns))\n\nprint(\"\\nColumns in sub:\")\nprint(\",\".join(sub.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.427760Z","iopub.execute_input":"2024-12-17T16:25:22.428101Z","iopub.status.idle":"2024-12-17T16:25:22.433498Z","shell.execute_reply.started":"2024-12-17T16:25:22.428073Z","shell.execute_reply":"2024-12-17T16:25:22.432579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge the dataframes on the common columns\nmerged_df = pd.merge(new_train_df, label, on=['study_id', 'condition', 'level'], how='inner')\n# Merge the dataframes on the common column 'series_id'\nfinal_merged_df = pd.merge(merged_df, train_desc, on='series_id', how='inner')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.434483Z","iopub.execute_input":"2024-12-17T16:25:22.434743Z","iopub.status.idle":"2024-12-17T16:25:22.500329Z","shell.execute_reply.started":"2024-12-17T16:25:22.434718Z","shell.execute_reply":"2024-12-17T16:25:22.499304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge the dataframes on the common column 'series_id'\nfinal_merged_df = pd.merge(merged_df, train_desc, on=['series_id','study_id'], how='inner')\n# Display the first few rows of the final merged dataframe\nfinal_merged_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.501524Z","iopub.execute_input":"2024-12-17T16:25:22.501882Z","iopub.status.idle":"2024-12-17T16:25:22.525082Z","shell.execute_reply.started":"2024-12-17T16:25:22.501842Z","shell.execute_reply":"2024-12-17T16:25:22.524374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Create the row_id column\nfinal_merged_df['row_id'] = (\n    final_merged_df['study_id'].astype(str) + '_' +\n    final_merged_df['condition'].str.lower().str.replace(' ', '_') + '_' +\n    final_merged_df['level'].str.lower().str.replace('/', '_')\n)\n\n# Create the image_path column\nfinal_merged_df['image_path'] = (\n    f'{train_path}/train_images/' + \n    final_merged_df['study_id'].astype(str) + '/' +\n    final_merged_df['series_id'].astype(str) + '/' +\n    final_merged_df['instance_number'].astype(str) + '.dcm'\n)\n\n# Note: Check image path, since there's 1 instance id, for 1 image, but there's many more images other than the ones labelled in the instance ID. \n\n# Display the updated dataframe\nfinal_merged_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.526052Z","iopub.execute_input":"2024-12-17T16:25:22.526330Z","iopub.status.idle":"2024-12-17T16:25:22.669007Z","shell.execute_reply.started":"2024-12-17T16:25:22.526306Z","shell.execute_reply":"2024-12-17T16:25:22.668037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Normal/Mild\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.670088Z","iopub.execute_input":"2024-12-17T16:25:22.670364Z","iopub.status.idle":"2024-12-17T16:25:22.792259Z","shell.execute_reply.started":"2024-12-17T16:25:22.670337Z","shell.execute_reply":"2024-12-17T16:25:22.791340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Moderate\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.793587Z","iopub.execute_input":"2024-12-17T16:25:22.794045Z","iopub.status.idle":"2024-12-17T16:25:22.831766Z","shell.execute_reply.started":"2024-12-17T16:25:22.793999Z","shell.execute_reply":"2024-12-17T16:25:22.831057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the base path for test images\nbase_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images/'\n\n# Function to get image paths for a series\ndef get_image_paths(row):\n    series_path = os.path.join(base_path, str(row['study_id']), str(row['series_id']))\n    if os.path.exists(series_path):\n        return [os.path.join(series_path, f) for f in os.listdir(series_path) if os.path.isfile(os.path.join(series_path, f))]\n    return []\n\n# Mapping of series_description to conditions\ncondition_mapping = {\n    'Sagittal T1': {'left': 'left_neural_foraminal_narrowing', 'right': 'right_neural_foraminal_narrowing'},\n    'Axial T2': {'left': 'left_subarticular_stenosis', 'right': 'right_subarticular_stenosis'},\n    'Sagittal T2/STIR': 'spinal_canal_stenosis'\n}\n\n# Create a list to store the expanded rows\nexpanded_rows = []\n\n# Expand the dataframe by adding new rows for each file path\nfor index, row in test_desc.iterrows():\n    image_paths = get_image_paths(row)\n    conditions = condition_mapping.get(row['series_description'], {})\n    if isinstance(conditions, str):  # Single condition\n        conditions = {'left': conditions, 'right': conditions}\n    for side, condition in conditions.items():\n        for image_path in image_paths:\n            expanded_rows.append({\n                'study_id': row['study_id'],\n                'series_id': row['series_id'],\n                'series_description': row['series_description'],\n                'image_path': image_path,\n                'condition': condition,\n                'row_id': f\"{row['study_id']}_{condition}\"\n            })\n\n# Create a new dataframe from the expanded rows\nexpanded_test_desc = pd.DataFrame(expanded_rows)\n\n# Display the resulting dataframe\nexpanded_test_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.833004Z","iopub.execute_input":"2024-12-17T16:25:22.833275Z","iopub.status.idle":"2024-12-17T16:25:22.921522Z","shell.execute_reply.started":"2024-12-17T16:25:22.833249Z","shell.execute_reply":"2024-12-17T16:25:22.920788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# change severity column labels\n#Normal/Mild': 'normal_mild', 'Moderate': 'moderate', 'Severe': 'severe'}\nfinal_merged_df['severity'] = final_merged_df['severity'].map({'Normal/Mild': 'normal_mild', 'Moderate': 'moderate', 'Severe': 'severe'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.922933Z","iopub.execute_input":"2024-12-17T16:25:22.923273Z","iopub.status.idle":"2024-12-17T16:25:22.931422Z","shell.execute_reply.started":"2024-12-17T16:25:22.923247Z","shell.execute_reply":"2024-12-17T16:25:22.930729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = expanded_test_desc\ntrain_data = final_merged_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.932341Z","iopub.execute_input":"2024-12-17T16:25:22.932568Z","iopub.status.idle":"2024-12-17T16:25:22.943789Z","shell.execute_reply.started":"2024-12-17T16:25:22.932546Z","shell.execute_reply":"2024-12-17T16:25:22.942854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:46.724709Z","iopub.execute_input":"2024-12-17T16:25:46.724990Z","iopub.status.idle":"2024-12-17T16:25:46.738193Z","shell.execute_reply.started":"2024-12-17T16:25:46.724947Z","shell.execute_reply":"2024-12-17T16:25:46.737265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Define a function to check if a path exists\ndef check_exists(path):\n    return os.path.exists(path)\n\n# Define a function to check if a study ID directory exists\ndef check_study_id(row):\n    study_id = row['study_id']\n    path = f'{train_path}/train_images/{study_id}'\n    return check_exists(path)\n\n# Define a function to check if a series ID directory exists\ndef check_series_id(row):\n    study_id = row['study_id']\n    series_id = row['series_id']\n    path = f'{train_path}/train_images/{study_id}/{series_id}'\n    return check_exists(path)\n\n# Define a function to check if an image file exists\ndef check_image_exists(row):\n    image_path = row['image_path']\n    return check_exists(image_path)\n\n# Apply the functions to the train_data dataframe\ntrain_data['study_id_exists'] = train_data.apply(check_study_id, axis=1)\ntrain_data['series_id_exists'] = train_data.apply(check_series_id, axis=1)\ntrain_data['image_exists'] = train_data.apply(check_image_exists, axis=1)\n\n# Filter train_data\ntrain_data = train_data[(train_data['study_id_exists']) & (train_data['series_id_exists']) & (train_data['image_exists'])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:22.944760Z","iopub.execute_input":"2024-12-17T16:25:22.945093Z","iopub.status.idle":"2024-12-17T16:25:46.723716Z","shell.execute_reply.started":"2024-12-17T16:25:22.945066Z","shell.execute_reply":"2024-12-17T16:25:46.723024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['series_description'].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:46.739306Z","iopub.execute_input":"2024-12-17T16:25:46.739632Z","iopub.status.idle":"2024-12-17T16:25:46.756822Z","shell.execute_reply.started":"2024-12-17T16:25:46.739606Z","shell.execute_reply":"2024-12-17T16:25:46.756009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.dcmread(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:46.757914Z","iopub.execute_input":"2024-12-17T16:25:46.758191Z","iopub.status.idle":"2024-12-17T16:25:46.769324Z","shell.execute_reply.started":"2024-12-17T16:25:46.758166Z","shell.execute_reply":"2024-12-17T16:25:46.768431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load images randomly\nimport random\nimages = []\nrow_ids = []\nselected_indices = random.sample(range(len(train_data)), 2)\nfor i in selected_indices:\n    image = load_dicom(train_data['image_path'][i])\n    images.append(image)\n    row_ids.append(train_data['row_id'][i])\n\n# Plot images\nfig, ax = plt.subplots(1, 2, figsize=(8, 4))\nfor i in range(2):\n    ax[i].imshow(images[i], cmap='gray')\n    ax[i].set_title(f'Row ID: {row_ids[i]}', fontsize=8)\n    ax[i].axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:46.770437Z","iopub.execute_input":"2024-12-17T16:25:46.770700Z","iopub.status.idle":"2024-12-17T16:25:47.202606Z","shell.execute_reply.started":"2024-12-17T16:25:46.770676Z","shell.execute_reply":"2024-12-17T16:25:47.201782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:47.203812Z","iopub.execute_input":"2024-12-17T16:25:47.204487Z","iopub.status.idle":"2024-12-17T16:25:47.222361Z","shell.execute_reply.started":"2024-12-17T16:25:47.204447Z","shell.execute_reply":"2024-12-17T16:25:47.221458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data.dropna()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:47.223411Z","iopub.execute_input":"2024-12-17T16:25:47.223688Z","iopub.status.idle":"2024-12-17T16:25:47.250558Z","shell.execute_reply.started":"2024-12-17T16:25:47.223664Z","shell.execute_reply":"2024-12-17T16:25:47.249900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Save the DataFrame to \"clean_data.csv\" in the desired directory\ntrain_data.to_csv(\"/kaggle/working/clean_data.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:47.251693Z","iopub.execute_input":"2024-12-17T16:25:47.251951Z","iopub.status.idle":"2024-12-17T16:25:47.734733Z","shell.execute_reply.started":"2024-12-17T16:25:47.251926Z","shell.execute_reply":"2024-12-17T16:25:47.734019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_clean = pd.read_csv('/kaggle/working/clean_data.csv')\n\n# Define conditions for each group\ncondition_groups = {\n    'Spinal Canal Stenosis': ['Spinal Canal Stenosis'],\n    'Neural Foraminal Narrowing': ['Right Neural Foraminal Narrowing', 'Left Neural Foraminal Narrowing'],\n    'Subarticular Stenosis': ['Right Subarticular Stenosis', 'Left Subarticular Stenosis']\n}\n# Split and save to separate CSV files\nfor group_name, conditions in condition_groups.items():\n    # Filter rows based on condition\n    filtered_df = df_clean[df_clean['condition'].isin(conditions)]\n    # Save to new CSV file\n    group_name_save=group_name.replace(' ','_')\n    filtered_df.to_csv(f'{group_name_save}.csv', index=False)\n\nprint(\"CSV files have been split and saved.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:25:47.737740Z","iopub.execute_input":"2024-12-17T16:25:47.738001Z","iopub.status.idle":"2024-12-17T16:25:48.399090Z","shell.execute_reply.started":"2024-12-17T16:25:47.737958Z","shell.execute_reply":"2024-12-17T16:25:48.398199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nimport os\n\ndef cross_validation(csv_file, results_name):\n    # Load your CSV file into a pandas DataFrame\n    df = pd.read_csv(f'{csv_file}')\n\n    # Concatenate 'condition' and 'level' to create a unique class for each combination\n    df['condition_level'] = df['condition'] + '_' + df['level']\n    \n    # Now, we can assign numeric class labels if needed\n    df['class_id'] = df['condition_level'].astype('category').cat.codes\n\n    # Prepare for 10-fold stratified split\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n    # Split the data\n    df['fold'] = -1  # Initialize the fold column\n\n    # Assign fold numbers\n    for fold, (train_idx, val_idx) in enumerate(skf.split(df, df['class_id'])):\n        df.loc[val_idx, 'fold'] = fold\n\n    # Now, 'df' contains a 'fold' column that indicates the fold assignment (0-4)\n    # Save the new CSV with fold numbers if necessary\n    output_path = f'{results_name}.csv'\n    df.to_csv(output_path, index=False)\n\n    # Check if the file has been created\n    if os.path.exists(output_path):\n        print(f\"Data has been split into 5 folds and saved as '{output_path}'\")\n    else:\n        print(f\"Failed to save the file '{output_path}'\")\n\n# Run the cross validation function\ncross_validation('/kaggle/working/Spinal_Canal_Stenosis.csv', 'Spinal_Canal_Stenosis_folds')\ncross_validation('/kaggle/working/Neural_Foraminal_Narrowing.csv', 'Neural_Foraminal_Narrowing_folds')\ncross_validation('/kaggle/working/Subarticular_Stenosis.csv', 'Subarticular_Stenosis_folds')\n\n# List files in the current working directory to see if they were created\nfor file_name in os.listdir('/kaggle/working/'):\n    print(file_name)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:33:45.549296Z","iopub.execute_input":"2024-12-17T16:33:45.549579Z","iopub.status.idle":"2024-12-17T16:33:46.293097Z","shell.execute_reply.started":"2024-12-17T16:33:45.549553Z","shell.execute_reply":"2024-12-17T16:33:46.292198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport yaml\nimport cv2\nimport csv\nimport pandas as pd\nclass Detector_data_prepration:\n   \n    def __init__(\n        self,\n        dataset_directory='../train_images',\n        csv_directory='',\n\n        condition_level_classes={},\n        condition_name='',\n        val_fold=1,\n        width_box=16,\n        \n        ):\n\n        self.dataset_directory = dataset_directory\n\n        self.csv_directory=csv_directory\n\n\n        self.condition_level_classes=condition_level_classes\n\n        self.condition_name = condition_name\n\n        self.val_fold=val_fold\n\n        self.width_box=width_box\n\n\n        self.save_directory= f'./{self.condition_name}'\n\n\n        ## Create a folder for saving data \n        self.create_folder()\n\n\n        ## read data based on the cross validation and fold that define as a validation fold\n        self.read_cross_validation()\n\n\n        ### Convert train data to png for training data \n        self.dicom_to_png(self.training_data,self.train_image_path)\n#\n        self.save_weight_height_to_csv()\n\n        self.create_label_for_yolo(self.training_data, self.train_labels_path)\n\n\n        ### Convert train data to png for validation data \n        self.dicom_to_png(self.validation_data,self.val_images_path)\n#\n        self.save_weight_height_to_csv()\n\n        self.create_label_for_yolo(self.validation_data, self.val_labels_path)\n\n        self.creata_yaml_file()\n\n        '''\n\n            Due to the fact that in each study id we have a number of series id, and for each series id there are \n            number of instance id, we create a dictionary to read and load data in once to save time for loading the dcm.\n\n            The data will create like this :\n\n\n            for example :\n\n                study_id,series_id,instance_number\n                    1,       101,        1\n                    1,       101,        2\n                    1,       102,        3\n                    2,       201,        1\n                    2,       201,        2\n\n\n\n                data = [\n                        {1: {101: [1, 2], 102: [3]}},\n                        {2: {201: [1, 2]}}\n                    ]\n        '''\n    ##\n\n    def create_folder(self):\n\n        folder_path = Path(f'{self.save_directory}')\n\n        if not folder_path.exists():\n            folder_path.mkdir(parents=True, exist_ok=True)\n\n        \n        self.fold_path = Path(f'{self.save_directory}/fold_{self.val_fold}')\n\n        if not self.fold_path.exists():\n            self.fold_path.mkdir(parents=True, exist_ok=True)\n\n        self.dataset_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets')\n\n        if not self.dataset_path.exists():\n            self.dataset_path.mkdir(parents=True, exist_ok=True)       \n\n\n        folder_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets/train')\n\n        if not folder_path.exists():\n            folder_path.mkdir(parents=True, exist_ok=True)\n\n\n        self.train_image_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets/train/images')\n\n        if not self.train_image_path.exists():\n            self.train_image_path.mkdir(parents=True, exist_ok=True)\n\n        self.train_labels_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets/train/labels')\n\n        if not self.train_labels_path.exists():\n            self.train_labels_path.mkdir(parents=True, exist_ok=True)\n\n        folder_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets/val')\n\n        if not folder_path.exists():\n            folder_path.mkdir(parents=True, exist_ok=True)\n\n        self.val_images_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets/val/images')\n\n        if not self.val_images_path.exists():\n            self.val_images_path.mkdir(parents=True, exist_ok=True)\n\n        self.val_labels_path = Path(f'{self.save_directory}/fold_{self.val_fold}/datasets/val/labels')\n\n        if not self.val_labels_path.exists():\n            self.val_labels_path.mkdir(parents=True, exist_ok=True)\n\n    def read_cross_validation(self):\n\n        df = pd.read_csv(self.csv_directory)\n\n        # Filter the data for fold 1\n\n        self.validation_data = df[df['fold'] == self.val_fold]\n\n        print('len val',len(self.validation_data))\n\n        self.training_data = df[df['fold'] != self.val_fold]\n\n        print('len train',len(self.training_data))\n\n    def read_dicom(self,dicom_dir):\n        ds = pydicom.dcmread(dicom_dir)\n\n        image = ds.pixel_array\n\n        image = (image - image.min()) / (image.max() - image.min() +1e-6) * 255\n        image = np.stack([image]*3, axis=-1).astype('uint8')\n\n\n        return image\n\n    '''\n\n    In dicom to png we convert the dicom data to png and save it and also we calculate the \n    height and weight of image ( for creating the labels)\n\n\n    '''\n\n    def dicom_to_png(self,df,image_directory):\n\n        self.height_weight_info=[]\n\n\n        for study_id, study_group in df.groupby('study_id'):\n\n            for series_id, series_group in study_group.groupby('series_id'):\n\n\n                instances = series_group['instance_number'].tolist()\n\n                unique_instance= np.unique(instances)\n\n                ## Find the height and  weight and save it \n\n                dcm_diretory=f'{self.dataset_directory}/{study_id}/{series_id}/{instances[0]}.dcm'\n\n                dcm_image = self.read_dicom(dcm_diretory)\n\n                height, width, channels = dcm_image.shape\n\n\n                self.height_weight_info.append({'study_id':study_id , 'series_id': series_id, 'height': height , 'width':width})\n\n\n                for instance in unique_instance: \n                        \n                    dcm_diretory=f'{self.dataset_directory}/{study_id}/{series_id}/{instance}.dcm'\n\n                    dcm_image = self.read_dicom(dcm_diretory)\n\n                    height, width, channels = dcm_image.shape\n\n                    cv2.imwrite(f'./{image_directory}/{study_id}_{series_id}_{instance}.png', dcm_image)\n\n\n    '''\n        In this part we save the height and weight of each subject\n\n    '''\n\n    def save_weight_height_to_csv(self):\n\n\n        csv_file = f'./{self.fold_path}/{self.condition_name}_height_weight.csv'\n\n        # Write the data to a CSV file\n        with open(csv_file, mode='w', newline='') as file:\n            # Create a CSV DictWriter object\n            writer = csv.DictWriter(file, fieldnames=self.height_weight_info[0].keys())\n            \n            # Write the header (field names)\n            writer.writeheader()\n            \n            # Write the rows (each dictionary)\n            writer.writerows(self.height_weight_info)\n\n        print(f\"Data saved to {csv_file}\")\n\n        # Load the first CSV file\n        file2 = pd.read_csv(f'./{self.fold_path}/{self.condition_name}_height_weight.csv')\n\n        # Load the second CSV file\n        file1 = pd.read_csv(self.csv_directory)\n\n        # Merge the two DataFrames based on 'study_id' and 'series_id'\n        # This will keep all rows from file1 and add 'height' and 'weight' from file2\n        merged_data = pd.merge(file1, file2[['study_id', 'series_id', 'height', 'width']],\n                            on=['study_id', 'series_id'], how='left')\n\n        # Save the merged data into a new CSV file\n        merged_data.to_csv(f'./{self.fold_path}/{self.condition_name}_height_weight.csv', index=False)\n\n        print(\"Merged data has been saved to 'updated_file.csv'\")\n\n    def find_class_label(self,condition,level):\n\n        condition=condition.replace(' ','_')\n        level=level.replace('/','_')\n\n        condtion_level=f'{condition}_{level}'\n\n\n        return self.condition_level_classes[condtion_level]\n\n    \n    def create_label_for_yolo(self,df,labels_directory):\n\n        df = pd.read_csv(f'./{self.fold_path}/{self.condition_name}_height_weight.csv')\n\n        for study_id, study_group in df.groupby('study_id'):\n\n            for series_id, series_group in study_group.groupby('series_id'):\n\n                for instances_number ,instances_groups in series_group.groupby('instance_number'):\n                    labels=[]\n\n                    for (condition, level), group in instances_groups.groupby(['condition', 'level']):\n\n\n                        x=group['x']\n\n                        y= group['y']\n\n                        height=group['height']\n                        \n                        width=group['width']\n\n\n                        condtion_calss=self.find_class_label(condition,level)\n\n\n                        labels.append({\n                        'class_id':condtion_calss,\n                        'x': float(x/width),\n                        'y': float(y/height),\n                        'width': float(self.width_box/width),\n                        'height': float(self.width_box/height),\n                    \n                        })\n                    \n                    \n                    output_file=f'{labels_directory}/{study_id}_{series_id}_{instances_number}.txt'\n                    \n                    with open(output_file, 'w') as file:\n                        for item in labels:\n\n                            class_id = item['class_id']\n                            x = item['x']\n                            y = item['y']\n                            width = item['width']\n                            height = item['height']\n\n                            file.write(f\"{class_id} {x} {y} {width} {height}\\n\")    \n\n    def creata_yaml_file(self):\n        \n        yaml_file_path = f'{self.dataset_path}/yolo_config.yaml'\n\n        num_classes = len(self.condition_level_classes)\n\n        # Prepare the YAML data\n        yaml_data = {\n            'train':'./train' ,  # Assuming training images are in JPG format\n            'val': './val',    # Assuming validation images are also in JPG format\n            'nc': num_classes,\n            'names': [name for name in self.condition_level_classes.keys()]\n        }\n        \n        # Write the YAML file\n        with open(yaml_file_path, 'w') as file:\n            yaml.dump(yaml_data, file, default_flow_style=False)\n\n\nprint(\"ok\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:38:15.117311Z","iopub.execute_input":"2024-12-17T16:38:15.117662Z","iopub.status.idle":"2024-12-17T16:38:15.142819Z","shell.execute_reply.started":"2024-12-17T16:38:15.117625Z","shell.execute_reply":"2024-12-17T16:38:15.141938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#### Spinal canal stenosis \n\nfor fold in range(5):\n\n    condition_level_classes_spinal_canal={\n\n        'Spinal_Canal_Stenosis_L1_L2':0,\n\n        'Spinal_Canal_Stenosis_L2_L3':1,\n\n        'Spinal_Canal_Stenosis_L3_L4':2,\n\n        'Spinal_Canal_Stenosis_L4_L5':3,    \n\n        'Spinal_Canal_Stenosis_L5_S1':4,\n\n    }\n\n    d=Detector_data_prepration(\n\n        dataset_directory='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images',\n\n        csv_directory='/kaggle/working/Spinal_Canal_Stenosis_folds.csv',\n\n        condition_name='Spinal_Canal_Stenosis',\n        \n        condition_level_classes=condition_level_classes_spinal_canal,\n        \n        val_fold=fold,\n        \n        width_box=16,\n        \n        )\n    \nprint('Spinal canal stenosis  DONE')\n\n#### Subarticular_Stenosis\n\nfor fold in range(5):\n\n    condition_level_classes_spinal_canal={\n\n        'Left_Subarticular_Stenosis_L1_L2':0,\n\n        'Left_Subarticular_Stenosis_L2_L3':1,\n\n        'Left_Subarticular_Stenosis_L3_L4':2,\n\n        'Left_Subarticular_Stenosis_L4_L5':3,    \n\n        'Left_Subarticular_Stenosis_L5_S1':4,\n\n\n        'Right_Subarticular_Stenosis_L1_L2':5,\n\n        'Right_Subarticular_Stenosis_L2_L3':6,\n\n        'Right_Subarticular_Stenosis_L3_L4':7,\n\n        'Right_Subarticular_Stenosis_L4_L5':8,    \n\n        'Right_Subarticular_Stenosis_L5_S1':9,\n\n    }\n    \n    d=Detector_data_prepration(\n\n        dataset_directory='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images',\n\n        csv_directory='/kaggle/working/Subarticular_Stenosis_folds.csv',\n\n        condition_name='Subarticular_Stenosis',\n        \n        condition_level_classes=condition_level_classes_spinal_canal,\n\n        val_fold=fold,\n\n        width_box=16,\n        \n        )\n\nprint('Subarticular_Stenosis  DONE')\n\n# ####  Neural Foraminal Narrowing\n\nfor fold in range(5):\n\n    condition_level_classes_spinal_canal={\n\n        'Left_Neural_Foraminal_Narrowing_L1_L2':0,\n\n        'Left_Neural_Foraminal_Narrowing_L2_L3':1,\n\n        'Left_Neural_Foraminal_Narrowing_L3_L4':2,\n\n        'Left_Neural_Foraminal_Narrowing_L4_L5':3,    \n\n        'Left_Neural_Foraminal_Narrowing_L5_S1':4,\n\n\n        'Right_Neural_Foraminal_Narrowing_L1_L2':5,\n\n        'Right_Neural_Foraminal_Narrowing_L2_L3':6,\n\n        'Right_Neural_Foraminal_Narrowing_L3_L4':7,\n\n        'Right_Neural_Foraminal_Narrowing_L4_L5':8,    \n\n        'Right_Neural_Foraminal_Narrowing_L5_S1':9,\n\n    }\n\n    d=Detector_data_prepration(\n        \n        dataset_directory='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images',\n\n        csv_directory='/kaggle/working/Neural_Foraminal_Narrowing_folds.csv',\n\n        condition_name='Neural_Foraminal_Narrowing',\n        \n        condition_level_classes=condition_level_classes_spinal_canal,\n\n        val_fold=fold,\n\n        width_box=16,\n\n        )\n\nprint(' Neural Foraminal Narrowing  DONE')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:41:55.892241Z","iopub.execute_input":"2024-12-17T16:41:55.892592Z","iopub.status.idle":"2024-12-17T16:44:05.729285Z","shell.execute_reply.started":"2024-12-17T16:41:55.892562Z","shell.execute_reply":"2024-12-17T16:44:05.728023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nfrom matplotlib import pyplot as plt\nfrom PIL import Image\n\nfrom pathlib import Path\nimport yaml\n\nclass yolo_training:\n\n\n    def __init__(self,\n    \n        data_directory='',\n\n        condition='',\n        \n        fold=0,\n\n\n        results_directory='',\n\n        epochs=500,\n\n        patience=20,\n\n        batch=4,\n\n    \n    ):\n\n\n        self.condition=condition\n\n        self.fold=fold\n\n\n        self.results_directory=results_directory\n\n        self.data_directory=data_directory\n\n        self.epochs=epochs\n        self.patience=patience\n        self.batch=batch\n\n\n        # the yolo that we need for training\n\n        self.yolo_config_yaml=f'{self.data_directory}/{self.condition}/fold_{self.fold}/datasets/yolo_config.yaml'\n\n\n        ## Create a folder \n\n\n        self.update_yaml()\n\n        model=self.load_pretrain_model()\n\n        self.training(model)\n\n\n\n    def load_pretrain_model(self):\n\n        model = YOLO('yolov8n.yaml')  # build a new model from YAML\n\n        model = YOLO('yolov8n.pt')\n\n        return model\n\n\n\n    '''\n        We clear the dataset directory and let the yaml file that will read for training update that\n\n        The resoan is that sone time the yaml file do not let update \n\n    '''\n\n    def update_yaml(self):\n\n\n        # Define the path to your settings file\n        settings_file = '/users/amousavi/.config/Ultralytics/settings.yaml'\n\n        # Define the new dataset directory\n        new_dataset_dir=''\n        # Load the YAML file\n        with open(settings_file, 'r') as file:\n            settings = yaml.safe_load(file)\n\n        # Update the dataset directory in the YAML configuration\n        settings['datasets_dir'] = new_dataset_dir\n\n        # Save the updated YAML back to the file\n        with open(settings_file, 'w') as file:\n            yaml.dump(settings, file, default_flow_style=False)\n\n        print(f\"Dataset directory has been updated to: {new_dataset_dir}\")\n\n\n        self.result_condition_fold_directory=f'{self.results_directory}/{self.condition}/{self.fold}'\n\n    def training(self,model):\n\n        result_condition_path = Path(f'{self.results_directory}/{self.condition}')\n\n        if not result_condition_path.exists():\n            result_condition_path.mkdir(parents=True, exist_ok=True)\n\n\n        result_condition_fold_path = Path(f'{self.results_directory}/{self.condition}/fold_{self.fold}')\n\n        if not result_condition_fold_path.exists():\n            result_condition_fold_path.mkdir(parents=True, exist_ok=True)\n\n\n        #Define subdirectory for this specific training\n        name = \"epochs-\" \n\n        results = model.train(data=self.yolo_config_yaml,\n                            project=result_condition_fold_path,\n                            name=name,\n                            epochs=self.epochs,\n                            patience=self.patience, #I am setting patience=0 to disable early stopping.\n                            batch=self.batch,\n                            )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nimport torch\nimport torch.optim.lr_scheduler as lr_scheduler\nfrom tqdm import tqdm\n\n# Define a custom dataset class\nclass CustomDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, index):\n        image_path = self.dataframe['image_path'][index]\n        image = load_dicom(image_path)  # Define this function to load your DICOM images\n        label = self.dataframe['severity'][index]\n        \n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n\"\"\"# Function to create datasets and dataloaders for each series description\ndef create_datasets_and_loaders(df, series_description, transform, batch_size=8):\n    filtered_df = df[df['series_description'] == series_description]\n    \n    train_df, val_df = train_test_split(filtered_df, test_size=0.2, random_state=42)\n    train_df = train_df.reset_index(drop=True)\n    val_df = val_df.reset_index(drop=True)\n\n    train_dataset = CustomDataset(train_df, transform)\n    val_dataset = CustomDataset(val_df, transform)\n\n    trainloader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n    valloader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\n    \n    return trainloader, valloader, len(train_df), len(val_df)\"\"\"\n# Function to create datasets and dataloaders for each series description\ndef create_datasets_and_loaders(df, series_description, transform, batch_size=8):\n    filtered_df = df[df['series_description'] == series_description]\n    \n    # %5'ini al frac değerini değiştirerek trainde verinin ne kadarını kullanacagını belirlersin\n    filtered_df = filtered_df.sample(frac=1.0, random_state=42)  \n    \n    train_df, val_df = train_test_split(filtered_df, test_size=0.2, random_state=42)\n    train_df = train_df.reset_index(drop=True)\n    val_df = val_df.reset_index(drop=True)\n\n    train_dataset = CustomDataset(train_df, transform)\n    val_dataset = CustomDataset(val_df, transform)\n\n    trainloader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n    valloader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\n    \n    return trainloader, valloader, len(train_df), len(val_df)\n\n\n# Define the transforms\ntransform = transforms.Compose([\n    transforms.Lambda(lambda x: (x * 255).astype(np.uint8)),  # Convert back to uint8 for PIL\n    transforms.ToPILImage(),\n    transforms.Resize((224, 224)),\n    transforms.Grayscale(num_output_channels=3),\n    transforms.ToTensor(),\n])\n\n# Create dataloaders for each series description\ndataloaders = {}\nlengths = {}\n\ntrainloader_t1, valloader_t1, len_train_t1, len_val_t1 = create_datasets_and_loaders(train_data, 'Sagittal T1', transform)\ntrainloader_t2, valloader_t2, len_train_t2, len_val_t2 = create_datasets_and_loaders(train_data, 'Axial T2', transform)\ntrainloader_t2stir, valloader_t2stir, len_train_t2stir, len_val_t2stir = create_datasets_and_loaders(train_data, 'Sagittal T2/STIR', transform)\n\ndataloaders['Sagittal T1'] = (trainloader_t1, valloader_t1)\ndataloaders['Axial T2'] = (trainloader_t2, valloader_t2)\ndataloaders['Sagittal T2/STIR'] = (trainloader_t2stir, valloader_t2stir)\n\nlengths['Sagittal T1'] = (len_train_t1, len_val_t1)\nlengths['Axial T2'] = (len_train_t2, len_val_t2)\nlengths['Sagittal T2/STIR'] = (len_train_t2stir, len_val_t2stir)\n\n# Dictionary mapping labels to indices\nlabel_map = {'Mild': 0, 'Moderate': 1, 'Severe': 2}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:35.477575Z","iopub.execute_input":"2024-12-08T21:16:35.477854Z","iopub.status.idle":"2024-12-08T21:16:35.530739Z","shell.execute_reply.started":"2024-12-08T21:16:35.47783Z","shell.execute_reply":"2024-12-08T21:16:35.529753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Function to visualize a batch of images\ndef visualize_batch(dataloader):\n    images, labels = next(iter(dataloader))\n    fig, axes = plt.subplots(1, len(images), figsize=(20, 5))\n    for i, (img, lbl) in enumerate(zip(images, labels)):\n        ax = axes[i]\n        img = img.permute(1, 2, 0)  # Convert to HWC for visualization\n        ax.imshow(img)\n        ax.set_title(f\"Label: {lbl}\")\n        ax.axis('off')\n    plt.show()\n\n# Visualize samples from each dataloader\nprint(\"Visualizing Sagittal T1 samples\")\nvisualize_batch(trainloader_t1)\nprint(\"Visualizing Axial T2 samples\")\nvisualize_batch(trainloader_t2)\nprint(\"Visualizing Sagittal T2/STIR samples\")\nvisualize_batch(trainloader_t2stir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:35.532017Z","iopub.execute_input":"2024-12-08T21:16:35.532682Z","iopub.status.idle":"2024-12-08T21:16:37.63418Z","shell.execute_reply.started":"2024-12-08T21:16:35.53264Z","shell.execute_reply":"2024-12-08T21:16:37.633271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nimage, label = next(iter(trainloader_t2))\nsample = image[1].permute(1, 2, 0)  #sample\n\n# Plot images\nplt.figsize=(8, 4)\nplt.imshow(images[0], cmap='gray')\nplt.title(label[0])\nplt.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:37.635228Z","iopub.execute_input":"2024-12-08T21:16:37.635498Z","iopub.status.idle":"2024-12-08T21:16:38.184335Z","shell.execute_reply.started":"2024-12-08T21:16:37.63547Z","shell.execute_reply":"2024-12-08T21:16:38.183514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nfrom tqdm import tqdm\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:38.18554Z","iopub.execute_input":"2024-12-08T21:16:38.185903Z","iopub.status.idle":"2024-12-08T21:16:38.191788Z","shell.execute_reply.started":"2024-12-08T21:16:38.185864Z","shell.execute_reply":"2024-12-08T21:16:38.190899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nfrom tqdm import tqdm\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n\n\nclass CustomResNet50(nn.Module):\n    def __init__(self, num_classes=3, pretrained_weights=None):\n        super(CustomResNet50, self).__init__()\n        # pretrained=False ile modelin rastgele ağırlıklarla başlatılmasını sağla\n        self.model = models.resnet50(pretrained=False)\n        \n        # Eğer manuel ağırlık yolu verilmişse, bu ağırlıkları yükle\n        if pretrained_weights:\n            self.model.load_state_dict(torch.load(pretrained_weights))\n        \n        num_ftrs = self.model.fc.in_features  # Son katmanın özellik sayısını al\n        self.model.fc = nn.Linear(num_ftrs, num_classes)  # Son katmanı değiştir\n\n    def forward(self, x):\n        return self.model(x)\n\n    def unfreeze_model(self):\n        \"\"\"Tüm katmanları çöz.\"\"\"\n        for param in self.model.parameters():\n            param.requires_grad = True\n\n    def unfreeze_specific_layers(self, layer_names=None):\n        \"\"\"\n        Belirli katmanları çözmek için kullanılabilir.\n        Eğer layer_names None ise, tüm katmanlar çözülür.\n        \"\"\"\n        for name, param in self.model.named_parameters():\n            if layer_names is None or any(layer in name for layer in layer_names):\n                param.requires_grad = True\n            else:\n                param.requires_grad = False\n\n# Cihaz seçimi\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Modeli başlat\nsagittal_t1_model = CustomResNet50(num_classes=3).to(device)\naxial_t2_model = CustomResNet50(num_classes=3).to(device)\nsagittal_t2stir_model = CustomResNet50(num_classes=3).to(device)\n\n\"\"\"# Son fully connected katmanı çözme\nfor param in sagittal_t1_model.model.fc.parameters():\n    param.requires_grad = True\nfor param in axial_t2_model.model.fc.parameters():\n    param.requires_grad = True\nfor param in sagittal_t2stir_model.model.fc.parameters():\n    param.requires_grad = True\n\n# Başlangıç katmanlarını dondurma\nfor param in sagittal_t1_model.model.parameters():\n    if param is not sagittal_t1_model.model.fc.weight and param is not sagittal_t1_model.model.fc.bias:\n        param.requires_grad = False\n\nfor param in axial_t2_model.model.parameters():\n    if param is not axial_t2_model.model.fc.weight and param is not axial_t2_model.model.fc.bias:\n        param.requires_grad = False\n\nfor param in sagittal_t2stir_model.model.parameters():\n    if param is not sagittal_t2stir_model.model.fc.weight and param is not sagittal_t2stir_model.model.fc.bias:\n        param.requires_grad = False\n\n# Eğitim parametreleri\ncriterion = nn.CrossEntropyLoss()\"\"\"\n# Tüm katmanları çözmek için\nfor model in [sagittal_t1_model, axial_t2_model, sagittal_t2stir_model]:\n    model.unfreeze_model()  # Bütün katmanları çöz\n\n# Eğitim parametreleri 05.12.2024 saat 0423'de güncellendi.\nweights = torch.tensor([1.0, 2.0, 4.0])\ncriterion = nn.CrossEntropyLoss(weight=weights.to(device))\n\n# Optimizer ayarları\noptimizer_sagittal_t1 = torch.optim.Adam(sagittal_t1_model.model.fc.parameters(), lr=0.001)\noptimizer_axial_t2 = torch.optim.Adam(axial_t2_model.model.fc.parameters(), lr=0.001)\noptimizer_sagittal_t2stir = torch.optim.Adam(sagittal_t2stir_model.model.fc.parameters(), lr=0.001)\n\n# Modelleri ve optimizörleri saklamak için dictionary\nmodels = {\n    'Sagittal T1': sagittal_t1_model,\n    'Axial T2': axial_t2_model,\n    'Sagittal T2/STIR': sagittal_t2stir_model,\n}\n\noptimizers = {\n    'Sagittal T1': optimizer_sagittal_t1,\n    'Axial T2': optimizer_axial_t2,\n    'Sagittal T2/STIR': optimizer_sagittal_t2stir,\n}\n\n\n# Eğitim yapılabilir parametrelerin sayısını yazdır\nfor model_name, model in models.items():\n    trainable_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\n    print(f\"Trainable parameters for {model_name}: {trainable_params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:38.192989Z","iopub.execute_input":"2024-12-08T21:16:38.193248Z","iopub.status.idle":"2024-12-08T21:16:39.822921Z","shell.execute_reply.started":"2024-12-08T21:16:38.193224Z","shell.execute_reply":"2024-12-08T21:16:39.822048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_map = {'normal_mild': 0, 'moderate': 1, 'severe': 2}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:39.823964Z","iopub.execute_input":"2024-12-08T21:16:39.824217Z","iopub.status.idle":"2024-12-08T21:16:39.828112Z","shell.execute_reply.started":"2024-12-08T21:16:39.824192Z","shell.execute_reply":"2024-12-08T21:16:39.827254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for images, labels in trainloader_t2:\n    labels = torch.tensor([label_map[label] for label in labels])\n    labels = labels.to(device)\n    print(labels)\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:39.829045Z","iopub.execute_input":"2024-12-08T21:16:39.829278Z","iopub.status.idle":"2024-12-08T21:16:39.958563Z","shell.execute_reply.started":"2024-12-08T21:16:39.829256Z","shell.execute_reply":"2024-12-08T21:16:39.957674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.optim.lr_scheduler as lr_scheduler\nfrom copy import deepcopy\n\ndef train_model(model, trainloader, valloader, len_train, len_val, optimizer, num_epochs=10, patience=3):\n    # Learning rate scheduler\n    scheduler = lr_scheduler.StepLR(optimizer, step_size=2, gamma=0.1)\n    \n    best_val_acc = 0.0\n    best_model_wts = deepcopy(model.state_dict())\n    counter = 0\n    \n    for epoch in range(num_epochs):\n        model.train()\n        train_loss = 0\n        correct_train = 0\n        \n        with tqdm(trainloader, unit=\"batch\") as tepoch:\n            for images, labels in tepoch:\n                images, labels = images.to(device), torch.tensor([label_map[label] for label in labels]).to(device)\n                optimizer.zero_grad()\n                outputs = model(images)\n                loss = criterion(outputs, labels)\n                loss.backward()\n                optimizer.step()\n                train_loss += loss.item()\n                \n                probabilities = torch.softmax(outputs, dim=1)\n                _, predicted = torch.max(probabilities, 1)\n                correct_train += (predicted == labels).sum().item()\n                \n                tepoch.set_postfix(epoch=epoch+1)\n        \n        scheduler.step()\n        \n        train_loss /= len(trainloader)\n        train_acc = 100 * correct_train / len_train\n        \n        model.eval()\n        val_loss, correct_val = 0, 0\n        with torch.no_grad():\n            with tqdm(valloader, unit=\"batch\") as vepoch:\n                for images, labels in vepoch:\n                    images, labels = images.to(device), torch.tensor([label_map[label] for label in labels]).to(device)\n                    outputs = model(images)\n                    loss = criterion(outputs, labels)\n                    val_loss += loss.item()\n                    \n                    probabilities = torch.softmax(outputs, dim=1)\n\n                    # Eğer batch size 1 ise, dim=0 kullanarak doğru boyutta işlem yapabilirsiniz\n                    if probabilities.dim() == 1:\n                        _, predicted = torch.max(probabilities, 0)  # batch size 1 ise dim=0\n                    else:\n                        _, predicted = torch.max(probabilities, 1)  # normal durumda dim=1\n                    correct_val += (predicted == labels).sum().item()\n                    \n                    vepoch.set_postfix(epoch=epoch+1)\n        \n        val_loss /= len(valloader)\n        val_acc = 100 * correct_val / len_val\n        \n        print(f\"Epoch {epoch+1}, Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.2f}%, Val Loss: {val_loss:.4f}, Val Acc: {val_acc:.2f}%\")\n        \n        # Save the best model and check for early stopping\n        if val_acc > best_val_acc:\n            best_val_acc = val_acc\n            best_model_wts = deepcopy(model.state_dict())\n            counter = 0\n            torch.save(best_model_wts, f'best_model_{epoch+1}.pth')\n        else:\n            counter += 1\n        \n        # Early stopping\n        if counter >= patience:\n            print(f\"Early stopping triggered after {epoch+1} epochs\")\n            break\n    \n    # Load best model weights\n    model.load_state_dict(best_model_wts)\n    return model, best_val_acc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:39.959578Z","iopub.execute_input":"2024-12-08T21:16:39.959863Z","iopub.status.idle":"2024-12-08T21:16:39.971166Z","shell.execute_reply.started":"2024-12-08T21:16:39.959838Z","shell.execute_reply":"2024-12-08T21:16:39.970479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training all models\nfor desc, model in models.items():\n    if desc == 'Sagittal T1':\n        trainloader, valloader, len_train, len_val = trainloader_t1, valloader_t1, len_train_t1, len_val_t1\n    elif desc == 'Axial T2':\n        trainloader, valloader, len_train, len_val = trainloader_t2, valloader_t2, len_train_t2, len_val_t2\n    elif desc == 'Sagittal T2/STIR':\n        trainloader, valloader, len_train, len_val = trainloader_t2stir, valloader_t2stir, len_train_t2stir, len_val_t2stir\n    \n    print(f\"Training model for {desc}\")\n    train_model(model, trainloader, valloader, len_train, len_val, optimizers[desc])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T21:16:39.971897Z","iopub.execute_input":"2024-12-08T21:16:39.972162Z","iopub.status.idle":"2024-12-08T22:14:31.819436Z","shell.execute_reply.started":"2024-12-08T21:16:39.972135Z","shell.execute_reply":"2024-12-08T22:14:31.818519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['level'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.820878Z","iopub.execute_input":"2024-12-08T22:14:31.821466Z","iopub.status.idle":"2024-12-08T22:14:31.831116Z","shell.execute_reply.started":"2024-12-08T22:14:31.821425Z","shell.execute_reply":"2024-12-08T22:14:31.830357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expanded_test_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.832147Z","iopub.execute_input":"2024-12-08T22:14:31.832451Z","iopub.status.idle":"2024-12-08T22:14:31.846328Z","shell.execute_reply.started":"2024-12-08T22:14:31.832426Z","shell.execute_reply":"2024-12-08T22:14:31.845526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"levels = ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']\n\n# Function to update row_id with levels\ndef update_row_id(row, levels):\n    level = levels[row.name % len(levels)]\n    return f\"{row['study_id']}_{row['condition']}_{level}\"\n\n# Update row_id in expanded_test_desc to include levels\nexpanded_test_desc['row_id'] = expanded_test_desc.apply(lambda row: update_row_id(row, levels), axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.847354Z","iopub.execute_input":"2024-12-08T22:14:31.847693Z","iopub.status.idle":"2024-12-08T22:14:31.864539Z","shell.execute_reply.started":"2024-12-08T22:14:31.847649Z","shell.execute_reply":"2024-12-08T22:14:31.863828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expanded_test_desc.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.865357Z","iopub.execute_input":"2024-12-08T22:14:31.865671Z","iopub.status.idle":"2024-12-08T22:14:31.879874Z","shell.execute_reply.started":"2024-12-08T22:14:31.865634Z","shell.execute_reply":"2024-12-08T22:14:31.879091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a custom test dataset class\nclass TestDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, index):\n        image_path = self.dataframe['image_path'][index]\n        image = load_dicom(image_path)  # Define this function to load your DICOM images\n        if self.transform:\n            image = self.transform(image)\n        return image\n\n# Define the transforms\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((224, 224)),\n    transforms.Grayscale(num_output_channels=3),\n    transforms.ToTensor(),\n])\n\n# Create a test dataset and dataloader\ntest_dataset = TestDataset(expanded_test_desc, transform)\ntestloader = DataLoader(test_dataset, batch_size=1, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.880893Z","iopub.execute_input":"2024-12-08T22:14:31.881154Z","iopub.status.idle":"2024-12-08T22:14:31.889539Z","shell.execute_reply.started":"2024-12-08T22:14:31.881127Z","shell.execute_reply":"2024-12-08T22:14:31.888668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for image in testloader:\n    print(image.shape)\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.890502Z","iopub.execute_input":"2024-12-08T22:14:31.891062Z","iopub.status.idle":"2024-12-08T22:14:31.920945Z","shell.execute_reply.started":"2024-12-08T22:14:31.891023Z","shell.execute_reply":"2024-12-08T22:14:31.92023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to get the model based on series_description\ndef get_model(series_description):\n    return models.get(series_description, None)\n\n# Function to make predictions on the test data\ndef predict_test_data(testloader, expanded_test_desc):\n    predictions = []\n    normal_mild_probs = []\n    moderate_probs = []\n    severe_probs = []\n    \n    for model in models.values():\n        model.eval()\n        \n    with torch.no_grad():\n        for idx, images in enumerate(tqdm(testloader)):\n            images = images.to(device)\n            series_description = expanded_test_desc.iloc[idx]['series_description']\n            model = get_model(series_description)\n            if model:\n                model.eval()  # Set the model to eval mode\n                outputs = model(images)\n                probs = torch.softmax(outputs, dim=1).squeeze(0)\n                normal_mild_probs.append(probs[0].item())\n                moderate_probs.append(probs[1].item())\n                severe_probs.append(probs[2].item())\n                predictions.append(probs)\n            else:\n                normal_mild_probs.append(None)\n                moderate_probs.append(None)\n                severe_probs.append(None)\n                predictions.append(None)\n    return normal_mild_probs, moderate_probs, severe_probs, predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.921767Z","iopub.execute_input":"2024-12-08T22:14:31.922018Z","iopub.status.idle":"2024-12-08T22:14:31.930103Z","shell.execute_reply.started":"2024-12-08T22:14:31.921991Z","shell.execute_reply":"2024-12-08T22:14:31.929286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test data\nnormal_mild_probs, moderate_probs, severe_probs, test_predictions = predict_test_data(testloader, expanded_test_desc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:31.93113Z","iopub.execute_input":"2024-12-08T22:14:31.931394Z","iopub.status.idle":"2024-12-08T22:14:36.113313Z","shell.execute_reply.started":"2024-12-08T22:14:31.93137Z","shell.execute_reply":"2024-12-08T22:14:36.112448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.114457Z","iopub.execute_input":"2024-12-08T22:14:36.114758Z","iopub.status.idle":"2024-12-08T22:14:36.121458Z","shell.execute_reply.started":"2024-12-08T22:14:36.114732Z","shell.execute_reply":"2024-12-08T22:14:36.120646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add predictions and probabilities to the test DataFrame\nexpanded_test_desc['normal_mild'] = normal_mild_probs\nexpanded_test_desc['moderate'] = moderate_probs\nexpanded_test_desc['severe'] = severe_probs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.122787Z","iopub.execute_input":"2024-12-08T22:14:36.123089Z","iopub.status.idle":"2024-12-08T22:14:36.131635Z","shell.execute_reply.started":"2024-12-08T22:14:36.123061Z","shell.execute_reply":"2024-12-08T22:14:36.130791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = expanded_test_desc[[\"row_id\",\"normal_mild\",\"moderate\",\"severe\"]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.132703Z","iopub.execute_input":"2024-12-08T22:14:36.132975Z","iopub.status.idle":"2024-12-08T22:14:36.142696Z","shell.execute_reply.started":"2024-12-08T22:14:36.132947Z","shell.execute_reply":"2024-12-08T22:14:36.14196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.143709Z","iopub.execute_input":"2024-12-08T22:14:36.144Z","iopub.status.idle":"2024-12-08T22:14:36.160824Z","shell.execute_reply.started":"2024-12-08T22:14:36.143974Z","shell.execute_reply":"2024-12-08T22:14:36.160032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'row_id' and sum the values\ngrouped_submission = submission.groupby('row_id').max().reset_index()\n\n# Normalize the columns\ngrouped_submission[['normal_mild', 'moderate', 'severe']] = grouped_submission[['normal_mild', 'moderate', 'severe']].div(grouped_submission[['normal_mild', 'moderate', 'severe']].sum(axis=1), axis=0)\n\n# Check the first 3 rows\ngrouped_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.161785Z","iopub.execute_input":"2024-12-08T22:14:36.162048Z","iopub.status.idle":"2024-12-08T22:14:36.183948Z","shell.execute_reply.started":"2024-12-08T22:14:36.162021Z","shell.execute_reply":"2024-12-08T22:14:36.183116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(grouped_submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.18491Z","iopub.execute_input":"2024-12-08T22:14:36.185221Z","iopub.status.idle":"2024-12-08T22:14:36.196469Z","shell.execute_reply.started":"2024-12-08T22:14:36.185175Z","shell.execute_reply":"2024-12-08T22:14:36.195685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub[['normal_mild', 'moderate', 'severe']] = grouped_submission[['normal_mild', 'moderate', 'severe']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.197443Z","iopub.execute_input":"2024-12-08T22:14:36.197767Z","iopub.status.idle":"2024-12-08T22:14:36.207567Z","shell.execute_reply.started":"2024-12-08T22:14:36.197732Z","shell.execute_reply":"2024-12-08T22:14:36.20683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Save the DataFrame to \"submission.csv\" in the desired directory\nsub.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.208588Z","iopub.execute_input":"2024-12-08T22:14:36.208951Z","iopub.status.idle":"2024-12-08T22:14:36.219663Z","shell.execute_reply.started":"2024-12-08T22:14:36.208916Z","shell.execute_reply":"2024-12-08T22:14:36.21894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T22:14:36.220545Z","iopub.execute_input":"2024-12-08T22:14:36.220911Z","iopub.status.idle":"2024-12-08T22:14:36.233746Z","shell.execute_reply.started":"2024-12-08T22:14:36.220882Z","shell.execute_reply":"2024-12-08T22:14:36.232918Z"}},"outputs":[],"execution_count":null}]}