{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":992,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":846,"modelId":101}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ncounter = 0  # Sayaç başlat\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        counter += 1\n        if counter == 15:  # 5 dosya yazdırdıktan sonra dur\n            break\n    if counter == 15:  # İç döngü kırıldığında dış döngüyü de kır\n        break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:03:16.636201Z","iopub.execute_input":"2025-01-07T20:03:16.636502Z","iopub.status.idle":"2025-01-07T20:03:18.214913Z","shell.execute_reply.started":"2025-01-07T20:03:16.636470Z","shell.execute_reply":"2025-01-07T20:03:18.213904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nimport matplotlib.pyplot as plt\nimport os\nimport time\nimport numpy as np\nimport glob\nimport json\nimport collections\nimport torch\nimport torch.nn as nn\n\nimport pydicom as dicom\nimport matplotlib.patches as patches\n\nfrom matplotlib import animation, rc\nimport pandas as pd\n\nimport pydicom as dicom # dicom\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:03:23.150002Z","iopub.execute_input":"2025-01-07T20:03:23.150291Z","iopub.status.idle":"2025-01-07T20:03:27.282515Z","shell.execute_reply.started":"2025-01-07T20:03:23.150270Z","shell.execute_reply":"2025-01-07T20:03:27.281895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read data\ntrain_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\n\ntrain  = pd.read_csv(train_path + 'train.csv')\nlabel = pd.read_csv(train_path + 'train_label_coordinates.csv')\ntrain_desc  = pd.read_csv(train_path + 'train_series_descriptions.csv')\ntest_desc   = pd.read_csv(train_path + 'test_series_descriptions.csv')\nsub         = pd.read_csv(train_path + 'sample_submission.csv')\nlen(test_desc) #number of test_description.csv rows ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:04:02.678226Z","iopub.execute_input":"2025-01-07T20:04:02.678668Z","iopub.status.idle":"2025-01-07T20:04:02.824652Z","shell.execute_reply.started":"2025-01-07T20:04:02.678644Z","shell.execute_reply":"2025-01-07T20:04:02.823744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:04:04.714371Z","iopub.execute_input":"2025-01-07T20:04:04.714686Z","iopub.status.idle":"2025-01-07T20:04:04.727716Z","shell.execute_reply.started":"2025-01-07T20:04:04.714659Z","shell.execute_reply":"2025-01-07T20:04:04.726941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:04:07.709950Z","iopub.execute_input":"2025-01-07T20:04:07.710266Z","iopub.status.idle":"2025-01-07T20:04:07.717691Z","shell.execute_reply.started":"2025-01-07T20:04:07.710240Z","shell.execute_reply":"2025-01-07T20:04:07.716887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:04:11.224914Z","iopub.execute_input":"2025-01-07T20:04:11.225331Z","iopub.status.idle":"2025-01-07T20:04:11.248956Z","shell.execute_reply.started":"2025-01-07T20:04:11.225298Z","shell.execute_reply":"2025-01-07T20:04:11.248028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to generate image paths based on directory structure\ndef generate_image_paths(df, data_dir):\n    image_paths = []\n    for study_id, series_id in zip(df['study_id'], df['series_id']):\n        study_dir = os.path.join(data_dir, str(study_id))\n        series_dir = os.path.join(study_dir, str(series_id))\n        images = os.listdir(series_dir)\n        image_paths.extend([os.path.join(series_dir, img) for img in images])\n    return image_paths\n\n# Generate image paths for train and test data\ntrain_image_paths = generate_image_paths(train_desc, f'{train_path}/train_images')\ntest_image_paths = generate_image_paths(test_desc, f'{train_path}/test_images')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:04:14.194189Z","iopub.execute_input":"2025-01-07T20:04:14.194468Z","iopub.status.idle":"2025-01-07T20:04:49.708511Z","shell.execute_reply.started":"2025-01-07T20:04:14.194446Z","shell.execute_reply":"2025-01-07T20:04:49.707869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_desc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:16.602332Z","iopub.execute_input":"2025-01-07T20:05:16.602649Z","iopub.status.idle":"2025-01-07T20:05:16.607542Z","shell.execute_reply.started":"2025-01-07T20:05:16.602620Z","shell.execute_reply":"2025-01-07T20:05:16.606697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:18.839180Z","iopub.execute_input":"2025-01-07T20:05:18.839503Z","iopub.status.idle":"2025-01-07T20:05:18.844286Z","shell.execute_reply.started":"2025-01-07T20:05:18.839475Z","shell.execute_reply":"2025-01-07T20:05:18.843514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to reshape a single row of the DataFrame\ndef reshape_row(row):\n    data = {'study_id': [], 'condition': [], 'level': [], 'severity': []}\n    \n    for column, value in row.items():\n        if column not in ['study_id', 'series_id', 'instance_number', 'x', 'y', 'series_description']:\n            parts = column.split('_')\n            condition = ' '.join([word.capitalize() for word in parts[:-2]])\n            level = parts[-2].capitalize() + '/' + parts[-1].capitalize()\n            data['study_id'].append(row['study_id'])\n            data['condition'].append(condition)\n            data['level'].append(level)\n            data['severity'].append(value)\n    \n    return pd.DataFrame(data)\n\n# Reshape the DataFrame for all rows\nnew_train_df = pd.concat([reshape_row(row) for _, row in train.iterrows()], ignore_index=True)\n\n# Display the first few rows of the reshaped dataframe\nnew_train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:22.681976Z","iopub.execute_input":"2025-01-07T20:05:22.682269Z","iopub.status.idle":"2025-01-07T20:05:23.755090Z","shell.execute_reply.started":"2025-01-07T20:05:22.682246Z","shell.execute_reply":"2025-01-07T20:05:23.754208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print columns in a neat way\nprint(\"\\nColumns in new_train_df:\")\nprint(\",\".join(new_train_df.columns))\n\nprint(\"\\nColumns in label:\")\nprint(\",\".join(label.columns))\n\nprint(\"\\nColumns in test_desc:\")\nprint(\",\".join(test_desc.columns))\n\nprint(\"\\nColumns in sub:\")\nprint(\",\".join(sub.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:26.326604Z","iopub.execute_input":"2025-01-07T20:05:26.326933Z","iopub.status.idle":"2025-01-07T20:05:26.333039Z","shell.execute_reply.started":"2025-01-07T20:05:26.326909Z","shell.execute_reply":"2025-01-07T20:05:26.332173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge the dataframes on the common columns\nmerged_df = pd.merge(new_train_df, label, on=['study_id', 'condition', 'level'], how='inner')\n# Merge the dataframes on the common column 'series_id'\nfinal_merged_df = pd.merge(merged_df, train_desc, on='series_id', how='inner')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:30.730085Z","iopub.execute_input":"2025-01-07T20:05:30.730399Z","iopub.status.idle":"2025-01-07T20:05:30.795218Z","shell.execute_reply.started":"2025-01-07T20:05:30.730374Z","shell.execute_reply":"2025-01-07T20:05:30.794313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge the dataframes on the common column 'series_id'\nfinal_merged_df = pd.merge(merged_df, train_desc, on=['series_id','study_id'], how='inner')\n# Display the first few rows of the final merged dataframe\nfinal_merged_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:32.833456Z","iopub.execute_input":"2025-01-07T20:05:32.833746Z","iopub.status.idle":"2025-01-07T20:05:32.859824Z","shell.execute_reply.started":"2025-01-07T20:05:32.833718Z","shell.execute_reply":"2025-01-07T20:05:32.859096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Create the row_id column\nfinal_merged_df['row_id'] = (\n    final_merged_df['study_id'].astype(str) + '_' +\n    final_merged_df['condition'].str.lower().str.replace(' ', '_') + '_' +\n    final_merged_df['level'].str.lower().str.replace('/', '_')\n)\n\n# Create the image_path column\nfinal_merged_df['image_path'] = (\n    f'{train_path}/train_images/' + \n    final_merged_df['study_id'].astype(str) + '/' +\n    final_merged_df['series_id'].astype(str) + '/' +\n    final_merged_df['instance_number'].astype(str) + '.dcm'\n)\n\n# Note: Check image path, since there's 1 instance id, for 1 image, but there's many more images other than the ones labelled in the instance ID. \n\n# Display the updated dataframe\nfinal_merged_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:35.407901Z","iopub.execute_input":"2025-01-07T20:05:35.408180Z","iopub.status.idle":"2025-01-07T20:05:35.587104Z","shell.execute_reply.started":"2025-01-07T20:05:35.408159Z","shell.execute_reply":"2025-01-07T20:05:35.586279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Normal/Mild\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:39.256134Z","iopub.execute_input":"2025-01-07T20:05:39.256417Z","iopub.status.idle":"2025-01-07T20:05:39.358607Z","shell.execute_reply.started":"2025-01-07T20:05:39.256394Z","shell.execute_reply":"2025-01-07T20:05:39.357841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Moderate\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:41.556317Z","iopub.execute_input":"2025-01-07T20:05:41.556596Z","iopub.status.idle":"2025-01-07T20:05:41.590314Z","shell.execute_reply.started":"2025-01-07T20:05:41.556575Z","shell.execute_reply":"2025-01-07T20:05:41.589615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Severe\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:43.860892Z","iopub.execute_input":"2025-01-07T20:05:43.861218Z","iopub.status.idle":"2025-01-07T20:05:43.882698Z","shell.execute_reply.started":"2025-01-07T20:05:43.861196Z","shell.execute_reply":"2025-01-07T20:05:43.882024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nimport pydicom\nfrom torchvision import transforms\nfrom PIL import Image\n\n# Veri artırma için dönüşüm kümesi\naugmentation_transforms = transforms.Compose([\n    transforms.RandomRotation(degrees=20),\n    transforms.RandomHorizontalFlip(p=0.5),\n    transforms.RandomVerticalFlip(p=0.5),\n    transforms.RandomResizedCrop(size=(224, 224), scale=(0.8, 1.0)),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),\n])\n\n# DICOM dosyasını yükleme ve görüntüye dönüştürme fonksiyonu\ndef load_dicom_as_image(file_path):\n    try:\n        dicom = pydicom.dcmread(file_path)  # DICOM dosyasını oku\n        image = dicom.pixel_array  # Piksel verilerini al\n        image = ((image - np.min(image)) / (np.max(image) - np.min(image)) * 255).astype(np.uint8)  # Normalize et\n        return Image.fromarray(image)  # PIL görüntüsüne dönüştür\n    except Exception as e:\n        print(f\"Error loading DICOM file {file_path}: {e}\")\n        return None\n\n# Severe sınıfındaki verileri augment etmek için fonksiyon\ndef augment_severe_class(dataframe, target_count, output_dir=\"augmented_images\"):\n    os.makedirs(output_dir, exist_ok=True)  # Augmented görüntüler için klasör oluştur\n    severe_samples = dataframe[dataframe[\"severity\"] == \"Severe\"]\n    augmented_rows = []\n    \n    while len(severe_samples) + len(augmented_rows) < target_count:\n        for _, row in severe_samples.iterrows():\n            try:\n                image_path = row[\"image_path\"]\n                image = load_dicom_as_image(image_path)  # DICOM dosyasını yükle ve görüntüye çevir\n                \n                if image is None:\n                    continue  # Hatalı görüntüyü atla\n                \n                augmented_image = augmentation_transforms(image)  # Augmentation uygula\n                \n                # Yeni dosya adını ve yolunu belirle\n                new_image_name = f\"augmented_{len(augmented_rows)}.png\"\n                new_image_path = os.path.join(output_dir, new_image_name)\n                \n                # Augment edilen görseli kaydet\n                augmented_image.save(new_image_path)\n                \n                # Yeni satırı oluştur ve ekle\n                augmented_row = row.copy()\n                augmented_row[\"image_path\"] = new_image_path\n                augmented_rows.append(augmented_row)\n                \n                # Hedefe ulaşırsak döngüyü kır\n                if len(severe_samples) + len(augmented_rows) >= target_count:\n                    break\n            except Exception as e:\n                print(f\"Error processing {row['image_path']}: {e}\")\n    \n    # Yeni verileri DataFrame'e ekle\n    augmented_df = pd.DataFrame(augmented_rows)\n    return pd.concat([dataframe, augmented_df]).reset_index(drop=True)\n\n# Moderate sınıfındaki örnek sayısını belirleyelim\nmoderate_count = len(final_merged_df[final_merged_df[\"severity\"] == \"Moderate\"])\n\n# Severe sınıfını augment ederek diğer sınıflarla eşitle\nbalanced_df = augment_severe_class(final_merged_df, target_count=moderate_count)\n\n# Sonuçları kontrol edelim\nprint(balanced_df[\"severity\"].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:05:46.830140Z","iopub.execute_input":"2025-01-07T20:05:46.830420Z","iopub.status.idle":"2025-01-07T20:07:25.805120Z","shell.execute_reply.started":"2025-01-07T20:05:46.830399Z","shell.execute_reply":"2025-01-07T20:07:25.804198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Severe sınıfını augment ederek diğer sınıflarla eşitle\nbalanced_df = augment_severe_class(final_merged_df, target_count=moderate_count)\n\n# Güncellenmiş DataFrame'i final_merged_df ile değiştir\nfinal_merged_df = balanced_df\n\n# Severe sınıfının yeni sayısını kontrol et\nprint(final_merged_df[final_merged_df[\"severity\"] == \"Severe\"].value_counts().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:07:37.798692Z","iopub.execute_input":"2025-01-07T20:07:37.799189Z","iopub.status.idle":"2025-01-07T20:08:57.486012Z","shell.execute_reply.started":"2025-01-07T20:07:37.799158Z","shell.execute_reply":"2025-01-07T20:08:57.485176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Moderate sınıfındaki örnek sayısını belirleyelim\nmoderate_count = len(final_merged_df[final_merged_df[\"severity\"] == \"Moderate\"])\n\n# Normal/Mild sınıfını azaltalım (downsampling)\nnormal_mild_df = final_merged_df[final_merged_df[\"severity\"] == \"Normal/Mild\"].sample(n=moderate_count, random_state=42)\n\n# Diğer sınıfları olduğu gibi alalım\nmoderate_df = final_merged_df[final_merged_df[\"severity\"] == \"Moderate\"]\nsevere_df = final_merged_df[final_merged_df[\"severity\"] == \"Severe\"]\n\n# Yeni veri setini birleştirelim\nfinal_merged_df = pd.concat([normal_mild_df, moderate_df, severe_df]).reset_index(drop=True)\n\n# Sonuçları kontrol edelim\nprint(final_merged_df[\"severity\"].value_counts())\n\n# Normal/Mild sınıfının yeni sayısını kontrol edelim\nprint(f\"Updated Normal/Mild count: {final_merged_df[final_merged_df['severity'] == 'Normal/Mild'].value_counts().sum()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:14.329603Z","iopub.execute_input":"2025-01-07T20:09:14.329957Z","iopub.status.idle":"2025-01-07T20:09:14.398640Z","shell.execute_reply.started":"2025-01-07T20:09:14.329929Z","shell.execute_reply":"2025-01-07T20:09:14.397901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Normal/Mild\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:17.154402Z","iopub.execute_input":"2025-01-07T20:09:17.154690Z","iopub.status.idle":"2025-01-07T20:09:17.189092Z","shell.execute_reply.started":"2025-01-07T20:09:17.154667Z","shell.execute_reply":"2025-01-07T20:09:17.188323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Moderate\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:18.782662Z","iopub.execute_input":"2025-01-07T20:09:18.783018Z","iopub.status.idle":"2025-01-07T20:09:18.814140Z","shell.execute_reply.started":"2025-01-07T20:09:18.782989Z","shell.execute_reply":"2025-01-07T20:09:18.813301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df[final_merged_df[\"severity\"] == \"Severe\"].value_counts().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:21.673616Z","iopub.execute_input":"2025-01-07T20:09:21.673964Z","iopub.status.idle":"2025-01-07T20:09:21.699993Z","shell.execute_reply.started":"2025-01-07T20:09:21.673935Z","shell.execute_reply":"2025-01-07T20:09:21.699289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the base path for test images\nbase_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images/'\n\n# Function to get image paths for a series\ndef get_image_paths(row):\n    series_path = os.path.join(base_path, str(row['study_id']), str(row['series_id']))\n    if os.path.exists(series_path):\n        return [os.path.join(series_path, f) for f in os.listdir(series_path) if os.path.isfile(os.path.join(series_path, f))]\n    return []\n\n# Mapping of series_description to conditions\ncondition_mapping = {\n    'Sagittal T1': {'left': 'left_neural_foraminal_narrowing', 'right': 'right_neural_foraminal_narrowing'},\n    'Axial T2': {'left': 'left_subarticular_stenosis', 'right': 'right_subarticular_stenosis'},\n    'Sagittal T2/STIR': 'spinal_canal_stenosis'\n}\n\n# Create a list to store the expanded rows\nexpanded_rows = []\n\n# Expand the dataframe by adding new rows for each file path\nfor index, row in test_desc.iterrows():\n    image_paths = get_image_paths(row)\n    conditions = condition_mapping.get(row['series_description'], {})\n    if isinstance(conditions, str):  # Single condition\n        conditions = {'left': conditions, 'right': conditions}\n    for side, condition in conditions.items():\n        for image_path in image_paths:\n            expanded_rows.append({\n                'study_id': row['study_id'],\n                'series_id': row['series_id'],\n                'series_description': row['series_description'],\n                'image_path': image_path,\n                'condition': condition,\n                'row_id': f\"{row['study_id']}_{condition}\"\n            })\n\n# Create a new dataframe from the expanded rows\nexpanded_test_desc = pd.DataFrame(expanded_rows)\n\n# Display the resulting dataframe\nexpanded_test_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:23.893961Z","iopub.execute_input":"2025-01-07T20:09:23.894252Z","iopub.status.idle":"2025-01-07T20:09:24.027481Z","shell.execute_reply.started":"2025-01-07T20:09:23.894231Z","shell.execute_reply":"2025-01-07T20:09:24.026626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# change severity column labels\n#Normal/Mild': 'normal_mild', 'Moderate': 'moderate', 'Severe': 'severe'}\nfinal_merged_df['severity'] = final_merged_df['severity'].map({'Normal/Mild': 'normal_mild', 'Moderate': 'moderate', 'Severe': 'severe'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:51.486084Z","iopub.execute_input":"2025-01-07T20:09:51.486373Z","iopub.status.idle":"2025-01-07T20:09:51.493400Z","shell.execute_reply.started":"2025-01-07T20:09:51.486351Z","shell.execute_reply":"2025-01-07T20:09:51.492572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = expanded_test_desc\ntrain_data = final_merged_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:54.222631Z","iopub.execute_input":"2025-01-07T20:09:54.222972Z","iopub.status.idle":"2025-01-07T20:09:54.226533Z","shell.execute_reply.started":"2025-01-07T20:09:54.222942Z","shell.execute_reply":"2025-01-07T20:09:54.225581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:09:57.972131Z","iopub.execute_input":"2025-01-07T20:09:57.972458Z","iopub.status.idle":"2025-01-07T20:09:57.984061Z","shell.execute_reply.started":"2025-01-07T20:09:57.972430Z","shell.execute_reply":"2025-01-07T20:09:57.983152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['series_description'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:10:01.221851Z","iopub.execute_input":"2025-01-07T20:10:01.222147Z","iopub.status.idle":"2025-01-07T20:10:01.229509Z","shell.execute_reply.started":"2025-01-07T20:10:01.222123Z","shell.execute_reply":"2025-01-07T20:10:01.228779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.dcmread(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:10:05.640048Z","iopub.execute_input":"2025-01-07T20:10:05.640332Z","iopub.status.idle":"2025-01-07T20:10:05.644691Z","shell.execute_reply.started":"2025-01-07T20:10:05.640310Z","shell.execute_reply":"2025-01-07T20:10:05.643800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport matplotlib.pyplot as plt\nimport pydicom\nimport numpy as np\n\n# DICOM dosyalarını yüklemek için düzenlenmiş fonksiyon\ndef load_dicom(path):\n    try:\n        # Hataları yok sayarak DICOM dosyasını yükle\n        dicom = pydicom.dcmread(path, force=True)\n        data = dicom.pixel_array\n        data = data - np.min(data)  # Normalize et\n        if np.max(data) != 0:\n            data = data / np.max(data)\n        return data\n    except Exception as e:\n        print(f\"Error loading DICOM file: {path} - {e}\")\n        return None\n\n# Yeni sıfırlanmış indekslerle rastgele seçim yapalım\nfinal_merged_df_reset = final_merged_df.reset_index(drop=True)\n\n# Rastgele iki indeks seçelim\nselected_indices = random.sample(range(len(final_merged_df_reset)), 2)\n\nimages = []\nrow_ids = []\n\n# Seçilen indekslerle görselleri yükleyelim\nfor i in selected_indices:\n    image = load_dicom(final_merged_df_reset['image_path'][i])  # Yeni sıfırlanmış indeksi kullan\n    if image is not None:  # Yalnızca geçerli görselleri ekle\n        images.append(image)\n        row_ids.append(final_merged_df_reset['row_id'][i])\n\n# Eğer geçerli görüntü yoksa hata vermeden devam et\nif len(images) == 0:\n    print(\"No valid DICOM images found. Unable to visualize.\")\nelse:\n    # Görselleri çizdirelim\n    fig, ax = plt.subplots(1, len(images), figsize=(8 * len(images), 4))\n    \n    # Eğer sadece bir görsel varsa, ax bir liste değil, bir Axes objesi olur\n    if len(images) == 1:\n        ax = [ax]  # Tekil Axes objesini listeye çevir\n\n    for i in range(len(images)):\n        ax[i].imshow(images[i], cmap='gray')\n        ax[i].set_title(f'Row ID: {row_ids[i]}', fontsize=8)\n        ax[i].axis('off')\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:10:17.732047Z","iopub.execute_input":"2025-01-07T20:10:17.732329Z","iopub.status.idle":"2025-01-07T20:10:18.136459Z","shell.execute_reply.started":"2025-01-07T20:10:17.732307Z","shell.execute_reply":"2025-01-07T20:10:18.135577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:10:34.797806Z","iopub.execute_input":"2025-01-07T20:10:34.798103Z","iopub.status.idle":"2025-01-07T20:10:34.812998Z","shell.execute_reply.started":"2025-01-07T20:10:34.798081Z","shell.execute_reply":"2025-01-07T20:10:34.812258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data.dropna()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:12:44.342751Z","iopub.execute_input":"2025-01-07T20:12:44.343107Z","iopub.status.idle":"2025-01-07T20:12:44.347991Z","shell.execute_reply.started":"2025-01-07T20:12:44.343081Z","shell.execute_reply":"2025-01-07T20:12:44.346986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nimport torch\nimport torch.optim.lr_scheduler as lr_scheduler\nfrom tqdm import tqdm\nimport numpy as np  # Ekledim çünkü numpy kullanılacak\n\n# Define a custom dataset class\nclass CustomDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, index):\n        image_path = self.dataframe['image_path'][index]\n        image = load_dicom(image_path)  # Define this function to load your DICOM images\n        label = self.dataframe['severity'][index]\n        \n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n# Function to create datasets and dataloaders for each series description\ndef create_datasets_and_loaders(df, series_description, transform, batch_size=8):\n    filtered_df = df[df['series_description'] == series_description]\n    \n    if filtered_df.empty:\n        raise ValueError(f\"No data found for series description: {series_description}\")\n    \n    # Shuffle data\n    filtered_df = filtered_df.sample(frac=1.0, random_state=42)\n    \n    train_df, val_df = train_test_split(filtered_df, test_size=0.2, random_state=42)\n    train_df = train_df.reset_index(drop=True)\n    val_df = val_df.reset_index(drop=True)\n\n    train_dataset = CustomDataset(train_df, transform)\n    val_dataset = CustomDataset(val_df, transform)\n\n    trainloader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n    valloader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\n    \n    return trainloader, valloader, len(train_df), len(val_df)\n\n# Define the transforms\ntransform = transforms.Compose([\n    transforms.Lambda(lambda x: (x * 255).astype(np.uint8)),  # Convert back to uint8 for PIL\n    transforms.ToPILImage(),\n    transforms.Resize((224, 224)),\n    transforms.Grayscale(num_output_channels=3),\n    transforms.ToTensor(),\n])\n\n# Example dataframe\ntrain_data = pd.DataFrame({\n    'image_path': ['path1.dcm', 'path2.dcm', 'path3.dcm'],\n    'series_description': ['Sagittal T1', 'Axial T2', 'Sagittal T2/STIR'],\n    'severity': ['Mild', 'Moderate', 'Severe']\n})\n\n# Create dataloaders for each series description\ndataloaders = {}\nlengths = {}\n\nseries_descriptions = ['Sagittal T1', 'Axial T2', 'Sagittal T2/STIR']\nfor series_description in series_descriptions:\n    try:\n        trainloader, valloader, len_train, len_val = create_datasets_and_loaders(\n            train_data, series_description, transform\n        )\n        dataloaders[series_description] = (trainloader, valloader)\n        lengths[series_description] = (len_train, len_val)\n    except ValueError as e:\n        print(e)  # Hata mesajını konsola yazdır\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:12:46.637297Z","iopub.execute_input":"2025-01-07T20:12:46.637581Z","iopub.status.idle":"2025-01-07T20:12:46.653035Z","shell.execute_reply.started":"2025-01-07T20:12:46.637558Z","shell.execute_reply":"2025-01-07T20:12:46.652193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DicomDataset(Dataset):\n    def __init__(self, filepaths, labels, transform=None):\n        self.filepaths = filepaths\n        self.labels = labels\n        self.transform = transform\n\n        # PNG yüklemesi için kontrol\n        self.valid_data = [\n            (fp, lbl) for fp, lbl in zip(filepaths, labels)\n            if os.path.exists(fp) and load_image(fp) is not None\n        ]\n\n    def __len__(self):\n        return len(self.valid_data)\n\n    def __getitem__(self, idx):\n        filepath, label = self.valid_data[idx]\n        image = load_image(filepath)  # PNG yükleme\n        if self.transform:\n            image = self.transform(image)\n        return image, label\n\ndef visualize_batch(dataloader):\n    images, labels = next(iter(dataloader))\n    fig, axes = plt.subplots(1, len(images), figsize=(15, 5))\n    for i, (img, lbl) in enumerate(zip(images, labels)):\n        ax = axes[i] if len(images) > 1 else axes\n        img = img.permute(1, 2, 0).numpy()\n        ax.imshow(img)\n        ax.set_title(f\"Label: {lbl}\")\n        ax.axis(\"off\")\n    plt.tight_layout()\n    plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:17:45.544855Z","iopub.execute_input":"2025-01-07T20:17:45.545199Z","iopub.status.idle":"2025-01-07T20:17:45.552081Z","shell.execute_reply.started":"2025-01-07T20:17:45.545172Z","shell.execute_reply":"2025-01-07T20:17:45.551103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_datasets_and_loaders(df, series_description, transform, batch_size=8):\n    # Verilen series_description'a göre filtreleme\n    filtered_df = df[df['series_description'] == series_description]\n    \n    # Eğer filtered_df boşsa hata ver\n    if filtered_df.empty:\n        raise ValueError(f\"No data found for series description: {series_description}\")\n    \n    # Satır sayısını kontrol et\n    if len(filtered_df) < 2:\n        raise ValueError(f\"Not enough data for series description: {series_description}. Minimum 2 rows required.\")\n    \n    # Veriyi karıştır\n    filtered_df = filtered_df.sample(frac=1.0, random_state=42)\n    \n    # train_test_split çağır\n    train_df, val_df = train_test_split(filtered_df, test_size=0.2, random_state=42)\n    train_df = train_df.reset_index(drop=True)\n    val_df = val_df.reset_index(drop=True)\n\n    # Dataset ve DataLoader oluştur\n    train_dataset = CustomDataset(train_df, transform)\n    val_dataset = CustomDataset(val_df, transform)\n\n    trainloader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n    valloader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\n    \n    return trainloader, valloader, len(train_df), len(val_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:17:48.539163Z","iopub.execute_input":"2025-01-07T20:17:48.539453Z","iopub.status.idle":"2025-01-07T20:17:48.544654Z","shell.execute_reply.started":"2025-01-07T20:17:48.539431Z","shell.execute_reply":"2025-01-07T20:17:48.543827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nfrom tqdm import tqdm\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:17:54.216223Z","iopub.execute_input":"2025-01-07T20:17:54.216543Z","iopub.status.idle":"2025-01-07T20:17:54.271163Z","shell.execute_reply.started":"2025-01-07T20:17:54.216513Z","shell.execute_reply":"2025-01-07T20:17:54.270043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nfrom tqdm import tqdm\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nclass CustomResNet50(nn.Module):\n    def __init__(self, num_classes=3, pretrained_weights=None):\n        super(CustomResNet50, self).__init__()\n        # Modeli tanımla\n        self.model = models.resnet50(weights=None)  # Varsayılan olarak weights=None\n\n        # Eğer manuel ağırlık yolu verilmişse, bu ağırlıkları yükle\n        if pretrained_weights:\n            self.model.load_state_dict(torch.load(pretrained_weights))\n        \n        # Son katmanı değiştir\n        num_ftrs = self.model.fc.in_features\n        self.model.fc = nn.Linear(num_ftrs, num_classes)\n\n    def forward(self, x):\n        return self.model(x)\n\n    def unfreeze_model(self):\n        \"\"\"Tüm katmanları çöz.\"\"\"\n        for param in self.model.parameters():\n            param.requires_grad = True\n\n    def unfreeze_specific_layers(self, layer_names=None):\n        \"\"\"\n        Belirli katmanları çözmek için kullanılabilir.\n        Eğer layer_names None ise, tüm katmanlar çözülür.\n        \"\"\"\n        for name, param in self.model.named_parameters():\n            if layer_names is None or any(layer in name for layer in layer_names):\n                param.requires_grad = True\n            else:\n                param.requires_grad = False\n\n# Modeli başlat\nsagittal_t1_model = CustomResNet50(num_classes=3).to(device)\naxial_t2_model = CustomResNet50(num_classes=3).to(device)\nsagittal_t2stir_model = CustomResNet50(num_classes=3).to(device)\n\n# Tüm katmanları çözmek için\nfor model in [sagittal_t1_model, axial_t2_model, sagittal_t2stir_model]:\n    model.unfreeze_model()\n\n# Eğitim parametreleri\nweights = torch.tensor([1.0, 2.0, 4.0])\ncriterion = nn.CrossEntropyLoss(weight=weights.to(device))\n\n# Optimizer ayarları\noptimizer_sagittal_t1 = torch.optim.Adam(sagittal_t1_model.model.fc.parameters(), lr=0.001)\noptimizer_axial_t2 = torch.optim.Adam(axial_t2_model.model.fc.parameters(), lr=0.001)\noptimizer_sagittal_t2stir = torch.optim.Adam(sagittal_t2stir_model.model.fc.parameters(), lr=0.001)\n\n# Modelleri ve optimizörleri saklamak için dictionary\nmodels = {\n    'Sagittal T1': sagittal_t1_model,\n    'Axial T2': axial_t2_model,\n    'Sagittal T2/STIR': sagittal_t2stir_model,\n}\n\noptimizers = {\n    'Sagittal T1': optimizer_sagittal_t1,\n    'Axial T2': optimizer_axial_t2,\n    'Sagittal T2/STIR': optimizer_sagittal_t2stir,\n}\n\n# Eğitim yapılabilir parametrelerin sayısını yazdır\nfor model_name, model in models.items():\n    trainable_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\n    print(f\"Trainable parameters for {model_name}: {trainable_params}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:17:58.363029Z","iopub.execute_input":"2025-01-07T20:17:58.363344Z","iopub.status.idle":"2025-01-07T20:17:59.815823Z","shell.execute_reply.started":"2025-01-07T20:17:58.363319Z","shell.execute_reply":"2025-01-07T20:17:59.814928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_map = {'normal_mild': 0, 'moderate': 1, 'severe': 2}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:18:03.504323Z","iopub.execute_input":"2025-01-07T20:18:03.504607Z","iopub.status.idle":"2025-01-07T20:18:03.508510Z","shell.execute_reply.started":"2025-01-07T20:18:03.504584Z","shell.execute_reply":"2025-01-07T20:18:03.507461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader, Dataset\n\n# Örnek Dataset sınıfı\nclass CustomDataset(Dataset):\n    def __init__(self, data, transform=None):\n        self.data = data  # data = [(image_path, label), ...]\n        self.transform = transform\n\n    def load_image(self, image_path):\n        if np.random.rand() > 0.9:  # %10 olasılıkla hata\n            return None\n        return np.random.rand(224, 224, 3)  # Örnek görüntü\n\n    def __getitem__(self, index):\n        image_path, label = self.data[index]\n        image = self.load_image(image_path)\n\n        if image is None:\n            return None, None\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n    def __len__(self):\n        return len(self.data)\n\n# Özel collate_fn\ndef custom_collate_fn(batch):\n    batch = [item for item in batch if item[0] is not None]\n    if len(batch) == 0:\n        return None, None\n    images, labels = zip(*batch)\n    images = torch.stack(images, dim=0)\n    labels = torch.tensor(labels)\n    return images, labels\n\n# Etiket kontrolü için örnek veri\ndata = [(f\"image_{i}.jpg\", i % 3) for i in range(100)]\n\n# Transform işlemleri\ntransform = transforms.Compose([\n    transforms.Lambda(lambda x: (x * 255).astype(np.uint8) if isinstance(x, np.ndarray) else ValueError(\"Invalid image type.\")),\n    transforms.ToPILImage(),\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n])\n\n# Dataset ve DataLoader\ndataset = CustomDataset(data, transform=transform)\ntrainloader_t2 = DataLoader(dataset, batch_size=8, shuffle=False, collate_fn=custom_collate_fn)\n\n# Label mapping\nlabel_map = {0: 0, 1: 1, 2: 2}  # Gerekirse burada düzenleme yap\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Eğitim döngüsü\nfor batch_idx, (images, labels) in enumerate(trainloader_t2):\n    if images is None or labels is None:\n        print(f\"Skipping batch {batch_idx} due to all None values.\")\n        continue\n\n    try:\n        labels = torch.tensor([label_map[label.item()] for label in labels])  # Label mapping\n        labels = labels.to(device)\n        print(f\"Processed labels for batch {batch_idx}: {labels}\")\n    except Exception as e:\n        print(f\"Error processing batch {batch_idx}: {e}\")\n        continue\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:18:07.419514Z","iopub.execute_input":"2025-01-07T20:18:07.419941Z","iopub.status.idle":"2025-01-07T20:18:07.744350Z","shell.execute_reply.started":"2025-01-07T20:18:07.419904Z","shell.execute_reply":"2025-01-07T20:18:07.743336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.optim.lr_scheduler as lr_scheduler\nfrom copy import deepcopy\n\ndef train_model(model, trainloader, valloader, len_train, len_val, optimizer, num_epochs=10, patience=3):\n    # Learning rate scheduler\n    scheduler = lr_scheduler.StepLR(optimizer, step_size=2, gamma=0.1)\n    \n    best_val_acc = 0.0\n    best_model_wts = deepcopy(model.state_dict())\n    counter = 0\n    \n    for epoch in range(num_epochs):\n        model.train()\n        train_loss = 0\n        correct_train = 0\n        \n        with tqdm(trainloader, unit=\"batch\") as tepoch:\n            for images, labels in tepoch:\n                images, labels = images.to(device), torch.tensor([label_map[label] for label in labels]).to(device)\n                optimizer.zero_grad()\n                outputs = model(images)\n                loss = criterion(outputs, labels)\n                loss.backward()\n                optimizer.step()\n                train_loss += loss.item()\n                \n                probabilities = torch.softmax(outputs, dim=1)\n                _, predicted = torch.max(probabilities, 1)\n                correct_train += (predicted == labels).sum().item()\n                \n                tepoch.set_postfix(epoch=epoch+1)\n        \n        scheduler.step()\n        \n        train_loss /= len(trainloader)\n        train_acc = 100 * correct_train / len_train\n        \n        model.eval()\n        val_loss, correct_val = 0, 0\n        with torch.no_grad():\n            with tqdm(valloader, unit=\"batch\") as vepoch:\n                for images, labels in vepoch:\n                    images, labels = images.to(device), torch.tensor([label_map[label] for label in labels]).to(device)\n                    outputs = model(images)\n                    loss = criterion(outputs, labels)\n                    val_loss += loss.item()\n                    \n                    probabilities = torch.softmax(outputs, dim=1)\n\n                    # Eğer batch size 1 ise, dim=0 kullanarak doğru boyutta işlem yapabilirsiniz\n                    if probabilities.dim() == 1:\n                        _, predicted = torch.max(probabilities, 0)  # batch size 1 ise dim=0\n                    else:\n                        _, predicted = torch.max(probabilities, 1)  # normal durumda dim=1\n                    correct_val += (predicted == labels).sum().item()\n                    \n                    vepoch.set_postfix(epoch=epoch+1)\n        \n        val_loss /= len(valloader)\n        val_acc = 100 * correct_val / len_val\n        \n        print(f\"Epoch {epoch+1}, Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.2f}%, Val Loss: {val_loss:.4f}, Val Acc: {val_acc:.2f}%\")\n        \n        # Save the best model and check for early stopping\n        if val_acc > best_val_acc:\n            best_val_acc = val_acc\n            best_model_wts = deepcopy(model.state_dict())\n            counter = 0\n            torch.save(best_model_wts, f'best_model_{epoch+1}.pth')\n        else:\n            counter += 1\n        \n        # Early stopping\n        if counter >= patience:\n            print(f\"Early stopping triggered after {epoch+1} epochs\")\n            break\n    \n    # Load best model weights\n    model.load_state_dict(best_model_wts)\n    return model, best_val_acc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:26:48.323896Z","iopub.execute_input":"2025-01-07T20:26:48.324233Z","iopub.status.idle":"2025-01-07T20:26:48.334565Z","shell.execute_reply.started":"2025-01-07T20:26:48.324207Z","shell.execute_reply":"2025-01-07T20:26:48.333671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def try_create_dataloader(df, series_description, transform, batch_size=8):\n    \"\"\"\n    Verilen seriler için DataLoader oluşturmayı dener. Yeterli veri yoksa None döner.\n    \"\"\"\n    filtered_df = df[df['SeriesDescription'] == series_description]\n    if len(filtered_df) < 2:\n        print(f\"Not enough data for series description: {series_description}. Skipping.\")\n        return None, None, 0, 0  # Yetersiz veri olduğunda dönen değerler\n    \n    filtered_df = filtered_df.sample(frac=1.0, random_state=42)  # Shuffle\n    \n    train_df, val_df = train_test_split(filtered_df, test_size=0.2, random_state=42)\n    train_df = train_df.reset_index(drop=True)\n    val_df = val_df.reset_index(drop=True)\n    \n    # CustomDataset kullanarak veri kümesi oluşturma\n    train_dataset = CustomDataset(\n        [(row['FilePath'], label_map[row['Label']]) for _, row in train_df.iterrows()],\n        transform=transform\n    )\n    val_dataset = CustomDataset(\n        [(row['FilePath'], label_map[row['Label']]) for _, row in val_df.iterrows()],\n        transform=transform\n    )\n    \n    # DataLoader oluşturma\n    train_loader = DataLoader(\n        train_dataset, batch_size=batch_size, shuffle=True, collate_fn=custom_collate_fn\n    )\n    val_loader = DataLoader(\n        val_dataset, batch_size=batch_size, shuffle=False, collate_fn=custom_collate_fn\n    )\n    \n    return train_loader, val_loader, len(train_df), len(val_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:28:34.917689Z","iopub.execute_input":"2025-01-07T20:28:34.918077Z","iopub.status.idle":"2025-01-07T20:28:34.924447Z","shell.execute_reply.started":"2025-01-07T20:28:34.918047Z","shell.execute_reply":"2025-01-07T20:28:34.923720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sütunları kontrol edin\nif 'level' in train_data.columns:\n    # Eğer 'level' sütunu varsa, benzersiz değerlerini yazdır\n    unique_levels = train_data['level'].unique()\n    print(\"Unique values in 'level' column:\", unique_levels)\nelse:\n    # Eğer 'level' sütunu yoksa, tüm sütun adlarını yazdır ve bir uyarı ver\n    print(\"The 'level' column does not exist in the DataFrame.\")\n    print(\"Available columns in train_data:\", train_data.columns)\n\n    # Opsiyonel: Eğer 'level' sütununu oluşturmak istiyorsanız, bunu ekleyebilirsiniz\n    train_data['level'] = 'default_value'  # Varsayılan bir değerle sütun eklenir\n    print(\"A 'level' column has been added with a default value.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:28:43.771158Z","iopub.execute_input":"2025-01-07T20:28:43.771444Z","iopub.status.idle":"2025-01-07T20:28:43.777485Z","shell.execute_reply.started":"2025-01-07T20:28:43.771423Z","shell.execute_reply":"2025-01-07T20:28:43.776857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expanded_test_desc.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:14.411458Z","iopub.execute_input":"2025-01-07T20:29:14.411754Z","iopub.status.idle":"2025-01-07T20:29:14.420255Z","shell.execute_reply.started":"2025-01-07T20:29:14.411720Z","shell.execute_reply":"2025-01-07T20:29:14.419523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"levels = ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']\n\n# Function to update row_id with levels\ndef update_row_id(row, levels):\n    level = levels[row.name % len(levels)]\n    return f\"{row['study_id']}_{row['condition']}_{level}\"\n\n# Update row_id in expanded_test_desc to include levels\nexpanded_test_desc['row_id'] = expanded_test_desc.apply(lambda row: update_row_id(row, levels), axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:17.940831Z","iopub.execute_input":"2025-01-07T20:29:17.941132Z","iopub.status.idle":"2025-01-07T20:29:17.948247Z","shell.execute_reply.started":"2025-01-07T20:29:17.941109Z","shell.execute_reply":"2025-01-07T20:29:17.947216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expanded_test_desc.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:21.148081Z","iopub.execute_input":"2025-01-07T20:29:21.148372Z","iopub.status.idle":"2025-01-07T20:29:21.157513Z","shell.execute_reply.started":"2025-01-07T20:29:21.148349Z","shell.execute_reply":"2025-01-07T20:29:21.156644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a custom test dataset class\nclass TestDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, index):\n        image_path = self.dataframe['image_path'][index]\n        image = load_dicom(image_path)  # Define this function to load your DICOM images\n        if self.transform:\n            image = self.transform(image)\n        return image\n\n# Define the transforms\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((224, 224)),\n    transforms.Grayscale(num_output_channels=3),\n    transforms.ToTensor(),\n])\n\n# Create a test dataset and dataloader\ntest_dataset = TestDataset(expanded_test_desc, transform)\ntestloader = DataLoader(test_dataset, batch_size=1, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:24.030054Z","iopub.execute_input":"2025-01-07T20:29:24.030349Z","iopub.status.idle":"2025-01-07T20:29:24.036699Z","shell.execute_reply.started":"2025-01-07T20:29:24.030326Z","shell.execute_reply":"2025-01-07T20:29:24.035712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for image in testloader:\n    print(image.shape)\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:29.583243Z","iopub.execute_input":"2025-01-07T20:29:29.583524Z","iopub.status.idle":"2025-01-07T20:29:29.648783Z","shell.execute_reply.started":"2025-01-07T20:29:29.583502Z","shell.execute_reply":"2025-01-07T20:29:29.647959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to get the model based on series_description\ndef get_model(series_description):\n    return models.get(series_description, None)\n\n# Function to make predictions on the test data\ndef predict_test_data(testloader, expanded_test_desc):\n    predictions = []\n    normal_mild_probs = []\n    moderate_probs = []\n    severe_probs = []\n    \n    for model in models.values():\n        model.eval()\n        \n    with torch.no_grad():\n        for idx, images in enumerate(tqdm(testloader)):\n            images = images.to(device)\n            series_description = expanded_test_desc.iloc[idx]['series_description']\n            model = get_model(series_description)\n            if model:\n                model.eval()  # Set the model to eval mode\n                outputs = model(images)\n                probs = torch.softmax(outputs, dim=1).squeeze(0)\n                normal_mild_probs.append(probs[0].item())\n                moderate_probs.append(probs[1].item())\n                severe_probs.append(probs[2].item())\n                predictions.append(probs)\n            else:\n                normal_mild_probs.append(None)\n                moderate_probs.append(None)\n                severe_probs.append(None)\n                predictions.append(None)\n    return normal_mild_probs, moderate_probs, severe_probs, predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:35.608720Z","iopub.execute_input":"2025-01-07T20:29:35.609093Z","iopub.status.idle":"2025-01-07T20:29:35.616043Z","shell.execute_reply.started":"2025-01-07T20:29:35.609065Z","shell.execute_reply":"2025-01-07T20:29:35.615038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test data\nnormal_mild_probs, moderate_probs, severe_probs, test_predictions = predict_test_data(testloader, expanded_test_desc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:29:38.568861Z","iopub.execute_input":"2025-01-07T20:29:38.569145Z","iopub.status.idle":"2025-01-07T20:29:44.536375Z","shell.execute_reply.started":"2025-01-07T20:29:38.569123Z","shell.execute_reply":"2025-01-07T20:29:44.535638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:06.991392Z","iopub.execute_input":"2025-01-07T20:30:06.991814Z","iopub.status.idle":"2025-01-07T20:30:07.269758Z","shell.execute_reply.started":"2025-01-07T20:30:06.991762Z","shell.execute_reply":"2025-01-07T20:30:07.268911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add predictions and probabilities to the test DataFrame\nexpanded_test_desc['normal_mild'] = normal_mild_probs\nexpanded_test_desc['moderate'] = moderate_probs\nexpanded_test_desc['severe'] = severe_probs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:14.215406Z","iopub.execute_input":"2025-01-07T20:30:14.215696Z","iopub.status.idle":"2025-01-07T20:30:14.221082Z","shell.execute_reply.started":"2025-01-07T20:30:14.215673Z","shell.execute_reply":"2025-01-07T20:30:14.220124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = expanded_test_desc[[\"row_id\",\"normal_mild\",\"moderate\",\"severe\"]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:18.244945Z","iopub.execute_input":"2025-01-07T20:30:18.245260Z","iopub.status.idle":"2025-01-07T20:30:18.249814Z","shell.execute_reply.started":"2025-01-07T20:30:18.245235Z","shell.execute_reply":"2025-01-07T20:30:18.248958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:21.524978Z","iopub.execute_input":"2025-01-07T20:30:21.525327Z","iopub.status.idle":"2025-01-07T20:30:21.535036Z","shell.execute_reply.started":"2025-01-07T20:30:21.525304Z","shell.execute_reply":"2025-01-07T20:30:21.534279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'row_id' and sum the values\ngrouped_submission = submission.groupby('row_id').max().reset_index()\n\n# Normalize the columns\ngrouped_submission[['normal_mild', 'moderate', 'severe']] = grouped_submission[['normal_mild', 'moderate', 'severe']].div(grouped_submission[['normal_mild', 'moderate', 'severe']].sum(axis=1), axis=0)\n\n# Check the first 3 rows\ngrouped_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:24.307965Z","iopub.execute_input":"2025-01-07T20:30:24.308288Z","iopub.status.idle":"2025-01-07T20:30:24.328345Z","shell.execute_reply.started":"2025-01-07T20:30:24.308260Z","shell.execute_reply":"2025-01-07T20:30:24.327638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(grouped_submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:28.626770Z","iopub.execute_input":"2025-01-07T20:30:28.627130Z","iopub.status.idle":"2025-01-07T20:30:28.632216Z","shell.execute_reply.started":"2025-01-07T20:30:28.627101Z","shell.execute_reply":"2025-01-07T20:30:28.631347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub[['normal_mild', 'moderate', 'severe']] = grouped_submission[['normal_mild', 'moderate', 'severe']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:30.930182Z","iopub.execute_input":"2025-01-07T20:30:30.930467Z","iopub.status.idle":"2025-01-07T20:30:30.936451Z","shell.execute_reply.started":"2025-01-07T20:30:30.930444Z","shell.execute_reply":"2025-01-07T20:30:30.935491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Save the DataFrame to \"submission.csv\" in the desired directory\nsub.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:33.495280Z","iopub.execute_input":"2025-01-07T20:30:33.495569Z","iopub.status.idle":"2025-01-07T20:30:33.502523Z","shell.execute_reply.started":"2025-01-07T20:30:33.495546Z","shell.execute_reply":"2025-01-07T20:30:33.501503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:30:35.739492Z","iopub.execute_input":"2025-01-07T20:30:35.739812Z","iopub.status.idle":"2025-01-07T20:30:35.748156Z","shell.execute_reply.started":"2025-01-07T20:30:35.739764Z","shell.execute_reply":"2025-01-07T20:30:35.747339Z"}},"outputs":[],"execution_count":null}]}