{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":10000068,"sourceType":"datasetVersion","datasetId":6151344},{"sourceId":10000072,"sourceType":"datasetVersion","datasetId":6151337},{"sourceId":10000083,"sourceType":"datasetVersion","datasetId":6155220},{"sourceId":10000089,"sourceType":"datasetVersion","datasetId":6155222},{"sourceId":10000093,"sourceType":"datasetVersion","datasetId":6155225},{"sourceId":10000465,"sourceType":"datasetVersion","datasetId":6155482},{"sourceId":10019991,"sourceType":"datasetVersion","datasetId":6047696}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"### import gc\nimport pandas as pd\nimport numpy as np\nimport datetime as dt\n\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nimport seaborn as sns\n\n# 🚫 Suppressing warnings 🚫\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:21:55.722580Z","iopub.execute_input":"2024-12-04T11:21:55.723074Z","iopub.status.idle":"2024-12-04T11:21:57.092686Z","shell.execute_reply.started":"2024-12-04T11:21:55.723034Z","shell.execute_reply":"2024-12-04T11:21:57.091359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy.optimize import minimize\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:21:57.094932Z","iopub.execute_input":"2024-12-04T11:21:57.095430Z","iopub.status.idle":"2024-12-04T11:21:57.693458Z","shell.execute_reply.started":"2024-12-04T11:21:57.095385Z","shell.execute_reply":"2024-12-04T11:21:57.692357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm.auto import tqdm \nfrom concurrent.futures import ThreadPoolExecutor\nfrom joblib import Parallel, delayed\nfrom time import sleep, time\nfrom multiprocessing import cpu_count\nfrom sklearn.preprocessing import MinMaxScaler\nimport concurrent.futures","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:21:57.695222Z","iopub.execute_input":"2024-12-04T11:21:57.695701Z","iopub.status.idle":"2024-12-04T11:21:57.716708Z","shell.execute_reply.started":"2024-12-04T11:21:57.695648Z","shell.execute_reply":"2024-12-04T11:21:57.715596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/colombian-frenchteam-problematicinternetusage/Dataset_problematic_internet_usage.csv')\nlen(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:21:57.718184Z","iopub.execute_input":"2024-12-04T11:21:57.718530Z","iopub.status.idle":"2024-12-04T11:21:57.803306Z","shell.execute_reply.started":"2024-12-04T11:21:57.718495Z","shell.execute_reply":"2024-12-04T11:21:57.802016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nfrom pathlib import Path\nimport json","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:21:57.805944Z","iopub.execute_input":"2024-12-04T11:21:57.806308Z","iopub.status.idle":"2024-12-04T11:21:57.812648Z","shell.execute_reply.started":"2024-12-04T11:21:57.806273Z","shell.execute_reply":"2024-12-04T11:21:57.811276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport pandas as pd\nimport shutil\nimport os\n\n# Define the input directories for the multiple parts\ninput_dirs = [\n    '/kaggle/input/colombianfrenchteam-imagesv1/imagesandannotations/',\n    '/kaggle/input/colombianfrenchteam-imagesv1-part2/imagesandannotations/',\n    '/kaggle/input/colombianfrenchteam-imagesv1-part-3/imagesandannotations/',\n    '/kaggle/input/colombianfrenchteam-imagesv1-part-4/imagesandannotations/',\n    '/kaggle/input/colombianfrenchteam-imagesv1-part-5/imagesandannotations/',\n    '/kaggle/input/colombianfrenchteam-imagesv1-part-6/imagesandannotations/'\n]\n\n# Define the destination folder for combined images\ncombined_images_dir = '/kaggle/working/combined_images/'\n\n# Ensure the destination folder exists\nos.makedirs(combined_images_dir, exist_ok=True)\n\n# Function to load and fix the JSON files if needed\ndef load_json_file(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            content = file.read()\n\n        # Fix invalid '][' and wrap content in square brackets if necessary\n        fixed_content = content.replace(\"][\", \",\")\n        if not fixed_content.startswith('['):\n            fixed_content = f\"[{fixed_content}\"\n        if not fixed_content.endswith(']'):\n            fixed_content = f\"{fixed_content}]\"\n        \n        data = json.loads(fixed_content)\n        print(f\"Successfully loaded {file_path}!\")\n        return data\n    except json.JSONDecodeError as e:\n        print(f\"Error decoding JSON in {file_path}: {e}\")\n        return None\n\n# Function to copy all files from source to destination\ndef copy_images(source_dir):\n    for filename in os.listdir(source_dir):\n        file_path = os.path.join(source_dir, filename)\n        if os.path.isfile(file_path):\n            shutil.copy(file_path, combined_images_dir)\n\n# Data containers for all parts\nall_window_properties_data = []\nall_all_events_data = []\n\n# Load data and copy images for each input directory part\nfor input_dir in input_dirs:\n    # Define the paths to the JSON files for this part\n    window_properties_path = os.path.join(input_dir, 'window_properties.json')\n    all_events_path = os.path.join(input_dir, 'all_events.json')\n\n    # Load the JSON data\n    window_properties_data = load_json_file(window_properties_path)\n    all_events_data = load_json_file(all_events_path)\n\n    # Append the data from this part\n    if window_properties_data:\n        all_window_properties_data.extend(window_properties_data)\n    if all_events_data:\n        all_all_events_data.extend(all_events_data)\n\n    # Copy images for this part\n    copy_images(input_dir)\n\n# Convert all collected window properties and events data into DataFrames\nwindow_properties_df = pd.DataFrame(all_window_properties_data)\nall_events_df = pd.DataFrame(all_all_events_data)\n\n# Display a sample of the DataFrames\nprint(\"Window Properties DataFrame:\")\nprint(window_properties_df.head())  # Show the first 5 rows as a sample\n\nprint(\"\\nAll Events DataFrame:\")\nprint(all_events_df.head())  # Show the first 5 rows as a sample\n\nprint(f\"All images have been successfully copied to {combined_images_dir}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:21:57.814707Z","iopub.execute_input":"2024-12-04T11:21:57.815687Z","iopub.status.idle":"2024-12-04T11:23:14.456559Z","shell.execute_reply.started":"2024-12-04T11:21:57.815632Z","shell.execute_reply":"2024-12-04T11:23:14.455369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path to the directory\ndirectory_path = '/kaggle/working/combined_images/'\n\n# List all files in the directory and filter out the .jpg files\njpg_files = [f for f in os.listdir(directory_path) if f.lower().endswith('.jpg')]\n\n# Count the number of .jpg files\nnum_jpg_files = len(jpg_files)\n\n# Print the result\nprint(f\"Number of .jpg images: {num_jpg_files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.458352Z","iopub.execute_input":"2024-12-04T11:23:14.458837Z","iopub.status.idle":"2024-12-04T11:23:14.475564Z","shell.execute_reply.started":"2024-12-04T11:23:14.458784Z","shell.execute_reply":"2024-12-04T11:23:14.474292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_df = all_events_df\ncombined_df = combined_df.rename(columns={'series_id': 'id'})\nlen(combined_df.id.unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.477420Z","iopub.execute_input":"2024-12-04T11:23:14.478130Z","iopub.status.idle":"2024-12-04T11:23:14.497467Z","shell.execute_reply.started":"2024-12-04T11:23:14.478029Z","shell.execute_reply":"2024-12-04T11:23:14.496201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.499034Z","iopub.execute_input":"2024-12-04T11:23:14.499475Z","iopub.status.idle":"2024-12-04T11:23:14.520428Z","shell.execute_reply.started":"2024-12-04T11:23:14.499424Z","shell.execute_reply":"2024-12-04T11:23:14.519347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(combined_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.521972Z","iopub.execute_input":"2024-12-04T11:23:14.522357Z","iopub.status.idle":"2024-12-04T11:23:14.530132Z","shell.execute_reply.started":"2024-12-04T11:23:14.522321Z","shell.execute_reply":"2024-12-04T11:23:14.528872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_df.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.531538Z","iopub.execute_input":"2024-12-04T11:23:14.531918Z","iopub.status.idle":"2024-12-04T11:23:14.546602Z","shell.execute_reply.started":"2024-12-04T11:23:14.531882Z","shell.execute_reply":"2024-12-04T11:23:14.545294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'sii' and count occurrences, including NaN as a category\nsii_counts = combined_df['label'].value_counts(dropna=False)\n\n# Calculate percentages\nsii_percentages = (sii_counts / sii_counts.sum()) * 100\n\n# Create a DataFrame to display counts and percentages together\nsii_summary = pd.DataFrame({\n    'Count': sii_counts,\n    'Percentage': sii_percentages\n})\n\n# Display the summary\nprint(sii_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.548465Z","iopub.execute_input":"2024-12-04T11:23:14.549548Z","iopub.status.idle":"2024-12-04T11:23:14.577505Z","shell.execute_reply.started":"2024-12-04T11:23:14.549508Z","shell.execute_reply":"2024-12-04T11:23:14.575890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.578897Z","iopub.execute_input":"2024-12-04T11:23:14.579363Z","iopub.status.idle":"2024-12-04T11:23:14.596508Z","shell.execute_reply.started":"2024-12-04T11:23:14.579302Z","shell.execute_reply":"2024-12-04T11:23:14.595257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_df['sii'] = pd.to_numeric(combined_df['label'], errors='coerce')\ncombined_df['sii'] = combined_df['sii'].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.600316Z","iopub.execute_input":"2024-12-04T11:23:14.600674Z","iopub.status.idle":"2024-12-04T11:23:14.619276Z","shell.execute_reply.started":"2024-12-04T11:23:14.600642Z","shell.execute_reply":"2024-12-04T11:23:14.617961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.621458Z","iopub.execute_input":"2024-12-04T11:23:14.621965Z","iopub.status.idle":"2024-12-04T11:23:14.644071Z","shell.execute_reply.started":"2024-12-04T11:23:14.621910Z","shell.execute_reply":"2024-12-04T11:23:14.642706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualize the images\n","metadata":{}},{"cell_type":"code","source":"import random\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\n# Assuming combined_df is already loaded\n# Load a random row from the dataframe\nrandom_row = combined_df.sample(n=1).iloc[0]\n\n# Get the image file name from the 'image_name' column\nimage_name = random_row['image_name']\n\n# Construct the full path to the image\nimage_path = f'/kaggle/working/combined_images/{image_name}'\n\n# Open the image\nimage = Image.open(image_path)\n\n# Get the dimensions of the image\nimage_width, image_height = image.size\n\n# Print the image name and dimensions\nprint(f\"Image: {image_name}\")\nprint(f\"Dimensions: {image_width}x{image_height}\")\n\n# Display the image\nplt.imshow(image)\nplt.axis('off')  # Hide axes for a cleaner view\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:14.645559Z","iopub.execute_input":"2024-12-04T11:23:14.645971Z","iopub.status.idle":"2024-12-04T11:23:15.007078Z","shell.execute_reply.started":"2024-12-04T11:23:14.645907Z","shell.execute_reply":"2024-12-04T11:23:15.005618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming combined_df is already loaded\n# Load a random row from the dataframe\nrandom_row = combined_df.sample(n=1).iloc[0]\n\n# Get the image file name from the 'image_name' column\nimage_name = random_row['image_name']\n\n# Construct the full path to the image\nimage_path = f'/kaggle/working/combined_images/{image_name}'\n\n# Open the image\nimage = Image.open(image_path)\n\n# Get the original dimensions of the image\nimage_width, image_height = image.size\n\n# Print the original image name and dimensions\nprint(f\"Original Image: {image_name}\")\nprint(f\"Original Dimensions: {image_width}x{image_height}\")\n\n# Resize the image while keeping the aspect ratio (resize width to 600px)\ntarget_width = 600\naspect_ratio = image_height / image_width\ntarget_height = int(target_width * aspect_ratio)\n\n# Resize the image\nresized_image = image.resize((target_width, target_height))\n\n# Print the resized image dimensions\nprint(f\"Resized Dimensions: {resized_image.size[0]}x{resized_image.size[1]}\")\n\n# Display the resized image\nplt.imshow(resized_image)\nplt.axis('off')  # Hide axes for a cleaner view\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:15.008476Z","iopub.execute_input":"2024-12-04T11:23:15.008845Z","iopub.status.idle":"2024-12-04T11:23:15.134174Z","shell.execute_reply.started":"2024-12-04T11:23:15.008797Z","shell.execute_reply":"2024-12-04T11:23:15.132671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Desired image dimensions\ntarget_width = 2160\ntarget_height = 600\n\n# Directory containing the images\nimage_dir = '/kaggle/working/combined_images/'\n\n# Iterate through each image in the directory\nfor image_name in os.listdir(image_dir):\n    image_path = os.path.join(image_dir, image_name)\n    \n    # Check if it's an image file (you can add more image formats if necessary)\n    if image_name.endswith(('.jpg', '.jpeg', '.png', '.bmp')):\n        # Open the image\n        image = Image.open(image_path)\n        \n        # Get the current dimensions of the image\n        image_width, image_height = image.size\n        \n        # Check if the dimensions match the target size\n        if image_width != target_width or image_height != target_height:\n            print(f\"Image {image_name} has dimensions {image_width}x{image_height}, \"\n                  f\"but the expected dimensions are {target_width}x{target_height}.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:15.135947Z","iopub.execute_input":"2024-12-04T11:23:15.136581Z","iopub.status.idle":"2024-12-04T11:23:16.400421Z","shell.execute_reply.started":"2024-12-04T11:23:15.136507Z","shell.execute_reply":"2024-12-04T11:23:16.399293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'sii' and count occurrences, including NaN as a category\nsii_counts = combined_df['label'].value_counts(dropna=False)\n\n# Calculate percentages\nsii_percentages = (sii_counts / sii_counts.sum()) * 100\n\n# Create a DataFrame to display counts and percentages together\nsii_summary = pd.DataFrame({\n    'Count': sii_counts,\n    'Percentage': sii_percentages\n})\n\n# Display the summary\nprint(sii_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T11:23:16.402255Z","iopub.execute_input":"2024-12-04T11:23:16.402732Z","iopub.status.idle":"2024-12-04T11:23:16.414380Z","shell.execute_reply.started":"2024-12-04T11:23:16.402682Z","shell.execute_reply":"2024-12-04T11:23:16.413171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CNN Vanilla","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\nimport os\nfrom torchvision import transforms\n\n# Path to the image folder\nimage_dir = '/kaggle/working/combined_images/'\n\n# Load the combined_df DataFrame (assuming it's already loaded)\n# combined_df = pd.read_csv('your_dataframe.csv')  # Replace with your actual DataFrame loading\n\n# Custom Dataset class to load images from the dataframe\nclass ImageDataset(Dataset):\n    def __init__(self, dataframe, image_dir, transform=None):\n        self.dataframe = dataframe\n        self.image_dir = image_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        img_name = os.path.join(self.image_dir, self.dataframe.iloc[idx, 1])  # image_name column\n        image = Image.open(img_name)\n        label = self.dataframe.iloc[idx, 3]  # sii column\n        image_id = self.dataframe.iloc[idx, 0]  # id column (assuming the 'id' is in the first column)\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label, image_id  # Return the image, label, and id\n\n# Function to calculate the target height maintaining the aspect ratio\ndef calculate_target_height(image, target_width):\n    width, height = image.size\n    target_height = int((target_width * height) / width)\n    return target_height\n\n# Define the CNN model\nclass SimpleCNN(nn.Module):\n    def __init__(self):\n        super(SimpleCNN, self).__init__()\n        self.conv1 = nn.Conv2d(3, 32, kernel_size=3, padding=1)\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=3, padding=1)\n        self.conv3 = nn.Conv2d(64, 128, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.fc1 = None  # We will initialize this dynamically\n        self.fc2 = nn.Linear(512, NUM_CLASSES)  # Assuming this part will remain constant\n        self.dropout = nn.Dropout(0.5)\n\n    def forward(self, x):\n        # Calculate the output size after conv1, conv2, conv3 and pooling layers\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.pool(F.relu(self.conv2(x)))\n        x = self.pool(F.relu(self.conv3(x)))\n\n        # Dynamically calculate the flattened size\n        flattened_size = x.size(1) * x.size(2) * x.size(3)\n        x = x.view(-1, flattened_size)  # Flatten the tensor dynamically\n\n        if self.fc1 is None:\n            self.fc1 = nn.Linear(flattened_size, 512)  # Initialize fc1 dynamically based on the size\n            self.fc1.apply(self.init_weights)  # Optionally initialize weights\n\n        x = F.relu(self.fc1(x))\n        x = self.dropout(x)\n        x = self.fc2(x)\n        return x\n    \n    # Initialize weights for the dynamically created fc1 layer\n    def init_weights(self, m):\n        if isinstance(m, nn.Linear):\n            nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')\n            if m.bias is not None:\n                nn.init.constant_(m.bias, 0)\n\n# Transforms for resizing while maintaining the aspect ratio and data normalization\ndef get_transform():\n    def transform(image):\n        target_height = calculate_target_height(image, TARGET_WIDTH)  # Calculate target height\n        resize_transform = transforms.Resize((target_height, TARGET_WIDTH))  # Resize to maintain aspect ratio\n        image = resize_transform(image)\n        image = transforms.ToTensor()(image)\n        image = transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])(image)\n        return image\n    return transform","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:55:55.310097Z","iopub.execute_input":"2024-12-04T17:55:55.310572Z","iopub.status.idle":"2024-12-04T17:55:55.331329Z","shell.execute_reply.started":"2024-12-04T17:55:55.310535Z","shell.execute_reply":"2024-12-04T17:55:55.329725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.cuda.is_available()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:55:57.916098Z","iopub.execute_input":"2024-12-04T17:55:57.916510Z","iopub.status.idle":"2024-12-04T17:55:57.924464Z","shell.execute_reply.started":"2024-12-04T17:55:57.916473Z","shell.execute_reply":"2024-12-04T17:55:57.923154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'sii' and count occurrences, including NaN as a category\nsii_counts = combined_df['label'].value_counts(dropna=False)\n\n# Calculate percentages\nsii_percentages = (sii_counts / sii_counts.sum()) * 100\n\n# Count the unique 'id's for each 'sii' value\nsii_unique_ids_count = combined_df.groupby('label')['id'].nunique()\n\n# Create a DataFrame to display counts, percentages, and unique id count together\nsii_summary = pd.DataFrame({\n    'Count': sii_counts,\n    'Percentage': sii_percentages,\n    'Unique ID Count': sii_unique_ids_count\n})\n\n# Display the summary\nprint(sii_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:55:58.627215Z","iopub.execute_input":"2024-12-04T17:55:58.627683Z","iopub.status.idle":"2024-12-04T17:55:58.666296Z","shell.execute_reply.started":"2024-12-04T17:55:58.627638Z","shell.execute_reply":"2024-12-04T17:55:58.665079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Determine if CUDA is available and set the device accordingly\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")  # Use GPU 0\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:56:00.866915Z","iopub.execute_input":"2024-12-04T17:56:00.867377Z","iopub.status.idle":"2024-12-04T17:56:00.875252Z","shell.execute_reply.started":"2024-12-04T17:56:00.867338Z","shell.execute_reply":"2024-12-04T17:56:00.873726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\nimport torch\n\n# Calculate class weights based on the label distribution\nclass_weights = compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(combined_df['label']),\n    y=combined_df['label']\n)\n\n# Convert class weights to a tensor and move it to the same device (GPU or CPU)\nclass_weights_tensor = torch.tensor(class_weights, dtype=torch.float).to(device)\n\nclass_weights_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:56:01.500990Z","iopub.execute_input":"2024-12-04T17:56:01.501416Z","iopub.status.idle":"2024-12-04T17:56:01.528673Z","shell.execute_reply.started":"2024-12-04T17:56:01.501379Z","shell.execute_reply":"2024-12-04T17:56:01.527472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define constants\nTARGET_WIDTH = 224  # Resize to 600px width while keeping the aspect ratio for height\nNUM_CLASSES = 4  # SII values: 0, 1, 2, 3\nBATCH_SIZE = 32  # Depending on available memory, you can adjust this\nNUM_EPOCHS = 10 # just to quickly test\nLEARNING_RATE = 0.0001","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T18:50:43.326433Z","iopub.execute_input":"2024-12-04T18:50:43.326885Z","iopub.status.idle":"2024-12-04T18:50:43.333391Z","shell.execute_reply.started":"2024-12-04T18:50:43.326847Z","shell.execute_reply":"2024-12-04T18:50:43.332168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import defaultdict\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\n\n# Arrays to store results for each case\nkappa_scores_image_level = []\nkappa_scores_id_level_no_argmax = []\nkappa_scores_id_level_custom_logic = []\n\nall_true_labels_image = []\nall_pred_labels_image = []\n\nall_true_labels_id_no_argmax = []\nall_pred_labels_id_no_argmax = []\n\nall_true_labels_id_custom = []\nall_pred_labels_id_custom = []\n\nmatrices_image_level = []\nmatrices_id_no_argmax = []\nmatrices_id_custom_logic = []\n\n# First, split the data based on the unique IDs\nids = combined_df['id'].unique()\nn_splits = 5  # Number of CV splits\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n# For each fold in cross-validation\nfor fold, (train_idx, val_idx) in enumerate(kf.split(ids, combined_df.groupby('id')['label'].first().loc[ids])):\n    print(f\"Training fold {fold + 1}/{n_splits}...\")\n\n    # Get the corresponding data for training and validation based on the split IDs\n    train_ids = ids[train_idx]\n    val_ids = ids[val_idx]\n    train_df = combined_df[combined_df['id'].isin(train_ids)]\n    val_df = combined_df[combined_df['id'].isin(val_ids)]\n\n    # Prepare datasets for this fold\n    train_dataset = ImageDataset(train_df, image_dir, transform=get_transform())\n    val_dataset = ImageDataset(val_df, image_dir, transform=get_transform())\n\n    # Create DataLoader for training and validation\n    train_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\n    val_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False)\n\n    # Initialize the model, loss function, and optimizer\n    model = SimpleCNN().to(device)\n    criterion = nn.CrossEntropyLoss(weight=class_weights_tensor)\n    optimizer = optim.Adam(model.parameters(), lr=LEARNING_RATE)\n\n    # Train the model for the specified number of epochs\n    for epoch in range(NUM_EPOCHS):\n        model.train()\n        running_loss = 0.0\n        for inputs, labels, img_ids in train_loader:\n            inputs, labels = inputs.to(device), labels.to(device)\n            optimizer.zero_grad()\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            running_loss += loss.item()\n\n        print(f\"Epoch {epoch + 1}/{NUM_EPOCHS}, Loss: {running_loss / len(train_loader):.4f}\")\n\n    # Evaluate the model on the validation set\n    model.eval()\n\n    # Dictionaries to hold true and predicted labels for each ID\n    id_true_labels_no_argmax = defaultdict(list)\n    id_pred_labels_no_argmax = defaultdict(list)\n    \n    id_true_labels_custom = defaultdict(list)\n    id_pred_labels_custom = defaultdict(list)\n\n    # For image-level confusion matrix\n    true_labels_image = []\n    pred_labels_image = []\n\n    with torch.no_grad():\n        for batch_idx, (inputs, labels, image_ids) in enumerate(val_loader):  # We now also get 'ids'\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs)\n            _, predicted = torch.max(outputs, 1)\n    \n            # For image-level confusion matrix and Kappa\n            true_labels_image.extend(labels.cpu().numpy())\n            pred_labels_image.extend(predicted.cpu().numpy())\n    \n            # For collecting labels by ID for both strategies\n            for i in range(len(image_ids)):  # Iterate over each image in the batch\n                image_id = image_ids[i]  # Use the image_id from the batch\n    \n                # For No `argmax` ID-level confusion matrix: Collect labels\n                id_true_labels_no_argmax[image_id].append(labels[i].cpu().numpy())\n                id_pred_labels_no_argmax[image_id].append(predicted[i].cpu().numpy())\n    \n                # For Custom Logic ID-level confusion matrix: Collect labels\n                id_true_labels_custom[image_id].append(labels[i].cpu().numpy())\n                id_pred_labels_custom[image_id].append(predicted[i].cpu().numpy())\n\n    # Calculate Kappa for Image-level\n    kappa_image = cohen_kappa_score(true_labels_image, pred_labels_image, weights='quadratic')\n    kappa_scores_image_level.append(kappa_image)\n\n    # Calculate Kappa for ID-level (No `argmax`)\n    id_true_labels_aggregated_no_argmax = []\n    id_pred_labels_aggregated_no_argmax = []\n    for image_id in id_true_labels_no_argmax:\n        # Most frequent label for each ID\n        most_frequent_true_label = np.bincount(id_true_labels_no_argmax[image_id]).argmax()\n        most_frequent_pred_label = np.bincount(id_pred_labels_no_argmax[image_id]).argmax()\n\n        id_true_labels_aggregated_no_argmax.append(most_frequent_true_label)\n        id_pred_labels_aggregated_no_argmax.append(most_frequent_pred_label)\n\n    kappa_no_argmax = cohen_kappa_score(id_true_labels_aggregated_no_argmax, id_pred_labels_aggregated_no_argmax, weights='quadratic')\n    kappa_scores_id_level_no_argmax.append(kappa_no_argmax)\n\n    # Calculate Kappa for ID-level (Custom Logic)\n    # After collecting true and predicted labels per ID, we apply custom logic\n    id_true_labels_custom_aggregated = []\n    id_pred_labels_custom_aggregated = []\n    \n    for image_id in id_true_labels_custom:\n        true_labels = id_true_labels_custom[image_id]\n        pred_labels = id_pred_labels_custom[image_id]\n    \n        # Apply custom logic:\n        # Priority: if the ID has at least one image labeled as 3.0, then it is 3.0, and so on\n        if 3 in pred_labels:\n            pred_label = 3\n        elif 2 in pred_labels:\n            pred_label = 2\n        elif 1 in pred_labels:\n            pred_label = 1\n        else:\n            pred_label = 0\n    \n        if 3 in true_labels:\n            true_label = 3\n        elif 2 in true_labels:\n            true_label = 2\n        elif 1 in true_labels:\n            true_label = 1\n        else:\n            true_label = 0\n    \n        id_true_labels_custom_aggregated.append(true_label)\n        id_pred_labels_custom_aggregated.append(pred_label)\n\n    kappa_custom_logic = cohen_kappa_score(id_true_labels_custom_aggregated, id_pred_labels_custom_aggregated, weights='quadratic')\n    kappa_scores_id_level_custom_logic.append(kappa_custom_logic)\n\n    print(f\"Fold {fold + 1} Image-level Kappa: {kappa_image:.4f}\")\n    print(f\"Fold {fold + 1} ID-level No `argmax` Kappa: {kappa_no_argmax:.4f}\")\n    print(f\"Fold {fold + 1} ID-level Custom Logic Kappa: {kappa_custom_logic:.4f}\")\n\n    # Calculate confusion matrix for each case (Image-level, ID-level No `argmax`, ID-level Custom Logic)\n    cm_image_level = confusion_matrix(true_labels_image, pred_labels_image)\n    cm_id_no_argmax = confusion_matrix(id_true_labels_aggregated_no_argmax, id_pred_labels_aggregated_no_argmax)\n    cm_id_custom_logic = confusion_matrix(id_true_labels_custom_aggregated, id_pred_labels_custom_aggregated)\n\n    matrices_image_level.append(cm_image_level)\n    matrices_id_no_argmax.append(cm_id_no_argmax)\n    matrices_id_custom_logic.append(cm_id_custom_logic)\n\n# Calculate the average Kappa scores for all folds\naverage_kappa_image = np.mean(kappa_scores_image_level)\naverage_kappa_no_argmax = np.mean(kappa_scores_id_level_no_argmax)\naverage_kappa_custom_logic = np.mean(kappa_scores_id_level_custom_logic)\n\nprint(f\"\\nAverage Image-level Kappa: {average_kappa_image:.4f}\")\nprint(f\"Average ID-level No `argmax` Kappa: {average_kappa_no_argmax:.4f}\")\nprint(f\"Average ID-level Custom Logic Kappa: {average_kappa_custom_logic:.4f}\")\n\ntotal_image_level = np.sum(matrices_image_level, axis=0).astype(int)\ntotal_id_no_argmax = np.sum(matrices_id_no_argmax, axis=0).astype(int)\ntotal_id_custom_logic = np.sum(matrices_id_custom_logic, axis=0).astype(int)\n\n# Display confusion matrices\nprint(\"\\nConfusion Matrix (Image-level):\")\nprint(total_image_level)\nprint(\"\\nConfusion Matrix (ID-level No `argmax`):\")\nprint(total_id_no_argmax)\nprint(\"\\nConfusion Matrix (ID-level Custom Logic):\")\nprint(total_id_custom_logic)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:35:03.775747Z","iopub.execute_input":"2024-12-04T19:35:03.776199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_confusion_matrix = total_image_level\n\n# Calculate the percentage confusion matrix by row\nrow_sums = np.sum(total_confusion_matrix, axis=1, keepdims=True)  # Sum of each row\npercentage_confusion_matrix_by_row = total_confusion_matrix / row_sums * 100  # Normalize by row\n\n# Manually format the percentages by adding '%' to each label\nformatted_percentage_matrix = np.array([['{:.2f}%'.format(val) for val in row] for row in percentage_confusion_matrix_by_row])\n\n# Create a figure with two subplots: one for the counts and one for percentages\nfig, axes = plt.subplots(1, 2, figsize=(14, 3))\n\n# Plot the confusion matrix with counts\nsns.heatmap(total_confusion_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], ax=axes[0], annot_kws={\"size\": 14})\naxes[0].set_title(\"Confusion Matrix - Counts\", fontsize=16)\naxes[0].set_xlabel(\"Predicted\", fontsize=14)\naxes[0].set_ylabel(\"Actual\", fontsize=14)\n\n# Increase font size for ticks\naxes[0].tick_params(axis='x', labelsize=12)\naxes[0].tick_params(axis='y', labelsize=12)\n\n# Plot the confusion matrix with formatted percentages by row\nsns.heatmap(percentage_confusion_matrix_by_row, annot=formatted_percentage_matrix, fmt=\"s\", cmap=\"Blues\", xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], ax=axes[1], annot_kws={\"size\": 14})\naxes[1].set_title(\"Confusion Matrix - Percentages (by Row)\", fontsize=16)\naxes[1].set_xlabel(\"Predicted\", fontsize=14)\naxes[1].set_ylabel(\"Actual\", fontsize=14)\n\n# Increase font size for ticks\naxes[1].tick_params(axis='x', labelsize=12)\naxes[1].tick_params(axis='y', labelsize=12)\n\n# Adjust layout for better spacing\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_confusion_matrix = total_id_no_argmax\n\n# Calculate the percentage confusion matrix by row\nrow_sums = np.sum(total_confusion_matrix, axis=1, keepdims=True)  # Sum of each row\npercentage_confusion_matrix_by_row = total_confusion_matrix / row_sums * 100  # Normalize by row\n\n# Manually format the percentages by adding '%' to each label\nformatted_percentage_matrix = np.array([['{:.2f}%'.format(val) for val in row] for row in percentage_confusion_matrix_by_row])\n\n# Create a figure with two subplots: one for the counts and one for percentages\nfig, axes = plt.subplots(1, 2, figsize=(14, 3))\n\n# Plot the confusion matrix with counts\nsns.heatmap(total_confusion_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], ax=axes[0], annot_kws={\"size\": 14})\naxes[0].set_title(\"Confusion Matrix - Counts\", fontsize=16)\naxes[0].set_xlabel(\"Predicted\", fontsize=14)\naxes[0].set_ylabel(\"Actual\", fontsize=14)\n\n# Increase font size for ticks\naxes[0].tick_params(axis='x', labelsize=12)\naxes[0].tick_params(axis='y', labelsize=12)\n\n# Plot the confusion matrix with formatted percentages by row\nsns.heatmap(percentage_confusion_matrix_by_row, annot=formatted_percentage_matrix, fmt=\"s\", cmap=\"Blues\", xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], ax=axes[1], annot_kws={\"size\": 14})\naxes[1].set_title(\"Confusion Matrix - Percentages (by Row)\", fontsize=16)\naxes[1].set_xlabel(\"Predicted\", fontsize=14)\naxes[1].set_ylabel(\"Actual\", fontsize=14)\n\n# Increase font size for ticks\naxes[1].tick_params(axis='x', labelsize=12)\naxes[1].tick_params(axis='y', labelsize=12)\n\n# Adjust layout for better spacing\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_confusion_matrix = total_id_custom_logic\n\n# Calculate the percentage confusion matrix by row\nrow_sums = np.sum(total_confusion_matrix, axis=1, keepdims=True)  # Sum of each row\npercentage_confusion_matrix_by_row = total_confusion_matrix / row_sums * 100  # Normalize by row\n\n# Manually format the percentages by adding '%' to each label\nformatted_percentage_matrix = np.array([['{:.2f}%'.format(val) for val in row] for row in percentage_confusion_matrix_by_row])\n\n# Create a figure with two subplots: one for the counts and one for percentages\nfig, axes = plt.subplots(1, 2, figsize=(14, 3))\n\n# Plot the confusion matrix with counts\nsns.heatmap(total_confusion_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], ax=axes[0], annot_kws={\"size\": 14})\naxes[0].set_title(\"Confusion Matrix - Counts\", fontsize=16)\naxes[0].set_xlabel(\"Predicted\", fontsize=14)\naxes[0].set_ylabel(\"Actual\", fontsize=14)\n\n# Increase font size for ticks\naxes[0].tick_params(axis='x', labelsize=12)\naxes[0].tick_params(axis='y', labelsize=12)\n\n# Plot the confusion matrix with formatted percentages by row\nsns.heatmap(percentage_confusion_matrix_by_row, annot=formatted_percentage_matrix, fmt=\"s\", cmap=\"Blues\", xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], ax=axes[1], annot_kws={\"size\": 14})\naxes[1].set_title(\"Confusion Matrix - Percentages (by Row)\", fontsize=16)\naxes[1].set_xlabel(\"Predicted\", fontsize=14)\naxes[1].set_ylabel(\"Actual\", fontsize=14)\n\n# Increase font size for ticks\naxes[1].tick_params(axis='x', labelsize=12)\naxes[1].tick_params(axis='y', labelsize=12)\n\n# Adjust layout for better spacing\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}