{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":366714,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":304057,"modelId":324528}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport math\nimport pandas as pd\nimport numpy as np\nimport cv2\nfrom tqdm import tqdm\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.utils import class_weight, shuffle\nimport warnings\nfrom sklearn.model_selection import train_test_split\n#warnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:03.625596Z","iopub.execute_input":"2025-04-30T15:17:03.626209Z","iopub.status.idle":"2025-04-30T15:17:03.631431Z","shell.execute_reply.started":"2025-04-30T15:17:03.626184Z","shell.execute_reply":"2025-04-30T15:17:03.630421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fixing rnadom seed\nnp.random.seed(2020)\ntf.random.set_seed(2020)\nseed = 2020\nIMG_SIZE=224","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:06.042312Z","iopub.execute_input":"2025-04-30T15:17:06.043052Z","iopub.status.idle":"2025-04-30T15:17:06.047648Z","shell.execute_reply.started":"2025-04-30T15:17:06.043026Z","shell.execute_reply":"2025-04-30T15:17:06.04659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir('/kaggle/input/aptos2019-blindness-detection'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:09.634635Z","iopub.execute_input":"2025-04-30T15:17:09.634981Z","iopub.status.idle":"2025-04-30T15:17:09.640682Z","shell.execute_reply.started":"2025-04-30T15:17:09.634957Z","shell.execute_reply":"2025-04-30T15:17:09.639745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\n\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:12.118902Z","iopub.execute_input":"2025-04-30T15:17:12.119178Z","iopub.status.idle":"2025-04-30T15:17:12.136861Z","shell.execute_reply.started":"2025-04-30T15:17:12.119159Z","shell.execute_reply":"2025-04-30T15:17:12.135908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"id_code = df_train['id_code']\ndiagnosis = df_train['diagnosis']\n\nid_code, diagnosis = shuffle(id_code, diagnosis, random_state=seed)\n\ntrain_x, valid_x, train_y, valid_y = train_test_split(id_code, diagnosis, test_size=0.15, stratify=diagnosis, random_state=seed)\n\ntrain_x = train_x.reset_index(drop=True)\nvalid_x = valid_x.reset_index(drop=True)\ntrain_y = train_y.reset_index(drop=True)\nvalid_y = valid_y.reset_index(drop=True)\n\n\nprint(train_x.shape)\nprint(train_y.shape)\nprint(valid_x.shape)\nprint(valid_y.shape)\nprint(df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:15.422051Z","iopub.execute_input":"2025-04-30T15:17:15.422395Z","iopub.status.idle":"2025-04-30T15:17:15.438401Z","shell.execute_reply.started":"2025-04-30T15:17:15.422368Z","shell.execute_reply":"2025-04-30T15:17:15.437436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_distribution(df, labels, t_cv_te='train'):\n    '''\n    This function prints the distribution of output variable in a given dataframe and also prints the stats\n    '''\n    print('-'*80)\n    class_distribution = labels.value_counts().sort_index()\n    class_distribution.plot(kind='bar')\n    plt.xlabel('Class')\n    plt.ylabel('Data points per Class')\n    plt.title('Distribution of yi in ' + t_cv_te + ' data')\n    plt.grid()\n    plt.show()\n\n    sorted_yi = np.argsort(-class_distribution.values)\n    for i in sorted_yi:\n        print('Number of data points in class', i+1, ':',class_distribution.values[i], '(', np.round((class_distribution.values[i]/labels.shape[0]*100), 3), '%)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:18.286195Z","iopub.execute_input":"2025-04-30T15:17:18.286778Z","iopub.status.idle":"2025-04-30T15:17:18.29295Z","shell.execute_reply.started":"2025-04-30T15:17:18.286748Z","shell.execute_reply":"2025-04-30T15:17:18.29202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_distribution(df=df_train, labels=df_train['diagnosis'], t_cv_te='train')\nplot_distribution(df=valid_x, labels=valid_y, t_cv_te='valid')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:23.94887Z","iopub.execute_input":"2025-04-30T15:17:23.949174Z","iopub.status.idle":"2025-04-30T15:17:24.271257Z","shell.execute_reply.started":"2025-04-30T15:17:23.949151Z","shell.execute_reply":"2025-04-30T15:17:24.270393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image(path, sigmaX=10, img_size=256, enhance=True):\n    '''\n    Reads an image, converts to RGB, resizes, and optionally enhances by blending with a Gaussian blurred version.\n    '''\n    image = cv2.imread(path)\n    if image is None:\n        raise ValueError(f\"❌ Failed to load image: {path}\")\n\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = cv2.resize(image, (img_size, img_size))\n\n    if enhance:\n        image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), sigmaX), -4,128)\n\n    return image\n\ndef image_preprocessing_1(df, labels, ben=False, sub_dir='train_images', sigmaX=10, ext='.png', img_size=256, seed=42):\n    '''\n    Displays a 5x5 grid of processed images with their labels and ids.\n    '''\n    fig = plt.figure(figsize=(25, 16))\n    \n    for class_id in sorted(labels.unique()):\n        samples = df[df['diagnosis'] == class_id].sample(5, random_state=seed)\n        \n        for i, (idx, row) in enumerate(samples.iterrows()):\n            ax = fig.add_subplot(5, 5, class_id * 5 + i + 1, xticks=[], yticks=[])\n\n            \n            path = os.path.join('/kaggle/input/aptos2019-blindness-detection', sub_dir, row['id_code'] + ext)\n            \n            try:\n                image = preprocess_image(path, sigmaX=sigmaX, img_size=img_size, enhance=ben)\n            except Exception as e:\n                print(e)\n                continue\n            \n            ax.imshow(image)\n            ax.set_title(f'Label: {class_id}-{row[\"id_code\"]}', fontsize=8)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:28.022341Z","iopub.execute_input":"2025-04-30T15:17:28.022675Z","iopub.status.idle":"2025-04-30T15:17:28.034425Z","shell.execute_reply.started":"2025-04-30T15:17:28.022632Z","shell.execute_reply":"2025-04-30T15:17:28.03328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_preprocessing_1(df_train,df_train['diagnosis'],ben=False,sub_dir = 'train_images')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:32.371271Z","iopub.execute_input":"2025-04-30T15:17:32.371571Z","iopub.status.idle":"2025-04-30T15:17:38.211631Z","shell.execute_reply.started":"2025-04-30T15:17:32.371551Z","shell.execute_reply":"2025-04-30T15:17:38.209255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_preprocessing_1(df_train,df_train['diagnosis'],ben=True,sub_dir = 'train_images',sigmaX=20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:17:45.781868Z","iopub.execute_input":"2025-04-30T15:17:45.782558Z","iopub.status.idle":"2025-04-30T15:17:52.519825Z","shell.execute_reply.started":"2025-04-30T15:17:45.782527Z","shell.execute_reply":"2025-04-30T15:17:52.518118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def crop_dark_extras(img,tol=7):\n    '''\n    This function is used to crop out the additions dark areas in the images as this information is not useful\n    '''\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            \n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:18:47.08385Z","iopub.execute_input":"2025-04-30T15:18:47.084146Z","iopub.status.idle":"2025-04-30T15:18:47.091396Z","shell.execute_reply.started":"2025-04-30T15:18:47.084123Z","shell.execute_reply":"2025-04-30T15:18:47.090477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport pandas as pd  # Add this import for handling CSV data\n\n# 1. Load your training data\ndf_train = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# 2. Verify the DataFrame was loaded\n#print(df_train.head())  # Check the first few rows\n\n# 3. Load and display the image\nimage_path = \"/kaggle/input/aptos2019-blindness-detection/train_images/\" + df_train['id_code'][1] + '.png'\n#print(f\"Loading image from: {image_path}\")  # Debug path\n\nimg = cv2.imread(image_path)\nif img is None:\n    raise FileNotFoundError(f\"Image not found at: {image_path}\")\n\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nplt.imshow(img)\nplt.axis('on')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:18:58.543929Z","iopub.execute_input":"2025-04-30T15:18:58.544257Z","iopub.status.idle":"2025-04-30T15:18:59.654054Z","shell.execute_reply.started":"2025-04-30T15:18:58.544232Z","shell.execute_reply":"2025-04-30T15:18:59.652987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = crop_dark_extras(img=img,tol=7)\nimg = cv2.resize(img, (224,224))\nplt.imshow(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:19:05.202644Z","iopub.execute_input":"2025-04-30T15:19:05.203026Z","iopub.status.idle":"2025-04-30T15:19:05.612314Z","shell.execute_reply.started":"2025-04-30T15:19:05.203002Z","shell.execute_reply":"2025-04-30T15:19:05.611417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image2(image, sigmaX=10):\n    \"\"\"\n    The whole preprocessing pipeline:\n    1. Read in image\n    2. Apply masks\n    3. Resize image to desired size\n    4. Add Gaussian noise to increase Robustness\n    \n    :param img: A NumPy Array that will be cropped\n    :param sigmaX: Value used for add GaussianBlur to the image\n    \n    :return: A NumPy array containing the preprocessed image\n    \"\"\"\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#    image = crop_image_from_gray(image)\n    image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n    image = cv2.addWeighted (image,4, cv2.GaussianBlur(image, (0,0) ,sigmaX), -4, 96)\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:23:21.917817Z","iopub.execute_input":"2025-04-30T15:23:21.918115Z","iopub.status.idle":"2025-04-30T15:23:21.924079Z","shell.execute_reply.started":"2025-04-30T15:23:21.918094Z","shell.execute_reply":"2025-04-30T15:23:21.923189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def Preprocessing_images(df, sub_dir, ben=False, sigmaX=10, img_size=224, ext='.png'):\n    '''\n    This function reads images, converts them to RGB, resizes the image and also enhances the images/\n    by blending them with gaussianBluered version of itself\n    '''\n    rows = df.shape[0]\n    df_as_nd_array = np.empty((rows,img_size,img_size,3),dtype=np.uint8)\n\n    for i,id_code in enumerate(tqdm(df)):\n        path = \"/kaggle/input/aptos2019-blindness-detection\"+sub_dir+'/'+ id_code + ext\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = crop_dark_extras(img=img,tol=7)\n        img = cv2.resize(img, (img_size,img_size))\n        if ben == True:\n\n            img = cv2.addWeighted(img,4,cv2.GaussianBlur(img,(0,0),sigmaX),-4,128)\n        df_as_nd_array[i,:,:,:] = img\n        \n    return df_as_nd_array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:23:25.216363Z","iopub.execute_input":"2025-04-30T15:23:25.216764Z","iopub.status.idle":"2025-04-30T15:23:25.223917Z","shell.execute_reply.started":"2025-04-30T15:23:25.216735Z","shell.execute_reply":"2025-04-30T15:23:25.222894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom tqdm import tqdm\n\n# Settings\nIMG_SIZE = 224\nsigmaX = 50\nben = True\n\n# Load the full training CSV\ndf_all = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# Shuffle for reproducibility\ndf_all = df_all.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# Split into train (3112), valid (550), test (1928)\ntrain_x = df_all.iloc[:3112][['id_code', 'diagnosis']]\nvalid_x = df_all.iloc[3112:3662][['id_code', 'diagnosis']]\n#test_x = df_all.iloc[3662:]['id_code'].to_frame()\n\n# Test CSV (for consistency with original structure)\n#df_test = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\n\n# Image preprocessing function\ndef preprocess_image(image_path, sigmaX=10, img_size=224, ben=False):\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = cv2.resize(image, (img_size, img_size))\n    if ben:\n        image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), sigmaX), -4, 128)\n    return image\n\n# Batch image preprocessing\ndef Preprocessing_images(df, sub_dir, sigmaX=10, img_size=224, ben=False):\n    images = []\n    for _, row in tqdm(df.iterrows(), total=len(df), desc=f\"Processing {sub_dir}\"):\n        image_path = os.path.join(f\"/kaggle/input/aptos2019-blindness-detection/{sub_dir}\", row['id_code'] + '.png')\n        if os.path.exists(image_path):\n            img = preprocess_image(image_path, sigmaX=sigmaX, img_size=img_size, ben=ben)\n            images.append(img)\n        else:\n            print(f\"Image not found: {image_path}\")\n    return np.array(images)\n\n# === Load or Preprocess x_train ===\nif os.path.exists('x_train_224_2019.npy'):\n    x_train = np.load(\"x_train_224_2019.npy\")\nelse:\n    x_train = Preprocessing_images(df=train_x, sub_dir='train_images', ben=ben, sigmaX=sigmaX, img_size=IMG_SIZE)\n    np.save('x_train_224_2019.npy', x_train)\n\n# === Load or Preprocess x_valid ===\nif os.path.exists('x_valid_224_2019.npy'):\n    x_valid = np.load(\"x_valid_224_2019.npy\")\nelse:\n    x_valid = Preprocessing_images(df=valid_x, sub_dir='train_images', ben=ben, sigmaX=sigmaX, img_size=IMG_SIZE)\n    np.save('x_valid_224_2019.npy', x_valid)\n\n# === Load or Preprocess x_test ===\n#if os.path.exists('x_test_224_2019.npy'):\n #   x_test = np.load(\"x_test_224_2019.npy\")\n#else:\n  #  x_test = Preprocessing_images(df=test_x, sub_dir='test_images', ben=ben, sigmaX=sigmaX, img_size=IMG_SIZE)\n   # np.save('x_test_224_2019.npy', x_test)\n\n# === Print final shapes ===\nprint(x_train.shape)  # (3112, 224, 224, 3)\nprint(x_valid.shape)  # (550, 224, 224, 3)\n#print(x_test.shape)   # (1928, 224, 224, 3)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:23:43.533199Z","iopub.execute_input":"2025-04-30T15:23:43.533488Z","iopub.status.idle":"2025-04-30T15:46:04.155707Z","shell.execute_reply.started":"2025-04-30T15:23:43.533454Z","shell.execute_reply":"2025-04-30T15:46:04.154565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport cv2\nfrom tqdm import tqdm\nimport pandas as pd\n\n# Load the CSVs containing the dataset information\n#train_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\ntest_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\n\n# Extract the relevant columns for train and validation DataFrames\n#train_x = train_df[['id_code', 'diagnosis']]  # train has 'diagnosis' column\ntest_x = valid_df[['id_code']]  # valid has only 'id_code' column\n\ndef preprocess_image(image_path, sigmaX=10, img_size=224, ben=False):\n    \"\"\"\n    This function loads an image, converts it to RGB, resizes it to the given size, and applies optional enhancement\n    (Gaussian blur blending).\n    \"\"\"\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  # Convert to RGB\n\n    # Resize the image\n    image = cv2.resize(image, (img_size, img_size))\n\n    if ben:\n        image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), sigmaX), -4, 128)\n\n    return image\n\ndef preprocess_images_from_directory(df, sub_dir, sigmaX=10, img_size=224, ben=False):\n    processed_images = []\n\n    for idx, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing Images\"):\n        image_path = os.path.join(sub_dir, row['id_code'] + '.png')\n        if os.path.exists(image_path):\n            image = preprocess_image(image_path, sigmaX, img_size, ben)\n            processed_images.append(image)\n        else:\n            print(f\"Image {image_path} not found.\")\n    \n    return np.array(processed_images)\n\n# Directory paths for train and valid images\ntest_images_dir = '/kaggle/input/aptos2019-blindness-detection/test_images/'\n\n# Preprocess the validation images\nx_test = preprocess_images_from_directory(df=test_x, sub_dir=test_images_dir, ben=True, sigmaX=50, img_size=224)\nnp.save('x_test_224_2019.npy', x_test)  # Save the preprocessed images as numpy file\n\n# Print the shapes of the processed datasets\nprint(x_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T15:46:19.960441Z","iopub.execute_input":"2025-04-30T15:46:19.960885Z","iopub.status.idle":"2025-04-30T15:55:44.453934Z","shell.execute_reply.started":"2025-04-30T15:46:19.960853Z","shell.execute_reply":"2025-04-30T15:55:44.453015Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1.4 Transforming Output variables.****","metadata":{}},{"cell_type":"code","source":"#below code converts categorical outputs into onhot encoded like vectors\ny_train = pd.get_dummies(train_y).values\ny_valid = pd.get_dummies(valid_y).values\n\nprint(y_train.shape)\nprint(y_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:03:46.2931Z","iopub.execute_input":"2025-04-30T16:03:46.293481Z","iopub.status.idle":"2025-04-30T16:03:46.305743Z","shell.execute_reply.started":"2025-04-30T16:03:46.293437Z","shell.execute_reply":"2025-04-30T16:03:46.304686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ordinal_regression(y):\n    '''\n    This function takes in the categorical(one hot encoded like) output varaible as input example [0,0,0,1,0] the output will be\\\n    [1,1,1,1,0] i.e all the categories before actual category are set to one\n    '''\n    y_multi = np.empty(y.shape, dtype=y.dtype)\n    y_multi[:,4] = y[:,4] \n\n    for i in range(3,-1,-1):\n        y_multi[:,i] = np.logical_or(y[:,i],y_multi[:,i+1])\n\n    print(y_multi.shape)\n    return y_multi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:04:54.080698Z","iopub.execute_input":"2025-04-30T16:04:54.081032Z","iopub.status.idle":"2025-04-30T16:04:54.086785Z","shell.execute_reply.started":"2025-04-30T16:04:54.081009Z","shell.execute_reply":"2025-04-30T16:04:54.085643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train=ordinal_regression(y_train)\ny_valid=ordinal_regression(y_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:04:57.95687Z","iopub.execute_input":"2025-04-30T16:04:57.957216Z","iopub.status.idle":"2025-04-30T16:04:57.962818Z","shell.execute_reply.started":"2025-04-30T16:04:57.957191Z","shell.execute_reply":"2025-04-30T16:04:57.961748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train = np.load('x_train_224_2019.npy')  # Adjust the file path as needed\nprint(x_train.shape)  # Ensure it's (3112, 224, 224, 3)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:05:00.65158Z","iopub.execute_input":"2025-04-30T16:05:00.651988Z","iopub.status.idle":"2025-04-30T16:05:00.902726Z","shell.execute_reply.started":"2025-04-30T16:05:00.65196Z","shell.execute_reply":"2025-04-30T16:05:00.901578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_confusion_matrix(test_y, predict_y):\n    '''\n    Utility function to plot confuion matirx/ classification report, precision and recall matrices\n    '''\n    C = confusion_matrix(test_y, predict_y)\n    print(\"Number of misclassified points \",(len(test_y)-np.trace(C))/len(test_y)*100)\n    # C = 5,5 matrix, each cell (i,j) represents number of points of class i are predicted class j\n    \n    A =(((C.T)/(C.sum(axis=1))).T)\n\n    \n    B =(C/C.sum(axis=0))\n \n    \n    labels = [1,2,3,4,5]\n    cmap=sns.light_palette(\"purple\")\n    # representing A in heatmap format\n    print(\"-\"*50, \"Confusion matrix\", \"-\"*50)\n    plt.figure(figsize=(10,5))\n    sns.heatmap(C, annot=True, cmap=cmap, fmt=\".3f\", xticklabels=labels, yticklabels=labels)\n    plt.xlabel('Predicted Class')\n    plt.ylabel('Original Class')\n    plt.show()\n\n    print(\"-\"*50, \"Precision matrix\", \"-\"*50)\n    plt.figure(figsize=(10,5))\n    sns.heatmap(B, annot=True, cmap=cmap, fmt=\".3f\", xticklabels=labels, yticklabels=labels)\n    plt.xlabel('Predicted Class')\n    plt.ylabel('Original Class')\n    plt.show()\n    print(\"Sum of columns in precision matrix\",B.sum(axis=0))\n    \n    # representing B in heatmap format\n    print(\"-\"*50, \"Recall matrix\"    , \"-\"*50)\n    plt.figure(figsize=(10,5))\n    sns.heatmap(A, annot=True, cmap=cmap, fmt=\".3f\", xticklabels=labels, yticklabels=labels)\n    plt.xlabel('Predicted Class')\n    plt.ylabel('Original Class')\n    plt.show()\n    print(\"Sum of rows in precision matrix\",A.sum(axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:05:09.050061Z","iopub.execute_input":"2025-04-30T16:05:09.051063Z","iopub.status.idle":"2025-04-30T16:05:09.066942Z","shell.execute_reply.started":"2025-04-30T16:05:09.051024Z","shell.execute_reply":"2025-04-30T16:05:09.065738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(x_train[1])  # Display the image at index 1\nplt.axis('on')  # Optionally, turn off axis labels\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:05:13.074194Z","iopub.execute_input":"2025-04-30T16:05:13.074532Z","iopub.status.idle":"2025-04-30T16:05:13.371054Z","shell.execute_reply.started":"2025-04-30T16:05:13.074507Z","shell.execute_reply":"2025-04-30T16:05:13.369752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(x_train[1].reshape(IMG_SIZE,IMG_SIZE,3))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:05:17.301804Z","iopub.execute_input":"2025-04-30T16:05:17.302622Z","iopub.status.idle":"2025-04-30T16:05:17.616173Z","shell.execute_reply.started":"2025-04-30T16:05:17.302586Z","shell.execute_reply":"2025-04-30T16:05:17.615194Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2# . Modelling# ","metadata":{}},{"cell_type":"code","source":"#sklearn's method\ndef cohen_kappa_score_cus(y1, y2, labels=None, weights=\"quadratic\", sample_weight=None):\n    confusion = confusion_matrix(y1, y2, labels=labels,\n                                 sample_weight=sample_weight)\n    n_classes = confusion.shape[0]\n    sum0 = np.sum(confusion, axis=0)\n    sum1 = np.sum(confusion, axis=1)\n    expected = np.outer(sum0, sum1) / np.sum(sum0)\n\n    if weights is None:\n        w_mat = np.ones([n_classes, n_classes], dtype=np.int)\n        w_mat.flat[:: n_classes + 1] = 0\n    elif weights == \"linear\" or weights == \"quadratic\":\n        w_mat = np.zeros([n_classes, n_classes], dtype=np.int)\n        w_mat += np.arange(n_classes)\n        if weights == \"linear\":\n            w_mat = np.abs(w_mat - w_mat.T)\n        else:\n            w_mat = (w_mat - w_mat.T) ** 2\n    else:\n        raise ValueError(\"Unknown kappa weighting type.\")\n\n    k = np.sum(w_mat * confusion) / np.sum(w_mat * expected)\n    return 1 - k","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:33:29.775334Z","iopub.execute_input":"2025-04-30T16:33:29.775696Z","iopub.status.idle":"2025-04-30T16:33:29.784093Z","shell.execute_reply.started":"2025-04-30T16:33:29.775673Z","shell.execute_reply":"2025-04-30T16:33:29.782749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nCustom metric function defined to get cohen kappa score \n'''\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nfrom keras.callbacks import Callback, ModelCheckpoint\n\ndef kappa(y_true, y_pred):\n        \n        y_actual = tf.cast(tf.math.reduce_sum(y_true,axis=1)- 1,dtype=tf.int32)\n        y_pred = tf.cast(y_pred,dtype=tf.float32)\n        limit = tf.constant([0.5],dtype=tf.float32)\n    \n        y_pred_reversed = tf.reverse(tf.math.greater(y_pred,limit),axis=[1])\n        y_pred_reversed = tf.cast(y_pred_reversed,tf.int32)\n    \n        indices = tf.math.argmax(y_pred_reversed,axis=1)\n        y_pred_ordinal = 4-indices\n        y_pred_ordinal = tf.cast(y_pred_ordinal,tf.int32)\n        \n        \n        # calculating using sklearn's kappa socre function\n        val_kappa = tf.py_function(cohen_kappa_score_cus,[y_actual,y_pred_ordinal,],tf.double)\n#         print(f\"val_kappa: {val_kappa:.4f}\")\n        return val_kappa","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:33:34.189801Z","iopub.execute_input":"2025-04-30T16:33:34.190102Z","iopub.status.idle":"2025-04-30T16:33:34.197149Z","shell.execute_reply.started":"2025-04-30T16:33:34.19008Z","shell.execute_reply":"2025-04-30T16:33:34.195841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2.1 Models with self built architectures\n","metadata":{}},{"cell_type":"code","source":"import keras as keras\n\n## from keras.layers.normalization import BatchNormalization \"chat\"\n\nfrom keras.layers import BatchNormalization \n\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalMaxPooling2D, GlobalAveragePooling2D\nfrom keras import backend as K\nfrom keras.layers import Reshape, Permute, Activation\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.callbacks import TensorBoard , EarlyStopping\nimport time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:33:37.97424Z","iopub.execute_input":"2025-04-30T16:33:37.974545Z","iopub.status.idle":"2025-04-30T16:33:37.979703Z","shell.execute_reply.started":"2025-04-30T16:33:37.974524Z","shell.execute_reply":"2025-04-30T16:33:37.978706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nName = \"Model_baseline_2019-{}\".format(int(time.time()))\n\ntensorboard = TensorBoard(log_dir='logs_new6\\{}'.format(Name))\n\nmodel_base = tf.keras.Sequential()\nmodel_base.add(tf.keras.layers.Conv2D(16, kernel_size=(3, 3),padding='same',\n                 activation='relu',\n                 input_shape=(IMG_SIZE,IMG_SIZE,3)))\n\nmodel_base.add(tf.keras.layers.Conv2D(32, (3, 3), activation='relu',padding='same'))\nmodel_base.add(tf.keras.layers.MaxPooling2D(pool_size=(2, 2)))\nmodel_base.add(tf.keras.layers.Dropout(0.3))\n\n\nmodel_base.add(tf.keras.layers.Flatten())\n\nmodel_base.add(tf.keras.layers.BatchNormalization())\n\nmodel_base.add(tf.keras.layers.Dense(5, activation='sigmoid'))\n\nmodel_base.compile(loss='binary_crossentropy',\n              optimizer=tf.keras.optimizers.Adam(decay=0.001),\n              metrics=[kappa])\n\nfilepath = \"weights_baseline.best.weights.h5\"  # Change the extension to .weights.h5\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(filepath, monitor='val_kappa', verbose=1, save_best_only=True, mode='max', save_weights_only=True)\n\n\ncallbacks_list = [checkpoint] + [tensorboard]\n\nmodel_base.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:36:18.293863Z","iopub.execute_input":"2025-04-30T16:36:18.294163Z","iopub.status.idle":"2025-04-30T16:36:18.409226Z","shell.execute_reply.started":"2025-04-30T16:36:18.294145Z","shell.execute_reply":"2025-04-30T16:36:18.408219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# From the above confusion matrices we can see that model_two has improved from our baseline model performance on classes 4 and 5 has also improved to a certain extent yet the dominance of class 3 is still evident.","metadata":{}},{"cell_type":"markdown","source":"# 2.2.1 Transfer Learning with Image Augmentation.\n","metadata":{}},{"cell_type":"markdown","source":"# Data Augmentation Technique-Image Data Generator.\n","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom matplotlib import pyplot as plt\n\n# define data preparation\nshift = 0.2\ndatagen = ImageDataGenerator(horizontal_flip=True, vertical_flip=True, rotation_range=180, validation_split=0.15)\n\nfor x_batch, y_batch in datagen.flow(x_train, y_train, batch_size=9):\n    # create a grid of 3x3 images\n    for i in range(0, 9):\n        plt.subplot(330 + 1 + i)\n        plt.imshow(x_batch[i].astype('uint8'))\n    # show the plot\n    plt.show()\n    break\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:57:26.131754Z","iopub.execute_input":"2025-04-30T17:57:26.132115Z","iopub.status.idle":"2025-04-30T17:57:28.892906Z","shell.execute_reply.started":"2025-04-30T17:57:26.132089Z","shell.execute_reply":"2025-04-30T17:57:28.891834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_generator = datagen.flow(x_train, y_train, batch_size=8,subset='training',seed=7)\nvalidation_generator = datagen.flow(x_train, y_train, batch_size=8,subset='validation',seed=7)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:58:08.112628Z","iopub.execute_input":"2025-04-30T17:58:08.112971Z","iopub.status.idle":"2025-04-30T17:58:09.878914Z","shell.execute_reply.started":"2025-04-30T17:58:08.112947Z","shell.execute_reply":"2025-04-30T17:58:09.87802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DenseNet121\n","metadata":{}},{"cell_type":"code","source":"from keras.applications import DenseNet121\n\n\nName = \"Model_image_gen_2019-{}\".format(int(time.time()))\n\ntensorboard = TensorBoard(log_dir='logs_new6\\{}'.format(Name))\n\ndensenet = tf.keras.applications.DenseNet121(weights='imagenet',\n                       include_top=False,\n                       input_shape=(224,224,3))\nmodel_three = tf.keras.Sequential()\nmodel_three.add(densenet)\nmodel_three.add(tf.keras.layers.GlobalAveragePooling2D())\nmodel_three.add(tf.keras.layers.Dropout(0.5))\nmodel_three.add(tf.keras.layers.Dense(5, activation='sigmoid'))\n    \nmodel_three.compile(\n    loss='binary_crossentropy',\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.00005),\n    metrics=[kappa]\n)\n\n    \nfilepath=\"weights_image_gen.keras\"\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(filepath, monitor='val_kappa', verbose=1, save_best_only=True, mode='max')\n\ncallbacks_list = [checkpoint] + [tensorboard]\n\nmodel_three.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T18:13:43.877146Z","iopub.execute_input":"2025-04-30T18:13:43.878029Z","iopub.status.idle":"2025-04-30T18:13:45.717491Z","shell.execute_reply.started":"2025-04-30T18:13:43.877998Z","shell.execute_reply":"2025-04-30T18:13:45.716532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''''''''''''''''''''''\nimport tensorflow as tf\nfrom keras.applications import DenseNet121\nfrom keras.callbacks import TensorBoard, ModelCheckpoint\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport matplotlib.pyplot as plt\nimport time\nimport os\n\n# اسم التجربة للتنظيم في TensorBoard\nName = \"Model_image_gen_2019-{}\".format(int(time.time()))\ntensorboard = TensorBoard(log_dir=os.path.join('logs_new6', Name))\n\n# تحميل DenseNet بدون الطبقة النهائية\ndensenet = tf.keras.applications.DenseNet121(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\n\n# بناء النموذج النهائي\nmodel_three = tf.keras.Sequential()\nmodel_three.add(densenet)\nmodel_three.add(tf.keras.layers.GlobalAveragePooling2D())\nmodel_three.add(tf.keras.layers.Dropout(0.5))\nmodel_three.add(tf.keras.layers.Dense(5, activation='sigmoid'))  # عدد الفئات 5\n\n# Compile\nmodel_three.compile(\n    loss='binary_crossentropy',\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.00005),\n    metrics=['accuracy']\n)\n\n# ModelCheckpoint\nfilepath = \"weights_image_gen.keras\"\ncheckpoint = ModelCheckpoint(filepath, monitor='val_accuracy', verbose=1, save_best_only=True, mode='max')\n\ncallbacks_list = [checkpoint, tensorboard]\n\n# Data augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=10,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True\n)\ndatagen.fit(x_train)\n\n# تدريب النموذج\nhistory_three = model_three.fit(\n    datagen.flow(x_train, y_train, batch_size=8),\n    steps_per_epoch=len(x_train) // 8,\n    epochs=10,\n    initial_epoch=0,\n    verbose=1,\n    validation_data=(x_valid, y_valid),\n    validation_steps=len(x_valid) // 8,\n    callbacks=callbacks_list\n)\n\n# ==========================\n# رسم Loss و Accuracy\n# ==========================\nhistory = history_three.history\n\nplt.figure(figsize=(12, 5))\n\n# Loss\nplt.subplot(1, 2, 1)\nplt.plot(history['loss'], label='Training Loss')\nplt.plot(history['val_loss'], label='Validation Loss')\nplt.title('Loss over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Binary Crossentropy')\nplt.legend()\n\n# Accuracy\nplt.subplot(1, 2, 2)\nplt.plot(history['accuracy'], label='Training Accuracy')\nplt.plot(history['val_accuracy'], label='Validation Accuracy')\nplt.title('Accuracy over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n''''''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T18:30:49.247144Z","iopub.execute_input":"2025-04-30T18:30:49.247439Z","iopub.status.idle":"2025-04-30T18:30:49.255171Z","shell.execute_reply.started":"2025-04-30T18:30:49.247419Z","shell.execute_reply":"2025-04-30T18:30:49.253979Z"}},"outputs":[],"execution_count":null}]}