{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport tensorflow as tf\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split\nfrom pathlib import Path\nimport os.path\nimport warnings\nimport tensorflow as tf\nprint(\"GPU Available:\", tf.config.list_physical_devices('GPU'))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.241763Z","iopub.execute_input":"2025-04-13T17:22:44.242053Z","iopub.status.idle":"2025-04-13T17:22:44.249503Z","shell.execute_reply.started":"2025-04-13T17:22:44.242032Z","shell.execute_reply":"2025-04-13T17:22:44.248700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = '/kaggle/input/happy-whale-and-dolphin/train_images'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.250839Z","iopub.execute_input":"2025-04-13T17:22:44.251095Z","iopub.status.idle":"2025-04-13T17:22:44.267971Z","shell.execute_reply.started":"2025-04-13T17:22:44.251075Z","shell.execute_reply":"2025-04-13T17:22:44.267112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.269670Z","iopub.execute_input":"2025-04-13T17:22:44.269896Z","iopub.status.idle":"2025-04-13T17:22:44.334847Z","shell.execute_reply.started":"2025-04-13T17:22:44.269877Z","shell.execute_reply":"2025-04-13T17:22:44.334229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv['image']  = train_csv['image'].apply(lambda x : train + '/'+ x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.335922Z","iopub.execute_input":"2025-04-13T17:22:44.336179Z","iopub.status.idle":"2025-04-13T17:22:44.356299Z","shell.execute_reply.started":"2025-04-13T17:22:44.336160Z","shell.execute_reply":"2025-04-13T17:22:44.355465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv['species'] = train_csv['species'].replace({\n    'false_killer_whale' : 'killer_whale',\n    'bottlenose_dolpin' : 'bottlenose_dolphin',\n    'kiler_whale' : 'killer_whale',\n    'short_finned_pilot_whale' : 'pilot_whale',\n    'long_finned_pilot_whale' :  'pilot_whale',\n    'pygmy_killer_whale' : 'killer_whale'\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.357104Z","iopub.execute_input":"2025-04-13T17:22:44.357312Z","iopub.status.idle":"2025-04-13T17:22:44.384989Z","shell.execute_reply.started":"2025-04-13T17:22:44.357291Z","shell.execute_reply":"2025-04-13T17:22:44.384160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv['species'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.385743Z","iopub.execute_input":"2025-04-13T17:22:44.386035Z","iopub.status.idle":"2025-04-13T17:22:44.394649Z","shell.execute_reply.started":"2025-04-13T17:22:44.386014Z","shell.execute_reply":"2025-04-13T17:22:44.393837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_csv['species'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.395290Z","iopub.execute_input":"2025-04-13T17:22:44.395497Z","iopub.status.idle":"2025-04-13T17:22:44.411630Z","shell.execute_reply.started":"2025-04-13T17:22:44.395478Z","shell.execute_reply":"2025-04-13T17:22:44.410948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndef adjust_brightness_contrast(image, alpha=1.5, beta=50):\n    return cv2.convertScaleAbs(image, alpha=alpha, beta=beta)\n\ndef equalize_histogram(image):\n    if len(image.shape) == 2:  # Grayscale image\n        return cv2.equalizeHist(image)\n    elif len(image.shape) == 3:  # Color image\n        ycrcb = cv2.cvtColor(image, cv2.COLOR_RGB2YCrCb)\n        y, cr, cb = cv2.split(ycrcb)\n        y_eq = cv2.equalizeHist(y)\n        ycrcb_eq = cv2.merge([y_eq, cr, cb])\n        return cv2.cvtColor(ycrcb_eq, cv2.COLOR_YCrCb2RGB)\n    return image\ndef sharpen_image(image):\n    sharpening_kernel = np.array([[-1, -1, -1],\n                                  [-1,  9, -1],\n                                  [-1, -1, -1]])\n    return cv2.filter2D(image, -1, sharpening_kernel)\n\ndef remove_noise(image):\n    return cv2.GaussianBlur(image, (5, 5), 0)\ndef custom_preprocessing(image):\n    # Convert image from range [0, 1] to [0, 255]\n    image = image * 255.0\n    image = image.astype(np.uint8)\n    \n    # Apply preprocessing steps\n    image = remove_noise(image)\n    image = adjust_brightness_contrast(image)\n    image = equalize_histogram(image)\n    image = sharpen_image(image)\n    image = tf.keras.applications.mobilenet_v2.preprocess_input(image)\n    \n    # Convert image back to range [0, 1]\n    image = image.astype(np.float32) / 255.0\n    \n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.413735Z","iopub.execute_input":"2025-04-13T17:22:44.413926Z","iopub.status.idle":"2025-04-13T17:22:44.425096Z","shell.execute_reply.started":"2025-04-13T17:22:44.413910Z","shell.execute_reply":"2025-04-13T17:22:44.424282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df , test_df = train_test_split(train_csv, test_size = 0.20, shuffle = True, random_state = 42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.426047Z","iopub.execute_input":"2025-04-13T17:22:44.426261Z","iopub.status.idle":"2025-04-13T17:22:44.448687Z","shell.execute_reply.started":"2025-04-13T17:22:44.426233Z","shell.execute_reply":"2025-04-13T17:22:44.447929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = custom_preprocessing,validation_split = 0.2)\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = custom_preprocessing)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.449352Z","iopub.execute_input":"2025-04-13T17:22:44.449616Z","iopub.status.idle":"2025-04-13T17:22:44.453125Z","shell.execute_reply.started":"2025-04-13T17:22:44.449585Z","shell.execute_reply":"2025-04-13T17:22:44.452335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#K = 2\n#skf = StratifiedKFold(n_splits=K, random_state=SEED, shuffle=True)\n\n#DISASTER = df_train['target'] == 1\n# print('Whole Training Set Shape = {}'.format(df_train.shape))\n# print('Whole Training Set Unique keyword Count = {}'.format(df_train['keyword'].nunique()))\n# print('Whole Training Set Target Rate (Disaster) {}/{} (Not Disaster)'.format(df_train[DISASTER]['target_relabeled'].count(), df_train[~DISASTER]['target_relabeled'].count()))\n\n# for fold, (trn_idx, val_idx) in enumerate(skf.split(df_train['text_cleaned'], df_train['target']), 1):\n#     print('\\nFold {} Training Set Shape = {} - Validation Set Shape = {}'.format(fold, df_train.loc[trn_idx, 'text_cleaned'].shape, df_train.loc[val_idx, 'text_cleaned'].shape))\n#     print('Fold {} Training Set Unique keyword Count = {} - Validation Set Unique keyword Count = {}'.format(fold, df_train.loc[trn_idx, 'keyword'].nunique(), df_train.loc[val_idx, 'keyword'].nunique()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.453987Z","iopub.execute_input":"2025-04-13T17:22:44.454260Z","iopub.status.idle":"2025-04-13T17:22:44.468590Z","shell.execute_reply.started":"2025-04-13T17:22:44.454233Z","shell.execute_reply":"2025-04-13T17:22:44.467879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (224, 224),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'training'\n)\nval_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (224, 224),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'validation'\n)\ntest_image = test_gen.flow_from_dataframe(\n    dataframe = test_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (224, 224),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:22:44.469549Z","iopub.execute_input":"2025-04-13T17:22:44.469846Z","iopub.status.idle":"2025-04-13T17:23:31.705508Z","shell.execute_reply.started":"2025-04-13T17:22:44.469820Z","shell.execute_reply":"2025-04-13T17:23:31.704818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, Input\nfrom tensorflow.keras.layers import LeakyReLU\nfrom tensorflow.keras.models import Sequential\nphysical_devices = tf.config.list_physical_devices('GPU')\ntf.config.set_visible_devices(physical_devices, 'GPU')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:23:31.706333Z","iopub.execute_input":"2025-04-13T17:23:31.706671Z","iopub.status.idle":"2025-04-13T17:23:31.710671Z","shell.execute_reply.started":"2025-04-13T17:23:31.706628Z","shell.execute_reply":"2025-04-13T17:23:31.709741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Input(shape=(224,224,3)))\nmodel.add(Conv2D(32, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(32, (3, 3), activation='sigmoid'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(32, (3, 3), activation=tf.keras.layers.LeakyReLU(alpha=0.27)))\nmodel.add(Flatten())\nmodel.add(Dense(24, kernel_initializer = 'normal', activation='softmax'))\n\nmodel.compile(\n    optimizer = 'adam',\n    loss = 'categorical_crossentropy',\n    metrics = ['accuracy',\"mean_squared_error\"]\n)\n\nhistory = model.fit(\n    train_image,\n    validation_data = val_image,\n    epochs = 100,\n    callbacks = [\n        tf.keras.callbacks.EarlyStopping(\n            monitor = 'val_loss',\n            patience = 3,\n            restore_best_weights = True\n        )\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T17:23:31.711467Z","iopub.execute_input":"2025-04-13T17:23:31.711804Z"}},"outputs":[],"execution_count":null}]}