{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"}],"dockerImageVersionId":30301,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport tensorflow as tf\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split\nfrom pathlib import Path\nimport os.path\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:12.660472Z","iopub.execute_input":"2024-06-26T21:23:12.660841Z","iopub.status.idle":"2024-06-26T21:23:18.837955Z","shell.execute_reply.started":"2024-06-26T21:23:12.660758Z","shell.execute_reply":"2024-06-26T21:23:18.836861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:18.840533Z","iopub.execute_input":"2024-06-26T21:23:18.841589Z","iopub.status.idle":"2024-06-26T21:23:18.967190Z","shell.execute_reply.started":"2024-06-26T21:23:18.841535Z","shell.execute_reply":"2024-06-26T21:23:18.966142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = '/kaggle/input/happy-whale-and-dolphin/train_images'","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:18.968323Z","iopub.execute_input":"2024-06-26T21:23:18.968612Z","iopub.status.idle":"2024-06-26T21:23:18.973679Z","shell.execute_reply.started":"2024-06-26T21:23:18.968585Z","shell.execute_reply":"2024-06-26T21:23:18.972640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['image']  = train_csv['image'].apply(lambda x : train + '/'+ x)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:18.974964Z","iopub.execute_input":"2024-06-26T21:23:18.975402Z","iopub.status.idle":"2024-06-26T21:23:19.013633Z","shell.execute_reply.started":"2024-06-26T21:23:18.975359Z","shell.execute_reply":"2024-06-26T21:23:19.012835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.017060Z","iopub.execute_input":"2024-06-26T21:23:19.017674Z","iopub.status.idle":"2024-06-26T21:23:19.043060Z","shell.execute_reply.started":"2024-06-26T21:23:19.017643Z","shell.execute_reply":"2024-06-26T21:23:19.042069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.044266Z","iopub.execute_input":"2024-06-26T21:23:19.044555Z","iopub.status.idle":"2024-06-26T21:23:19.086258Z","shell.execute_reply.started":"2024-06-26T21:23:19.044527Z","shell.execute_reply":"2024-06-26T21:23:19.085107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['species'] = train_csv['species'].replace({\n    'false_killer_whale' : 'killer_whale',\n    'bottlenose_dolpin' : 'bottlenose_dolphin',\n    'kiler_whale' : 'killer_whale',\n    'short_finned_pilot_whale' : 'pilot_whale',\n    'long_finned_pilot_whale' :  'pilot_whale',\n    'pygmy_killer_whale' : 'killer_whale'\n    \n})","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.087692Z","iopub.execute_input":"2024-06-26T21:23:19.087999Z","iopub.status.idle":"2024-06-26T21:23:19.117926Z","shell.execute_reply.started":"2024-06-26T21:23:19.087970Z","shell.execute_reply":"2024-06-26T21:23:19.117122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.119035Z","iopub.execute_input":"2024-06-26T21:23:19.119348Z","iopub.status.idle":"2024-06-26T21:23:19.135404Z","shell.execute_reply.started":"2024-06-26T21:23:19.119319Z","shell.execute_reply":"2024-06-26T21:23:19.134366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat = train_csv['species'].value_counts().loc[lambda x : x < 1000].index.tolist()\nfor i in cat:\n    drop = train_csv.loc[train_csv['species'] == i, 'species'].index.values.tolist()\n    train_csv = train_csv.drop(drop, axis = 0)\ntrain_csv.reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.136765Z","iopub.execute_input":"2024-06-26T21:23:19.137144Z","iopub.status.idle":"2024-06-26T21:23:19.314860Z","shell.execute_reply.started":"2024-06-26T21:23:19.137113Z","shell.execute_reply":"2024-06-26T21:23:19.313791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.317029Z","iopub.execute_input":"2024-06-26T21:23:19.317447Z","iopub.status.idle":"2024-06-26T21:23:19.332341Z","shell.execute_reply.started":"2024-06-26T21:23:19.317412Z","shell.execute_reply":"2024-06-26T21:23:19.331148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extracting only 300 samples from each category","metadata":{}},{"cell_type":"code","source":"samples = []\nfor i in train_csv['species'].unique():\n    x = train_csv.query('species == @i')\n    samples.append(x.sample(300, random_state = 1))\ntrain_csv = pd.concat(samples, axis = 0).sample(frac = 1.0, random_state = 1).reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.333655Z","iopub.execute_input":"2024-06-26T21:23:19.334321Z","iopub.status.idle":"2024-06-26T21:23:19.410106Z","shell.execute_reply.started":"2024-06-26T21:23:19.334285Z","shell.execute_reply":"2024-06-26T21:23:19.409232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.411470Z","iopub.execute_input":"2024-06-26T21:23:19.411811Z","iopub.status.idle":"2024-06-26T21:23:19.420678Z","shell.execute_reply.started":"2024-06-26T21:23:19.411781Z","shell.execute_reply":"2024-06-26T21:23:19.419579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndef adjust_brightness_contrast(image, alpha=1.5, beta=50):\n    return cv2.convertScaleAbs(image, alpha=alpha, beta=beta)\n\ndef equalize_histogram(image):\n    if len(image.shape) == 2:  # Grayscale image\n        return cv2.equalizeHist(image)\n    elif len(image.shape) == 3:  # Color image\n        ycrcb = cv2.cvtColor(image, cv2.COLOR_RGB2YCrCb)\n        y, cr, cb = cv2.split(ycrcb)\n        y_eq = cv2.equalizeHist(y)\n        ycrcb_eq = cv2.merge([y_eq, cr, cb])\n        return cv2.cvtColor(ycrcb_eq, cv2.COLOR_YCrCb2RGB)\n    return image\n\ndef sharpen_image(image):\n    sharpening_kernel = np.array([[-1, -1, -1],\n                                  [-1,  9, -1],\n                                  [-1, -1, -1]])\n    return cv2.filter2D(image, -1, sharpening_kernel)\n\ndef remove_noise(image):\n    return cv2.GaussianBlur(image, (5, 5), 0)\n\ndef custom_preprocessing(image):\n    # Convert image from range [0, 1] to [0, 255]\n    image = image * 255.0\n    image = image.astype(np.uint8)\n    \n    # Apply preprocessing steps\n    image = remove_noise(image)\n    image = adjust_brightness_contrast(image)\n    image = equalize_histogram(image)\n    image = sharpen_image(image)\n    image = tf.keras.applications.mobilenet_v2.preprocess_input(image)\n    \n    # Convert image back to range [0, 1]\n    image = image.astype(np.float32) / 255.0\n    \n    return image","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:19.422640Z","iopub.execute_input":"2024-06-26T21:23:19.423347Z","iopub.status.idle":"2024-06-26T21:23:20.525086Z","shell.execute_reply.started":"2024-06-26T21:23:19.423315Z","shell.execute_reply":"2024-06-26T21:23:20.524215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Test Spliiting","metadata":{}},{"cell_type":"code","source":"train_df , test_df = train_test_split(train_csv, test_size = 0.30, shuffle = True, random_state = 1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:20.528951Z","iopub.execute_input":"2024-06-26T21:23:20.529351Z","iopub.status.idle":"2024-06-26T21:23:20.551243Z","shell.execute_reply.started":"2024-06-26T21:23:20.529318Z","shell.execute_reply":"2024-06-26T21:23:20.550191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Images","metadata":{}},{"cell_type":"code","source":"train_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = custom_preprocessing,validation_split = 0.2)\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = custom_preprocessing)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:20.552485Z","iopub.execute_input":"2024-06-26T21:23:20.552864Z","iopub.status.idle":"2024-06-26T21:23:20.559080Z","shell.execute_reply.started":"2024-06-26T21:23:20.552820Z","shell.execute_reply":"2024-06-26T21:23:20.558056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (224, 224),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'training'\n)\nval_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (224, 224),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'validation'\n)\ntest_image = test_gen.flow_from_dataframe(\n    dataframe = test_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (224, 224),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:20.560486Z","iopub.execute_input":"2024-06-26T21:23:20.560803Z","iopub.status.idle":"2024-06-26T21:23:24.783731Z","shell.execute_reply.started":"2024-06-26T21:23:20.560774Z","shell.execute_reply":"2024-06-26T21:23:24.782693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"pre_model = tf.keras.applications.MobileNetV2(\n    input_shape = (224, 224, 3),\n    include_top = False,\n    pooling = 'avg'\n)\npre_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:24.785078Z","iopub.execute_input":"2024-06-26T21:23:24.785426Z","iopub.status.idle":"2024-06-26T21:23:29.048314Z","shell.execute_reply.started":"2024-06-26T21:23:24.785397Z","shell.execute_reply":"2024-06-26T21:23:29.047418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = pre_model.input\nx = tf.keras.layers.Dense(128, activation = 'relu')(pre_model.output)\nx = tf.keras.layers.Dense(128, activation = 'relu')(x)\noutputs = tf.keras.layers.Dense(11,activation = 'softmax')(x)# 8 for 8 Classes\nmodel = tf.keras.Model(inputs = inputs , outputs  = outputs)\nmodel.compile(\n    optimizer = 'adam',\n    loss = 'categorical_crossentropy',\n    metrics = ['accuracy',\"mean_squared_error\"]\n)\nhistory = model.fit(\n    train_image,\n    validation_data = val_image,\n    epochs = 100,\n    callbacks = [\n        tf.keras.callbacks.EarlyStopping(\n            monitor = 'val_loss',\n            patience = 3,\n            restore_best_weights = True\n        )\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T21:23:29.049607Z","iopub.execute_input":"2024-06-26T21:23:29.049951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Results","metadata":{}},{"cell_type":"code","source":"results = model.evaluate(test_image, verbose = 0)\npred = np.argmax(model.predict(test_image), axis = 1)\nclass_names = list(test_image.class_indices.keys())\ncm = confusion_matrix(test_image.labels, pred, labels = np.arange(11))\nclr = classification_report(test_image.labels, pred, labels = np.arange(11),target_names = class_names)\nprint(f'\\nTest Accuracy : {round(results[1], 4)*100}%\\n')\nplt.figure(figsize = (10,10))\nsns.heatmap(cm, annot = True, fmt = 'g', vmin = 0, cbar = False)\nplt.xticks(ticks = np.arange(11) + 0.5, labels = class_names, rotation = 90)\nplt.yticks(ticks = np.arange(11) + 0.5, labels = class_names, rotation = 0)\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()\nprint(f'classification Report------------>\\n{clr}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\n    'test_labels' : test_image.labels,\n    'predicted_labels' : pred\n})\nsubmission_df.head(20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}