{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf\nimport glob\nimport cv2\nimport matplotlib.pyplot as plt\nfrom skimage.transform import resize\nfrom sklearn.model_selection import train_test_split\nimport math\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:11:40.720322Z","iopub.execute_input":"2024-06-21T08:11:40.721140Z","iopub.status.idle":"2024-06-21T08:11:54.592021Z","shell.execute_reply.started":"2024-06-21T08:11:40.721104Z","shell.execute_reply":"2024-06-21T08:11:54.590928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## loading the data","metadata":{}},{"cell_type":"code","source":"# loading labels\ndata = pd.read_csv(\"/kaggle/input/happy-whale-and-dolphin/train.csv\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-21T08:11:54.593671Z","iopub.execute_input":"2024-06-21T08:11:54.594286Z","iopub.status.idle":"2024-06-21T08:11:54.700816Z","shell.execute_reply.started":"2024-06-21T08:11:54.594258Z","shell.execute_reply":"2024-06-21T08:11:54.699733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:11:54.702025Z","iopub.execute_input":"2024-06-21T08:11:54.702318Z","iopub.status.idle":"2024-06-21T08:11:54.725829Z","shell.execute_reply.started":"2024-06-21T08:11:54.702294Z","shell.execute_reply":"2024-06-21T08:11:54.724926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test for missing values\nnp.any(data.isnull().values.reshape(-1))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:11:54.727877Z","iopub.execute_input":"2024-06-21T08:11:54.728198Z","iopub.status.idle":"2024-06-21T08:11:54.744173Z","shell.execute_reply.started":"2024-06-21T08:11:54.728173Z","shell.execute_reply":"2024-06-21T08:11:54.743027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test image type (png/ jpeg...)\nnp.all(np.array([ path.split(\".\") for path in data['image'].values])[:, 1] == 'jpg')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:11:54.745367Z","iopub.execute_input":"2024-06-21T08:11:54.745713Z","iopub.status.idle":"2024-06-21T08:11:54.821709Z","shell.execute_reply.started":"2024-06-21T08:11:54.745661Z","shell.execute_reply":"2024-06-21T08:11:54.820611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = data['species'].values\nclasses, counts = np.unique(labels, return_counts=True)\nplt.barh(classes, counts)\nplt.title('Class distribution in training set')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:11:54.822910Z","iopub.execute_input":"2024-06-21T08:11:54.823251Z","iopub.status.idle":"2024-06-21T08:11:55.411593Z","shell.execute_reply.started":"2024-06-21T08:11:54.823217Z","shell.execute_reply":"2024-06-21T08:11:55.410619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH_PNG = r'/kaggle/input/happy-whale-and-dolphin/train_images/*'\nimage_paths = [path for path in glob.glob(DATA_PATH_PNG)]\n\n#visualizing the data\nfig, ax = plt.subplots(3,3,  figsize=(10,10))\nax = ax.reshape(-1)\n\nfor index, path in enumerate(image_paths[:9]):\n    img = cv2.imread(path)\n    ax[index].imshow(img[:, :, ::-1])\n    ax[index].set_axis_off()\n    ax[index].set_title( labels[index] + \"\\n\" + str(img.shape))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:11:55.413054Z","iopub.execute_input":"2024-06-21T08:11:55.413379Z","iopub.status.idle":"2024-06-21T08:12:03.302894Z","shell.execute_reply.started":"2024-06-21T08:11:55.413351Z","shell.execute_reply":"2024-06-21T08:12:03.301865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## version 2 loading using numpy arrays","metadata":{}},{"cell_type":"code","source":"#loading the images \nDATA_PATH_PNG = r'/kaggle/input/happy-whale-and-dolphin/train_images/*'\n\n#deciding on the size\nIMG_WIDTH = 128\nIMG_HEIGHT = 128\nIMG_CHANNEL = 3\n\n#get paths\nimage_paths = [path for path in glob.glob(DATA_PATH_PNG)][:1000]\n\nDATASET_SIZE = len(image_paths)\nNUM_OF_CLASSES = len(classes)\n\nprint(DATASET_SIZE, NUM_OF_CLASSES)\n\nX = np.zeros((len(image_paths), IMG_HEIGHT, IMG_WIDTH, IMG_CHANNEL), dtype=np.uint8)\n\n# resizing to correct shape\nfor n, img_path in tqdm(enumerate(image_paths)):\n    img = cv2.imread(img_path)[:,:,:IMG_CHANNEL]\n    img = resize(img, (IMG_HEIGHT, IMG_WIDTH), mode='constant', preserve_range=True)\n    X[n] = img\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:12:03.304724Z","iopub.execute_input":"2024-06-21T08:12:03.305106Z","iopub.status.idle":"2024-06-21T08:40:10.769945Z","shell.execute_reply.started":"2024-06-21T08:12:03.305073Z","shell.execute_reply":"2024-06-21T08:40:10.768950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalizing images\ndef normalize_images(img):\n    img = img.astype(\"float32\") / 255.0\n    return img\n\nX = normalize_images(X)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:10.771393Z","iopub.execute_input":"2024-06-21T08:40:10.771735Z","iopub.status.idle":"2024-06-21T08:40:10.852588Z","shell.execute_reply.started":"2024-06-21T08:40:10.771704Z","shell.execute_reply":"2024-06-21T08:40:10.851368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = data['species'].values\nmask = np.array(np.ones(len(classes)), dtype=bool)\nclasses[mask]","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:10.856963Z","iopub.execute_input":"2024-06-21T08:40:10.857408Z","iopub.status.idle":"2024-06-21T08:40:10.869485Z","shell.execute_reply.started":"2024-06-21T08:40:10.857367Z","shell.execute_reply":"2024-06-21T08:40:10.865793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoded_data = pd.get_dummies(y, columns = ['species']).values\nprint(one_hot_encoded_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:10.871958Z","iopub.execute_input":"2024-06-21T08:40:10.872415Z","iopub.status.idle":"2024-06-21T08:40:10.894702Z","shell.execute_reply.started":"2024-06-21T08:40:10.872379Z","shell.execute_reply":"2024-06-21T08:40:10.892056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train test split\nX_train, X_test, y_train, y_test = train_test_split(\n    X, one_hot_encoded_data[:1000], test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:10.896089Z","iopub.execute_input":"2024-06-21T08:40:10.896402Z","iopub.status.idle":"2024-06-21T08:40:10.984076Z","shell.execute_reply.started":"2024-06-21T08:40:10.896377Z","shell.execute_reply":"2024-06-21T08:40:10.983014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)\nprint(y_train.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:10.985436Z","iopub.execute_input":"2024-06-21T08:40:10.985858Z","iopub.status.idle":"2024-06-21T08:40:10.991925Z","shell.execute_reply.started":"2024-06-21T08:40:10.985824Z","shell.execute_reply":"2024-06-21T08:40:10.990958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#augmentation \nclass Augment(tf.keras.layers.Layer):\n  def __init__(self, seed):\n    super().__init__()\n    # both use the same seed, so they'll make the same random changes.\n    self.augment_inputs = tf.keras.layers.RandomFlip(mode=\"horizontal_and_vertical\", seed=seed)\n    self.rotate_inputs = tf.keras.layers.RandomRotation(0.2)\n    #self.crop_inputs = tf.keras.layers.RandomCrop(128, 128, seed=seed)\n\n  def call(self, inputs):\n    inputs = self.augment_inputs(inputs)\n    inputs = self.rotate_inputs(inputs)\n    #inputs = self.crop_inputs(inputs)\n    \n    return inputs\n\naugmenter = Augment(seed=42)\nX_train = augmenter(X_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:10.993081Z","iopub.execute_input":"2024-06-21T08:40:10.993392Z","iopub.status.idle":"2024-06-21T08:40:12.328933Z","shell.execute_reply.started":"2024-06-21T08:40:10.993367Z","shell.execute_reply":"2024-06-21T08:40:12.327744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualizing the data\nfig, ax = plt.subplots(2,3,  figsize=(10,10))\nax = ax.reshape(-1)\n\nfor index, img in enumerate(X_train[:6]):\n    ax[index].imshow(img[:, :, ::-1])\n    ax[index].set_axis_off()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:12.330233Z","iopub.execute_input":"2024-06-21T08:40:12.330567Z","iopub.status.idle":"2024-06-21T08:40:12.893502Z","shell.execute_reply.started":"2024-06-21T08:40:12.330539Z","shell.execute_reply":"2024-06-21T08:40:12.892517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#building model #\nINPUT_SHAPE = (IMG_WIDTH, IMG_HEIGHT, IMG_CHANNEL)\n\nmodel = tf.keras.Sequential()\nmodel.add(tf.keras.layers.Conv2D(filters=16, kernel_size=(3, 3), input_shape=INPUT_SHAPE, activation='relu', padding='same'))\nmodel.add(tf.keras.layers.BatchNormalization())\nmodel.add(tf.keras.layers.Conv2D(filters=16, kernel_size=(3, 3), input_shape=INPUT_SHAPE, activation='relu', padding='same'))\nmodel.add(tf.keras.layers.BatchNormalization())\n\nmodel.add(tf.keras.layers.MaxPool2D(pool_size=(2, 2)))\nmodel.add(tf.keras.layers.Dropout(0.25))\n\nmodel.add(tf.keras.layers.Conv2D(filters=32, kernel_size=(3, 3), input_shape=INPUT_SHAPE, activation='relu', padding='same'))\nmodel.add(tf.keras.layers.BatchNormalization())\nmodel.add(tf.keras.layers.Conv2D(filters=32, kernel_size=(3, 3), input_shape=INPUT_SHAPE, activation='relu', padding='same'))\nmodel.add(tf.keras.layers.BatchNormalization())\nmodel.add(tf.keras.layers.MaxPool2D(pool_size=(2, 2)))\n\nmodel.add(tf.keras.layers.Flatten())\nmodel.add(tf.keras.layers.Dense(32, activation='relu'))\nmodel.add(tf.keras.layers.Dense(NUM_OF_CLASSES, activation='softmax'))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:12.894881Z","iopub.execute_input":"2024-06-21T08:40:12.895269Z","iopub.status.idle":"2024-06-21T08:40:13.112382Z","shell.execute_reply.started":"2024-06-21T08:40:12.895235Z","shell.execute_reply":"2024-06-21T08:40:13.111446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LEARNING_RATE = 1e-3\n\n#compiling\nmodel.compile(loss='categorical_crossentropy', \n               optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE), \n               metrics=['accuracy']\n              )","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:13.113567Z","iopub.execute_input":"2024-06-21T08:40:13.113889Z","iopub.status.idle":"2024-06-21T08:40:13.128661Z","shell.execute_reply.started":"2024-06-21T08:40:13.113863Z","shell.execute_reply":"2024-06-21T08:40:13.127740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS=300\nBATCH_SIZE=32\n\n#fitting the model \nhistory = model.fit(x=X_train,\n                    y=y_train,\n                    epochs=EPOCHS,\n                    batch_size=BATCH_SIZE\n                   )","metadata":{"execution":{"iopub.status.busy":"2024-06-21T08:40:13.130063Z","iopub.execute_input":"2024-06-21T08:40:13.130374Z","iopub.status.idle":"2024-06-21T10:01:30.535472Z","shell.execute_reply.started":"2024-06-21T08:40:13.130350Z","shell.execute_reply":"2024-06-21T10:01:30.534464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#evaluating \nplt.figure(figsize=(12, 16))\n\nplt.subplot(4, 2, 1)\nplt.plot(history.history['loss'], label='Loss')\n#plt.plot(history.history['val_loss'], label='val_Loss')\nplt.title('Loss')\nplt.legend()\n\nplt.subplot(4, 2, 2)\nplt.plot(history.history['accuracy'], label='accuracy')\n#plt.plot(history.history['val_accuracy'], label='val_accuracy')\nplt.title('Accuracy')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T10:01:30.536850Z","iopub.execute_input":"2024-06-21T10:01:30.537139Z","iopub.status.idle":"2024-06-21T10:01:31.083978Z","shell.execute_reply.started":"2024-06-21T10:01:30.537111Z","shell.execute_reply":"2024-06-21T10:01:31.082958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nidx = random.randint(0, len(X_test))\nim = X_test[idx]\nplt.imshow(im)\n\ny_pred = model.predict(np.expand_dims(X_test[idx], axis=0))\nprediction_String = classes[y_pred.argmax()]\ncertainty = y_pred[:,y_pred.argmax()]\nreal = classes[y_test[idx]]\n\nprint(f\"our model predicts for an image of {real} that it is {prediction_String} with a certainty of {certainty}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T10:01:31.085679Z","iopub.execute_input":"2024-06-21T10:01:31.086141Z","iopub.status.idle":"2024-06-21T10:01:31.680368Z","shell.execute_reply.started":"2024-06-21T10:01:31.086102Z","shell.execute_reply":"2024-06-21T10:01:31.679283Z"},"trusted":true},"execution_count":null,"outputs":[]}]}