{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-30T13:09:16.162819Z","iopub.execute_input":"2023-12-30T13:09:16.163929Z","iopub.status.idle":"2023-12-30T13:10:52.389941Z","shell.execute_reply.started":"2023-12-30T13:09:16.163887Z","shell.execute_reply":"2023-12-30T13:10:52.388842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Import libraries:\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport random\nimport shutil\nimport glob\nfrom sklearn.utils import shuffle\n\n# for image:\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.image as mpimg\n\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\n# for model:\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Activation, Dropout, BatchNormalization, LeakyReLU\nfrom tensorflow.keras.layers import Conv2D, AveragePooling2D, MaxPooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:10:52.391938Z","iopub.execute_input":"2023-12-30T13:10:52.392427Z","iopub.status.idle":"2023-12-30T13:11:06.849240Z","shell.execute_reply.started":"2023-12-30T13:10:52.392395Z","shell.execute_reply":"2023-12-30T13:11:06.848092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.__version__\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:06.850524Z","iopub.execute_input":"2023-12-30T13:11:06.851126Z","iopub.status.idle":"2023-12-30T13:11:06.859024Z","shell.execute_reply.started":"2023-12-30T13:11:06.851097Z","shell.execute_reply":"2023-12-30T13:11:06.857815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir('../input'))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:06.862601Z","iopub.execute_input":"2023-12-30T13:11:06.863141Z","iopub.status.idle":"2023-12-30T13:11:06.906524Z","shell.execute_reply.started":"2023-12-30T13:11:06.863099Z","shell.execute_reply":"2023-12-30T13:11:06.905623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set paths:\n\ntrain = '../input/happy-whale-and-dolphin/train_images'\ntest = '../input/happy-whale-and-dolphin/test_images'","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:06.907974Z","iopub.execute_input":"2023-12-30T13:11:06.908593Z","iopub.status.idle":"2023-12-30T13:11:06.921780Z","shell.execute_reply.started":"2023-12-30T13:11:06.908556Z","shell.execute_reply":"2023-12-30T13:11:06.920897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir('../input/happy-whale-and-dolphin'))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:06.923149Z","iopub.execute_input":"2023-12-30T13:11:06.924063Z","iopub.status.idle":"2023-12-30T13:11:06.937012Z","shell.execute_reply.started":"2023-12-30T13:11:06.924028Z","shell.execute_reply":"2023-12-30T13:11:06.936145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(len(os.listdir(train)))\nprint(len(os.listdir(test)))","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:06.938312Z","iopub.execute_input":"2023-12-30T13:11:06.938906Z","iopub.status.idle":"2023-12-30T13:11:06.984726Z","shell.execute_reply.started":"2023-12-30T13:11:06.938875Z","shell.execute_reply":"2023-12-30T13:11:06.983623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(train)[:5])\nprint(os.listdir(test)[:5])\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:06.987628Z","iopub.execute_input":"2023-12-30T13:11:06.988129Z","iopub.status.idle":"2023-12-30T13:11:07.028714Z","shell.execute_reply.started":"2023-12-30T13:11:06.988085Z","shell.execute_reply":"2023-12-30T13:11:07.027575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set image paths:\n\ntrain_jpg = tf.io.gfile.glob(train+'/*.jpg')\ntest_jpg = tf.io.gfile.glob(test+'/*.jpg')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:07.030113Z","iopub.execute_input":"2023-12-30T13:11:07.030451Z","iopub.status.idle":"2023-12-30T13:11:17.342233Z","shell.execute_reply.started":"2023-12-30T13:11:07.030421Z","shell.execute_reply":"2023-12-30T13:11:17.340992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View train dataset:\n\ntrain_data = pd.read_csv('../input/happy-whale-and-dolphin/train.csv', sep = ',')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.346780Z","iopub.execute_input":"2023-12-30T13:11:17.347178Z","iopub.status.idle":"2023-12-30T13:11:17.488154Z","shell.execute_reply.started":"2023-12-30T13:11:17.347144Z","shell.execute_reply":"2023-12-30T13:11:17.486854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.sample(5)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.489713Z","iopub.execute_input":"2023-12-30T13:11:17.490081Z","iopub.status.idle":"2023-12-30T13:11:17.507746Z","shell.execute_reply.started":"2023-12-30T13:11:17.490050Z","shell.execute_reply":"2023-12-30T13:11:17.506338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.509353Z","iopub.execute_input":"2023-12-30T13:11:17.509718Z","iopub.status.idle":"2023-12-30T13:11:17.519773Z","shell.execute_reply.started":"2023-12-30T13:11:17.509689Z","shell.execute_reply":"2023-12-30T13:11:17.518773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.521097Z","iopub.execute_input":"2023-12-30T13:11:17.521404Z","iopub.status.idle":"2023-12-30T13:11:17.563152Z","shell.execute_reply.started":"2023-12-30T13:11:17.521380Z","shell.execute_reply":"2023-12-30T13:11:17.562298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe(include = 'all')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.564883Z","iopub.execute_input":"2023-12-30T13:11:17.565646Z","iopub.status.idle":"2023-12-30T13:11:17.664194Z","shell.execute_reply.started":"2023-12-30T13:11:17.565601Z","shell.execute_reply":"2023-12-30T13:11:17.662937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.665499Z","iopub.execute_input":"2023-12-30T13:11:17.665858Z","iopub.status.idle":"2023-12-30T13:11:17.692013Z","shell.execute_reply.started":"2023-12-30T13:11:17.665830Z","shell.execute_reply":"2023-12-30T13:11:17.691064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.species.value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.693352Z","iopub.execute_input":"2023-12-30T13:11:17.694358Z","iopub.status.idle":"2023-12-30T13:11:17.714360Z","shell.execute_reply.started":"2023-12-30T13:11:17.694324Z","shell.execute_reply":"2023-12-30T13:11:17.713564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(train_data.individual_id.duplicated())\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.715962Z","iopub.execute_input":"2023-12-30T13:11:17.716587Z","iopub.status.idle":"2023-12-30T13:11:17.737182Z","shell.execute_reply.started":"2023-12-30T13:11:17.716551Z","shell.execute_reply":"2023-12-30T13:11:17.736322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef Load_Image(path):\n    image_path = tf.io.read_file(path)\n    image_path = tf.image.decode_image(image_path, channels = 3)\n    image_path = tf.image.convert_image_dtype(image_path, tf.float32)\n    return image_path","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.738697Z","iopub.execute_input":"2023-12-30T13:11:17.739313Z","iopub.status.idle":"2023-12-30T13:11:17.749613Z","shell.execute_reply.started":"2023-12-30T13:11:17.739281Z","shell.execute_reply":"2023-12-30T13:11:17.748598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Fix mis-spellings from species variable:\n\ntrain_data['species'] = train_data['species'].replace({'kiler_whale': 'killer_whale', \n                               'globis': 'pilot_whale', \n                               'beluga': 'beluga_whale',\n                               'bottlenose_dolpin': 'bottlenose_dolphin',\n                               'short_finned_pilot_whale': 'pilot_whale',\n                               'long_finned_pilot_whale': 'pilot_whale'})","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.751254Z","iopub.execute_input":"2023-12-30T13:11:17.751808Z","iopub.status.idle":"2023-12-30T13:11:17.792548Z","shell.execute_reply.started":"2023-12-30T13:11:17.751768Z","shell.execute_reply":"2023-12-30T13:11:17.791692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Declare dolphin and whale variables for further analysis:\n\ndolphin = ['bottlenose_dolphin','common_dolphin','dusky_dolphin', 'spinner_dolphin', 'spotted_dolphin', 'commersons_dolphin', \n           'white_sided_dolphin', 'rough_toothed_dolphin', 'pantropic_spotted_dolphin', 'frasiers_dolphin']\n\n\nwhale = ['melon_headed_whale', 'humpback_whale', 'false_killer_whale', 'belug_whale', 'minke_whale', 'fin_whale', 'blue_whale', 'gray_whale',\n         'southern_right_whale', 'killer_whale', 'pilot_whale', 'sei_whale', 'cuviers_beaked_whale', 'brydes_whale', 'pygmy_killer_whale']\n\n\n# Add to train dataset:\ntrain_data['family'] = 'dolphin'\n\nfor ele in range(len(train_data)):\n    if train_data.species[ele] in whale:\n        train_data.family[ele] = 'whale'\n        \n        \ntrain_data.sample(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:17.793982Z","iopub.execute_input":"2023-12-30T13:11:17.794580Z","iopub.status.idle":"2023-12-30T13:11:25.328676Z","shell.execute_reply.started":"2023-12-30T13:11:17.794544Z","shell.execute_reply":"2023-12-30T13:11:25.327569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set globals:\n\nrandom_state = 42\nbatch_size = 256\nepochs = 3\nseed = 42\ntarget_size = (64, 64)\ninput_shape = (64, 64, 3)","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:25.329896Z","iopub.execute_input":"2023-12-30T13:11:25.330244Z","iopub.status.idle":"2023-12-30T13:11:25.335902Z","shell.execute_reply.started":"2023-12-30T13:11:25.330215Z","shell.execute_reply":"2023-12-30T13:11:25.334803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = shuffle(train_data, random_state = random_state)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:25.337835Z","iopub.execute_input":"2023-12-30T13:11:25.338355Z","iopub.status.idle":"2023-12-30T13:11:25.364761Z","shell.execute_reply.started":"2023-12-30T13:11:25.338315Z","shell.execute_reply":"2023-12-30T13:11:25.363671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_norm = ImageDataGenerator(rescale = 1.0/255, validation_split = 0.20)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:25.366257Z","iopub.execute_input":"2023-12-30T13:11:25.367285Z","iopub.status.idle":"2023-12-30T13:11:25.372298Z","shell.execute_reply.started":"2023-12-30T13:11:25.367239Z","shell.execute_reply":"2023-12-30T13:11:25.371391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Eet up training batching:\n\ngen_train = data_norm.flow_from_dataframe(train_data,\n                                          directory = train,\n                                          x_col = 'image',\n                                          y_col = 'species',\n                                          subset = 'training',\n                                          batch_size = batch_size,\n                                          class_mode = 'categorical',\n                                          seed = seed,\n                                          target_size = target_size)","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:25.373752Z","iopub.execute_input":"2023-12-30T13:11:25.374078Z","iopub.status.idle":"2023-12-30T13:11:53.824246Z","shell.execute_reply.started":"2023-12-30T13:11:25.374050Z","shell.execute_reply":"2023-12-30T13:11:53.822923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up testing/validation batching:\n\ngen_valid = data_norm.flow_from_dataframe(train_data,\n                                          directory = train,\n                                          x_col = 'image',\n                                          y_col = 'species',\n                                          subset = 'validation',\n                                          batch_size = batch_size,\n                                          class_mode = 'categorical',\n                                          seed = seed,\n                                          target_size = target_size)","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:11:53.825904Z","iopub.execute_input":"2023-12-30T13:11:53.826255Z","iopub.status.idle":"2023-12-30T13:12:26.160610Z","shell.execute_reply.started":"2023-12-30T13:11:53.826226Z","shell.execute_reply":"2023-12-30T13:12:26.159588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and Train Simple Model:\n\nmod = Sequential()\n\n# set up base model (simple base):\nmod.add(Conv2D(filters = 32, kernel_size = (5, 5), strides = (1, 1), input_shape = input_shape, padding ='valid'))\nmod.add(BatchNormalization())\nmod.add(Activation(LeakyReLU()))\n\nmod.add(Conv2D(filters = 32, kernel_size = (5, 5), strides = (1, 1), input_shape = input_shape, padding ='valid'))\nmod.add(BatchNormalization())\nmod.add(Activation(LeakyReLU()))\nmod.add(MaxPooling2D(pool_size = (2, 2)))\nmod.add(Dropout(0.1))\n\nmod.add(Conv2D(filters = 64, kernel_size = (5, 5), strides = (1, 1), input_shape = input_shape, padding ='valid'))\nmod.add(Activation(LeakyReLU()))\nmod.add(BatchNormalization())\n\nmod.add(Conv2D(filters = 128, kernel_size = (5, 5), strides = (1, 1), input_shape = input_shape, padding ='valid'))\nmod.add(BatchNormalization())\nmod.add(Activation(LeakyReLU()))\nmod.add(AveragePooling2D(pool_size = (2, 2)))\n\nmod.add(Conv2D(filters = 128, kernel_size = (5, 5), strides = (1, 1), input_shape = input_shape, padding ='valid'))\nmod.add(BatchNormalization())\nmod.add(Activation(LeakyReLU()))\nmod.add(AveragePooling2D(pool_size = (2, 2)))\nmod.add(Dropout(0.1))\n\nmod.add(Flatten())\n\n# set dense with activation as softmax:\nmod.add(Dense(train_data.species.nunique(), activation = 'softmax'))\n\n# set optimizer with small rate:\nopt = Adam(learning_rate = 0.0001)\n\n#set up loss function:\nlosses = tf.keras.losses.CategoricalCrossentropy() \n\n# compile model:\nmod.compile(loss = 'categorical_crossentropy', metrics = ['accuracy'], optimizer = opt)\n\n# view model summary:\nmod.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:12:26.161727Z","iopub.execute_input":"2023-12-30T13:12:26.162037Z","iopub.status.idle":"2023-12-30T13:12:26.659999Z","shell.execute_reply.started":"2023-12-30T13:12:26.162010Z","shell.execute_reply":"2023-12-30T13:12:26.658596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train model:\n\nfit = mod.fit(gen_train, epochs = epochs, validation_data = gen_valid)\nfit","metadata":{"execution":{"iopub.status.busy":"2023-12-30T13:12:26.662128Z","iopub.execute_input":"2023-12-30T13:12:26.662642Z","iopub.status.idle":"2023-12-30T15:25:46.190225Z","shell.execute_reply.started":"2023-12-30T13:12:26.662598Z","shell.execute_reply":"2023-12-30T15:25:46.184966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"validation_accuracy = mod.evaluate(gen_valid)[1]\nprint(f\"Validation Accuracy: {validation_accuracy * 100:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2023-12-30T15:25:46.205015Z","iopub.execute_input":"2023-12-30T15:25:46.205978Z","iopub.status.idle":"2023-12-30T15:32:17.182963Z","shell.execute_reply.started":"2023-12-30T15:25:46.205929Z","shell.execute_reply":"2023-12-30T15:32:17.181778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on a new image (replace 'path_to_new_image.jpg' with the actual path)\nnew_image_path = '/kaggle/input'\nnew_image = Load_Image(new_image_path)\nnew_image = np.expand_dims(new_image, axis=0)  # Add an extra dimension for batch size\n\n# Preprocess the image\nnew_image = data_norm.rescale(new_image / 255.0)\n\n# Make predictions\npredictions = mod.predict(new_image)\n\n# Decode the predictions (assuming one-hot encoding)\npredicted_class = np.argmax(predictions)\n\n# Map the predicted class index to the corresponding label\nclass_labels = gen_train.class_indices\npredicted_label = [k for k, v in class_labels.items() if v == predicted_class][0]\n\nprint(f\"The image is predicted to belong to class: {predicted_label}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}