{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-18T18:58:23.139781Z","iopub.execute_input":"2024-07-18T18:58:23.140156Z","iopub.status.idle":"2024-07-18T18:58:23.16918Z","shell.execute_reply.started":"2024-07-18T18:58:23.140125Z","shell.execute_reply":"2024-07-18T18:58:23.167987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport gc\nimport sys\nimport math\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\nfrom tqdm.autonotebook import tqdm\nfrom random import shuffle\n\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport keras.backend as K\nfrom keras.models import Sequential\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.models import Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import SGD,Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.utils import Sequence\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\n#train_df=train_df.head(n=2000)\ntrain_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 128\n\nBATCH_SIZE= 128\n\nEpochs=3\nLearning_rate=0.001\n\nnum_folds=5\nSelected_fold=1 #1,2,3,4,5 ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder\n\ndef load_images(x,dataset=\"train_images\"):\n    img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+x, target_size=(IMAGE_SIZE, IMAGE_SIZE, 3))\n    x = image.img_to_array(img)\n    x = preprocess_input(x)\n    x /= 255\n    return x","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yyy, label_encoder = prepare_labels(train_df['individual_id'])\nyyy.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_g=train_df.groupby('individual_id').size()\ntrain_df_g = train_df_g.to_frame()\ntrain_df_g = train_df_g.rename(columns={train_df_g.columns[0]: 'count_cls'})\ntrain_df=pd.merge(train_df,train_df_g,on='individual_id',how='left')\ntrain_df.head(5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold,StratifiedKFold\nsfolder = StratifiedKFold(n_splits=num_folds,random_state=1,shuffle=True)\ntrain_df[\"Fold\"]=\"train\"\nX = train_df[['image']]\ny = train_df[['individual_id']]\n\nfold_no = 1\nfor train, valid in sfolder.split(X,y):\n    train,\n    if fold_no==Selected_fold:\n        train_df.loc[valid, \"Fold\"] = \"valid\"\n    fold_no += 1\n    \n\ntrain_df[\"Fold\"][train_df.count_cls < 3]=\"train\"\n\ny_t=yyy[train_df[train_df.Fold==\"train\"].index]\ny_v=yyy[train_df[train_df.Fold==\"valid\"].index]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=train_df[train_df.Fold==\"train\"]\ndf_valid=train_df[train_df.Fold==\"valid\"]\n\ndf_train.shape,y_t.shape,df_valid.shape,y_v.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of classes in Training dataset:\",len(df_train.groupby('individual_id').size()))\nprint(\"Number of classes in validation dataset:\",len(df_valid.groupby('individual_id').size()))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Dataset(Sequence):\n    def __init__(self,df,yyy=None,is_train=True,batch_size=BATCH_SIZE,shuffle=True):\n        self.idx = df[\"image\"].values\n        self.paths = df[\"image\"].values \n        self.y = yyy   #df[\"individual_id_int\"].values\n        self.is_train = is_train\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n    def __len__(self):\n        return math.ceil(len(self.idx)/self.batch_size)\n   \n    def __getitem__(self,ids):\n        id_path= self.paths[ids]\n        batch_paths = self.paths[ids * self.batch_size:(ids + 1) * self.batch_size]\n        \n        if self.y is not None:\n            batch_y = self.y[ids * self.batch_size: (ids + 1) * self.batch_size]\n        \n        if self.is_train:\n            list_x =  [load_images(x,dataset=\"train_images\") for x in batch_paths]\n            batch_X = np.stack(list_x)\n            return batch_X,batch_y\n        else:\n            list_x =  [load_images(x,dataset=\"test_images\") for x in batch_paths]\n            batch_X = np.stack(list_x)\n            return batch_X\n    \n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.idx, self.y))\n            shuffle(ids_y)\n            self.idx, self.y = list(zip(*ids_y))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = Dataset(df_train,y_t,is_train=True,batch_size=BATCH_SIZE)\nvalid_dataset = Dataset(df_valid,y_v,is_train=True,batch_size=BATCH_SIZE)\nfor i in range(1):\n    images, label = train_dataset[i]\n    print(\"Dimension of the images is:\", images.shape)\n    print(\"label=\",label.shape)\n    print(label)\n    plt.imshow(images[0,:,:,:])\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\nlabels = df_train[\"individual_id\"]\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                  np.unique(labels),\n                                                  labels)\nclass_weights_dict = dict(enumerate(class_weights))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_save = ModelCheckpoint('./last.h5', \n                             save_best_only = True, \n                             save_weights_only = False,\n                             monitor = 'val_loss', \n                             mode = 'min', verbose = 1)\nearly_stop = EarlyStopping(monitor = 'val_loss', min_delta = 0.0001, \n                           patience = 5, mode = 'min', verbose = 1,\n                           restore_best_weights = True)\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.5, \n                              patience = 3, min_delta = 0.0001, \n                              mode = 'min', verbose = 1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\ndef create_model():\n    efficientnet_layers = EfficientNetB0(weights='imagenet', \n                                         include_top=False, \n                                         input_shape = (IMAGE_SIZE, IMAGE_SIZE, 3),\n                                         pooling='avg')\n\n    model = Sequential()\n    model.add(efficientnet_layers)\n    model.add(Dense(yyy.shape[1], activation='softmax'))\n    model.compile(optimizer = Adam(lr = Learning_rate),\n                  loss = \"categorical_crossentropy\",\n                  metrics = [\"accuracy\"])\n\n    return model\n\nmodel = create_model()\nmodel.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_dataset,\n                    validation_data=valid_dataset,\n                    epochs=Epochs,\n                    verbose=1,\n                    #class_weight=class_weights_dict,\n                    callbacks = [model_save, early_stop, reduce_lr]\n                   )\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluación","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.plot(history.history['accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.plot(history.history['loss'])\nplt.title('Model loss')\nplt.ylabel('loss')\nplt.xlabel('Epoch')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dataset\ndel valid_dataset\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inferencia","metadata":{}},{"cell_type":"code","source":"test = os.listdir(\"../input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''\n#test_df=test_df.head(100)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = Dataset(test_df,is_train=False,batch_size=16)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1):\n    images = test_dataset[i]\n    print(\"Dimension of the images is:\", images.shape)\n    plt.imshow(images[0,:,:,:])\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_dataset, verbose=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, pred in enumerate(predictions):\n    p=pred.argsort()[-5:][::-1]\n    idx=-1\n    s=''\n    s1=''\n    s2=''\n    for x in p:\n        idx=idx+1\n        if pred[x]>0.5:\n            s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n        else:\n            s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n    s= s1 + ' new_individual' + s2\n    s = s.strip(' ')\n    test_df.loc[i, 'predictions'] = s","metadata":{},"execution_count":null,"outputs":[]}]}