{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# CNN with Keras Stater","metadata":{}},{"cell_type":"markdown","source":"### Please if this kernel is useful, <font color='red'>please upvote !!</font>","metadata":{}},{"cell_type":"markdown","source":"This kernel is based on: [CNN with Keras for Humpback Whale ID](https://www.kaggle.com/anezka/cnn-with-keras-for-humpback-whale-id)","metadata":{}},{"cell_type":"markdown","source":"### Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport gc\nimport sys\nimport math\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\nfrom tqdm.autonotebook import tqdm\nfrom random import shuffle\n\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport keras.backend as K\nfrom keras.models import Sequential\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.models import Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import SGD,Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.utils import Sequence\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)","metadata":{"_uuid":"0d9c73ad23e6c2eae3028255ee00c3254fe66401","execution":{"iopub.status.busy":"2022-02-25T10:56:00.943087Z","iopub.execute_input":"2022-02-25T10:56:00.943650Z","iopub.status.idle":"2022-02-25T10:56:07.845998Z","shell.execute_reply.started":"2022-02-25T10:56:00.943525Z","shell.execute_reply":"2022-02-25T10:56:07.845197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading Data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\n#train_df=train_df.head(n=2000)\ntrain_df.head()","metadata":{"_uuid":"46a8839e13a14eb8d16ea6823de9927ea63d5001","execution":{"iopub.status.busy":"2022-02-25T10:56:07.847555Z","iopub.execute_input":"2022-02-25T10:56:07.847765Z","iopub.status.idle":"2022-02-25T10:56:07.960247Z","shell.execute_reply.started":"2022-02-25T10:56:07.847739Z","shell.execute_reply":"2022-02-25T10:56:07.959317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Confg","metadata":{}},{"cell_type":"code","source":"IMAGE_SIZE = 128\n\nBATCH_SIZE= 128\n\nEpochs=3\nLearning_rate=0.001\n\nnum_folds=5\nSelected_fold=1 #1,2,3,4,5 ","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:07.961650Z","iopub.execute_input":"2022-02-25T10:56:07.962328Z","iopub.status.idle":"2022-02-25T10:56:07.966805Z","shell.execute_reply.started":"2022-02-25T10:56:07.962294Z","shell.execute_reply":"2022-02-25T10:56:07.965998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Functions","metadata":{}},{"cell_type":"code","source":"def prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder\n\ndef load_images(x,dataset=\"train_images\"):\n    img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+x, target_size=(IMAGE_SIZE, IMAGE_SIZE, 3))\n    x = image.img_to_array(img)\n    x = preprocess_input(x)\n    x /= 255\n    return x","metadata":{"_uuid":"6587a101b58af064af0f9c60a1070c6c8f52d45f","execution":{"iopub.status.busy":"2022-02-25T10:56:07.969253Z","iopub.execute_input":"2022-02-25T10:56:07.970332Z","iopub.status.idle":"2022-02-25T10:56:07.978736Z","shell.execute_reply.started":"2022-02-25T10:56:07.970283Z","shell.execute_reply":"2022-02-25T10:56:07.978036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yyy, label_encoder = prepare_labels(train_df['individual_id'])\nyyy.shape","metadata":{"_uuid":"675924f8863aef27cf90dc668e0a68cd609dfc1c","execution":{"iopub.status.busy":"2022-02-25T10:56:07.980531Z","iopub.execute_input":"2022-02-25T10:56:07.981033Z","iopub.status.idle":"2022-02-25T10:56:08.309327Z","shell.execute_reply.started":"2022-02-25T10:56:07.980979Z","shell.execute_reply":"2022-02-25T10:56:08.308426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_g=train_df.groupby('individual_id').size()\ntrain_df_g = train_df_g.to_frame()\ntrain_df_g = train_df_g.rename(columns={train_df_g.columns[0]: 'count_cls'})\ntrain_df=pd.merge(train_df,train_df_g,on='individual_id',how='left')\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:08.310512Z","iopub.execute_input":"2022-02-25T10:56:08.310732Z","iopub.status.idle":"2022-02-25T10:56:08.382584Z","shell.execute_reply.started":"2022-02-25T10:56:08.310706Z","shell.execute_reply":"2022-02-25T10:56:08.381580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### n Fold","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold,StratifiedKFold\nsfolder = StratifiedKFold(n_splits=num_folds,random_state=1,shuffle=True)\ntrain_df[\"Fold\"]=\"train\"\nX = train_df[['image']]\ny = train_df[['individual_id']]\n\nfold_no = 1\nfor train, valid in sfolder.split(X,y):\n    train,\n    if fold_no==Selected_fold:\n        train_df.loc[valid, \"Fold\"] = \"valid\"\n    fold_no += 1\n    \n\ntrain_df[\"Fold\"][train_df.count_cls < 3]=\"train\"\n\ny_t=yyy[train_df[train_df.Fold==\"train\"].index]\ny_v=yyy[train_df[train_df.Fold==\"valid\"].index]","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:08.383682Z","iopub.execute_input":"2022-02-25T10:56:08.383887Z","iopub.status.idle":"2022-02-25T10:56:15.893499Z","shell.execute_reply.started":"2022-02-25T10:56:08.383861Z","shell.execute_reply":"2022-02-25T10:56:15.892516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:15.896152Z","iopub.execute_input":"2022-02-25T10:56:15.896712Z","iopub.status.idle":"2022-02-25T10:56:16.082544Z","shell.execute_reply.started":"2022-02-25T10:56:15.896656Z","shell.execute_reply":"2022-02-25T10:56:16.081618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=train_df[train_df.Fold==\"train\"]\ndf_valid=train_df[train_df.Fold==\"valid\"]\n\ndf_train.shape,y_t.shape,df_valid.shape,y_v.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:16.083852Z","iopub.execute_input":"2022-02-25T10:56:16.084165Z","iopub.status.idle":"2022-02-25T10:56:16.117479Z","shell.execute_reply.started":"2022-02-25T10:56:16.084132Z","shell.execute_reply":"2022-02-25T10:56:16.116524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of classes in Training dataset:\",len(df_train.groupby('individual_id').size()))\nprint(\"Number of classes in validation dataset:\",len(df_valid.groupby('individual_id').size()))","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:16.122050Z","iopub.execute_input":"2022-02-25T10:56:16.122286Z","iopub.status.idle":"2022-02-25T10:56:16.164066Z","shell.execute_reply.started":"2022-02-25T10:56:16.122259Z","shell.execute_reply":"2022-02-25T10:56:16.163184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:16.165321Z","iopub.execute_input":"2022-02-25T10:56:16.165594Z","iopub.status.idle":"2022-02-25T10:56:16.171207Z","shell.execute_reply.started":"2022-02-25T10:56:16.165561Z","shell.execute_reply":"2022-02-25T10:56:16.170250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset","metadata":{}},{"cell_type":"code","source":"class Dataset(Sequence):\n    def __init__(self,df,yyy=None,is_train=True,batch_size=BATCH_SIZE,shuffle=True):\n        self.idx = df[\"image\"].values\n        self.paths = df[\"image\"].values \n        self.y = yyy   #df[\"individual_id_int\"].values\n        self.is_train = is_train\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n    def __len__(self):\n        return math.ceil(len(self.idx)/self.batch_size)\n   \n    def __getitem__(self,ids):\n        id_path= self.paths[ids]\n        batch_paths = self.paths[ids * self.batch_size:(ids + 1) * self.batch_size]\n        \n        if self.y is not None:\n            batch_y = self.y[ids * self.batch_size: (ids + 1) * self.batch_size]\n        \n        if self.is_train:\n            list_x =  [load_images(x,dataset=\"train_images\") for x in batch_paths]\n            batch_X = np.stack(list_x)\n            return batch_X,batch_y\n        else:\n            list_x =  [load_images(x,dataset=\"test_images\") for x in batch_paths]\n            batch_X = np.stack(list_x)\n            return batch_X\n    \n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.idx, self.y))\n            shuffle(ids_y)\n            self.idx, self.y = list(zip(*ids_y))","metadata":{"_uuid":"14d243b19023e830b636bea16679e13bc40deae6","execution":{"iopub.status.busy":"2022-02-25T10:56:16.172918Z","iopub.execute_input":"2022-02-25T10:56:16.173252Z","iopub.status.idle":"2022-02-25T10:56:16.188009Z","shell.execute_reply.started":"2022-02-25T10:56:16.173209Z","shell.execute_reply":"2022-02-25T10:56:16.186889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = Dataset(df_train,y_t,is_train=True,batch_size=BATCH_SIZE)\nvalid_dataset = Dataset(df_valid,y_v,is_train=True,batch_size=BATCH_SIZE)\nfor i in range(1):\n    images, label = train_dataset[i]\n    print(\"Dimension of the images is:\", images.shape)\n    print(\"label=\",label.shape)\n    print(label)\n    plt.imshow(images[0,:,:,:])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:16.189741Z","iopub.execute_input":"2022-02-25T10:56:16.190277Z","iopub.status.idle":"2022-02-25T10:56:28.010094Z","shell.execute_reply.started":"2022-02-25T10:56:16.190232Z","shell.execute_reply":"2022-02-25T10:56:28.009241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Class Weights","metadata":{}},{"cell_type":"code","source":"le = LabelEncoder()\nlabels = df_train[\"individual_id\"]\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                  np.unique(labels),\n                                                  labels)\nclass_weights_dict = dict(enumerate(class_weights))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-02-25T10:56:28.011695Z","iopub.execute_input":"2022-02-25T10:56:28.012196Z","iopub.status.idle":"2022-02-25T10:56:40.306363Z","shell.execute_reply.started":"2022-02-25T10:56:28.012149Z","shell.execute_reply":"2022-02-25T10:56:40.305188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model","metadata":{}},{"cell_type":"code","source":"model_save = ModelCheckpoint('./last.h5', \n                             save_best_only = True, \n                             save_weights_only = False,\n                             monitor = 'val_loss', \n                             mode = 'min', verbose = 1)\nearly_stop = EarlyStopping(monitor = 'val_loss', min_delta = 0.0001, \n                           patience = 5, mode = 'min', verbose = 1,\n                           restore_best_weights = True)\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.5, \n                              patience = 3, min_delta = 0.0001, \n                              mode = 'min', verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:40.308949Z","iopub.execute_input":"2022-02-25T10:56:40.309393Z","iopub.status.idle":"2022-02-25T10:56:40.316309Z","shell.execute_reply.started":"2022-02-25T10:56:40.309329Z","shell.execute_reply":"2022-02-25T10:56:40.315244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\ndef create_model():\n    efficientnet_layers = EfficientNetB0(weights='imagenet', \n                                         include_top=False, \n                                         input_shape = (IMAGE_SIZE, IMAGE_SIZE, 3),\n                                         pooling='avg')\n\n    model = Sequential()\n    model.add(efficientnet_layers)\n    model.add(Dense(yyy.shape[1], activation='softmax'))\n    model.compile(optimizer = Adam(lr = Learning_rate),\n                  loss = \"categorical_crossentropy\",\n                  metrics = [\"accuracy\"])\n\n    return model\n\nmodel = create_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T10:56:40.317765Z","iopub.execute_input":"2022-02-25T10:56:40.318088Z","iopub.status.idle":"2022-02-25T10:56:43.721725Z","shell.execute_reply.started":"2022-02-25T10:56:40.318036Z","shell.execute_reply":"2022-02-25T10:56:43.720864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(train_dataset,\n                    validation_data=valid_dataset,\n                    epochs=Epochs,\n                    verbose=1,\n                    #class_weight=class_weights_dict,\n                    callbacks = [model_save, early_stop, reduce_lr]\n                   )\n","metadata":{"_uuid":"169f45e150c3a584e0f655a8eda523e0675da63a","execution":{"iopub.status.busy":"2022-02-25T10:56:43.722946Z","iopub.execute_input":"2022-02-25T10:56:43.723486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.plot(history.history['accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"_uuid":"7bca48a1d0963cbf70685b75431435cef9499895","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.plot(history.history['loss'])\nplt.title('Model loss')\nplt.ylabel('loss')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dataset\ndel valid_dataset\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## inference","metadata":{}},{"cell_type":"code","source":"test = os.listdir(\"../input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{"_uuid":"debe961c93b72bef151d9aad3ca2cb500ee00aaa","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''\n#test_df=test_df.head(100)","metadata":{"_uuid":"72ed8198f519f7b1ae3efbc688933c78d8cdd0e4","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = Dataset(test_df,is_train=False,batch_size=16)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1):\n    images = test_dataset[i]\n    print(\"Dimension of the images is:\", images.shape)\n    plt.imshow(images[0,:,:,:])\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_dataset, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, pred in enumerate(predictions):\n    p=pred.argsort()[-5:][::-1]\n    idx=-1\n    s=''\n    s1=''\n    s2=''\n    for x in p:\n        idx=idx+1\n        if pred[x]>0.5:\n            s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n        else:\n            s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n    s= s1 + ' new_individual' + s2\n    s = s.strip(' ')\n    test_df.loc[i, 'predictions'] = s","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('submission.csv',index=False)\ntest_df.head(30)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}