{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-04T07:59:21.402478Z","iopub.execute_input":"2022-05-04T07:59:21.402868Z","iopub.status.idle":"2022-05-04T08:00:20.910175Z","shell.execute_reply.started":"2022-05-04T07:59:21.402772Z","shell.execute_reply":"2022-05-04T08:00:20.909150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Happywhale - Whale and Dolphin Identification**","metadata":{}},{"cell_type":"markdown","source":"###  **Importing Library**","metadata":{}},{"cell_type":"code","source":"# These library are for data manipulation \nimport numpy as np\nimport pandas as pd\n\n# These library are for working with directories\nimport os\nfrom glob import glob\nfrom tqdm import tqdm\n\n# These library are for Visualization\nimport matplotlib.pyplot as plt\nimport plotly.express as px\n\n# These Library are for converting Label Encoding\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\n# These library are for building model \nfrom tensorflow.keras import layers\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\nfrom tensorflow.keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:20.912929Z","iopub.execute_input":"2022-05-04T08:00:20.913356Z","iopub.status.idle":"2022-05-04T08:00:28.764372Z","shell.execute_reply.started":"2022-05-04T08:00:20.913312Z","shell.execute_reply":"2022-05-04T08:00:28.763511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the Files present in the main Directories\n\npath = '/kaggle/input/happy-whale-and-dolphin/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:28.765836Z","iopub.execute_input":"2022-05-04T08:00:28.766150Z","iopub.status.idle":"2022-05-04T08:00:28.778985Z","shell.execute_reply.started":"2022-05-04T08:00:28.766112Z","shell.execute_reply":"2022-05-04T08:00:28.778074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the Train csv file and Sample Submission File using main dir\n\ntrain_data = pd.read_csv(path+'train.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:28.781677Z","iopub.execute_input":"2022-05-04T08:00:28.782075Z","iopub.status.idle":"2022-05-04T08:00:28.957208Z","shell.execute_reply.started":"2022-05-04T08:00:28.782030Z","shell.execute_reply":"2022-05-04T08:00:28.956277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the dimension of the train.csv file\n\nprint('Number train samples:', len(train_data))","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:28.958709Z","iopub.execute_input":"2022-05-04T08:00:28.958924Z","iopub.status.idle":"2022-05-04T08:00:28.964267Z","shell.execute_reply.started":"2022-05-04T08:00:28.958899Z","shell.execute_reply":"2022-05-04T08:00:28.963273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the column name present in the train.csv file\n\ntrain_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:28.965770Z","iopub.execute_input":"2022-05-04T08:00:28.966031Z","iopub.status.idle":"2022-05-04T08:00:28.975074Z","shell.execute_reply.started":"2022-05-04T08:00:28.966003Z","shell.execute_reply":"2022-05-04T08:00:28.974418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the first five in rows in the train.csv file\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:28.976378Z","iopub.execute_input":"2022-05-04T08:00:28.977010Z","iopub.status.idle":"2022-05-04T08:00:28.996335Z","shell.execute_reply.started":"2022-05-04T08:00:28.976964Z","shell.execute_reply":"2022-05-04T08:00:28.995554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the Number of training images available in the train image directory\n\nprint('Number train images:', len(os.listdir(path+'train_images/')))","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:28.997921Z","iopub.execute_input":"2022-05-04T08:00:28.998478Z","iopub.status.idle":"2022-05-04T08:00:29.027107Z","shell.execute_reply.started":"2022-05-04T08:00:28.998441Z","shell.execute_reply":"2022-05-04T08:00:29.026243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the Number of testing images available in the test image directory\n\nprint('Number test images:', len(os.listdir(path+'test_images/')))","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.028425Z","iopub.execute_input":"2022-05-04T08:00:29.028683Z","iopub.status.idle":"2022-05-04T08:00:29.046738Z","shell.execute_reply.started":"2022-05-04T08:00:29.028654Z","shell.execute_reply":"2022-05-04T08:00:29.045895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dispalying the different species availabel in the dataset\n\ncount_species=train_data['species'].value_counts()\ncount_species","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.050480Z","iopub.execute_input":"2022-05-04T08:00:29.050728Z","iopub.status.idle":"2022-05-04T08:00:29.071267Z","shell.execute_reply.started":"2022-05-04T08:00:29.050700Z","shell.execute_reply":"2022-05-04T08:00:29.070520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total number of species: {len(count_species)}\")","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.072814Z","iopub.execute_input":"2022-05-04T08:00:29.073108Z","iopub.status.idle":"2022-05-04T08:00:29.077960Z","shell.execute_reply.started":"2022-05-04T08:00:29.073069Z","shell.execute_reply":"2022-05-04T08:00:29.077205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting BarChart of the Species avialable\n\nplt.figure(figsize=(15, 12))\nplt.rcParams[\"font.size\"] = 18\nplt.barh(train_data[\"species\"].value_counts().sort_values(ascending=True).index,\n         train_data[\"species\"].value_counts().sort_values(ascending=True),\n         tick_label = train_data[\"species\"].value_counts().sort_values(ascending=True).index)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.079246Z","iopub.execute_input":"2022-05-04T08:00:29.079453Z","iopub.status.idle":"2022-05-04T08:00:29.510949Z","shell.execute_reply.started":"2022-05-04T08:00:29.079426Z","shell.execute_reply":"2022-05-04T08:00:29.510110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the individual id present in the dataset\n\ntrain_data['individual_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.512332Z","iopub.execute_input":"2022-05-04T08:00:29.512711Z","iopub.status.idle":"2022-05-04T08:00:29.536995Z","shell.execute_reply.started":"2022-05-04T08:00:29.512670Z","shell.execute_reply":"2022-05-04T08:00:29.536000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we will prepare our data for training and also plot few images \n\nBASE_PATH = \"../input/happy-whale-and-dolphin/train_images/\"\nTRAIN_IMAGES = glob(BASE_PATH + \"train/*.jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.538565Z","iopub.execute_input":"2022-05-04T08:00:29.538866Z","iopub.status.idle":"2022-05-04T08:00:29.543356Z","shell.execute_reply.started":"2022-05-04T08:00:29.538826Z","shell.execute_reply":"2022-05-04T08:00:29.542811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying Images Randomly for training dataset\n\npath = BASE_PATH + np.random.choice(train_data['image'])\nim = plt.imread(path)\nplt.figure(figsize=(15, 6))\nplt.imshow(im)\nplt.title(path.split(\"/\")[-1])\nplt.xticks([]), plt.yticks([])\ntrain_data[train_data['image']==path.split('/')[-1]]","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.544447Z","iopub.execute_input":"2022-05-04T08:00:29.544705Z","iopub.status.idle":"2022-05-04T08:00:29.866463Z","shell.execute_reply.started":"2022-05-04T08:00:29.544676Z","shell.execute_reply":"2022-05-04T08:00:29.865695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Storing the Base path and then creating test images for further use\n\nBASE_PATH = \"../input/happy-whale-and-dolphin/test_images/\"\nTEST_IMAGES = glob(BASE_PATH + \"*.jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.867901Z","iopub.execute_input":"2022-05-04T08:00:29.868227Z","iopub.status.idle":"2022-05-04T08:00:29.977119Z","shell.execute_reply.started":"2022-05-04T08:00:29.868198Z","shell.execute_reply":"2022-05-04T08:00:29.976525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying images randomly using test data\n\npath = np.random.choice(TEST_IMAGES)\nim = plt.imread(path)\nplt.figure(figsize=(15, 6))\nplt.imshow(im)\nplt.title(path.split(\"/\")[-1])","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:29.978020Z","iopub.execute_input":"2022-05-04T08:00:29.978727Z","iopub.status.idle":"2022-05-04T08:00:31.449293Z","shell.execute_reply.started":"2022-05-04T08:00:29.978691Z","shell.execute_reply":"2022-05-04T08:00:31.447716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating label in train.csv\n\ntrain_data['label'] = train_data.species.map(lambda x: 'whale' if 'whale' in x else 'dolphin')","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:31.450752Z","iopub.execute_input":"2022-05-04T08:00:31.451272Z","iopub.status.idle":"2022-05-04T08:00:31.473703Z","shell.execute_reply.started":"2022-05-04T08:00:31.451225Z","shell.execute_reply":"2022-05-04T08:00:31.472721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Barchart of Whale vs Dolphin count\n\ndata = train_data['label'].value_counts().reset_index()\nfig = px.bar(data, x='index', y='label', color='label', title='Whale Vs Dolphin', text_auto=True)\nfig.update_traces(textfont_size=12, textangle=0, textposition=\"outside\", cliponaxis=False)\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:31.474986Z","iopub.execute_input":"2022-05-04T08:00:31.475308Z","iopub.status.idle":"2022-05-04T08:00:32.468964Z","shell.execute_reply.started":"2022-05-04T08:00:31.475280Z","shell.execute_reply":"2022-05-04T08:00:32.468279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting proportion of Whales vs Dolphins\n\nfig, ax  = plt.subplots(figsize=(16, 8))\nfig.suptitle('Whales and Dolphins ', size = 20, font=\"Serif\")\nexplode = (0.05, 0.05)\nlabels = list(train_data.label.value_counts().index)\nsizes = train_data.label.value_counts().values\nax.pie(sizes, explode=explode,startangle=60, labels=labels,autopct='%1.0f%%', pctdistance=0.7, colors=[\"#0077b6\",\"#90e0ef\"])\nax.add_artist(plt.Circle((0,0),0.4,fc='white'))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.470427Z","iopub.execute_input":"2022-05-04T08:00:32.470653Z","iopub.status.idle":"2022-05-04T08:00:32.629287Z","shell.execute_reply.started":"2022-05-04T08:00:32.470627Z","shell.execute_reply":"2022-05-04T08:00:32.628235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Model building**","metadata":{}},{"cell_type":"code","source":"#loading train.csv as train_df\n\ntrain_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.631077Z","iopub.execute_input":"2022-05-04T08:00:32.631491Z","iopub.status.idle":"2022-05-04T08:00:32.721710Z","shell.execute_reply.started":"2022-05-04T08:00:32.631438Z","shell.execute_reply":"2022-05-04T08:00:32.720895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the dimension of the train_df \n\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.722902Z","iopub.execute_input":"2022-05-04T08:00:32.723107Z","iopub.status.idle":"2022-05-04T08:00:32.728005Z","shell.execute_reply.started":"2022-05-04T08:00:32.723083Z","shell.execute_reply":"2022-05-04T08:00:32.726781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying First Five column of the train_df\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.729176Z","iopub.execute_input":"2022-05-04T08:00:32.729419Z","iopub.status.idle":"2022-05-04T08:00:32.746262Z","shell.execute_reply.started":"2022-05-04T08:00:32.729391Z","shell.execute_reply":"2022-05-04T08:00:32.745376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking for null values\n\ntrain_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.747575Z","iopub.execute_input":"2022-05-04T08:00:32.747952Z","iopub.status.idle":"2022-05-04T08:00:32.775965Z","shell.execute_reply.started":"2022-05-04T08:00:32.747908Z","shell.execute_reply":"2022-05-04T08:00:32.774997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Removing duplicate values form the train_df of column individual_id\n\ntrain_df=train_df.drop_duplicates(subset=['individual_id'],keep='last')\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.777458Z","iopub.execute_input":"2022-05-04T08:00:32.777886Z","iopub.status.idle":"2022-05-04T08:00:32.798366Z","shell.execute_reply.started":"2022-05-04T08:00:32.777841Z","shell.execute_reply":"2022-05-04T08:00:32.797303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function will Load Images of certain dimension\n\ndef Loading_Images(data, m, dataset):\n    print(\"Loading images\")\n    X_train = np.zeros((m, 32, 32, 3))\n    count = 0\n    for fig in tqdm(data['image']):\n        img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+fig, target_size=(32, 32, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        X_train[count] = x\n        count += 1\n    return X_train","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.799720Z","iopub.execute_input":"2022-05-04T08:00:32.800404Z","iopub.status.idle":"2022-05-04T08:00:32.808115Z","shell.execute_reply.started":"2022-05-04T08:00:32.800357Z","shell.execute_reply":"2022-05-04T08:00:32.807277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function will convert the text category to numeric\n\ndef prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.809567Z","iopub.execute_input":"2022-05-04T08:00:32.810013Z","iopub.status.idle":"2022-05-04T08:00:32.824827Z","shell.execute_reply.started":"2022-05-04T08:00:32.809978Z","shell.execute_reply":"2022-05-04T08:00:32.824099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = Loading_Images(train_df, train_df.shape[0], \"train_images\")\nX /= 255","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:00:32.828359Z","iopub.execute_input":"2022-05-04T08:00:32.829287Z","iopub.status.idle":"2022-05-04T08:17:13.455886Z","shell.execute_reply.started":"2022-05-04T08:00:32.829149Z","shell.execute_reply":"2022-05-04T08:17:13.454637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, label_encoder = prepare_labels(train_df['individual_id'])","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:17:13.458066Z","iopub.execute_input":"2022-05-04T08:17:13.458977Z","iopub.status.idle":"2022-05-04T08:17:13.675242Z","shell.execute_reply.started":"2022-05-04T08:17:13.458923Z","shell.execute_reply":"2022-05-04T08:17:13.674372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:17:13.676641Z","iopub.execute_input":"2022-05-04T08:17:13.676897Z","iopub.status.idle":"2022-05-04T08:17:13.683791Z","shell.execute_reply.started":"2022-05-04T08:17:13.676867Z","shell.execute_reply":"2022-05-04T08:17:13.683043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a Module","metadata":{}},{"cell_type":"code","source":"# Creating Model\n\nmodel = Sequential()\n\nmodel.add(Conv2D(32, (6, 6), strides = (1, 1), input_shape = (32, 32, 3)))\nmodel.add(BatchNormalization(axis = 3))\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1)))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3)))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation=\"relu\"))\nmodel.add(Dropout(0.85))\n\nmodel.add(Dense(y.shape[1], activation='softmax'))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:17:13.684873Z","iopub.execute_input":"2022-05-04T08:17:13.685790Z","iopub.status.idle":"2022-05-04T08:17:14.105458Z","shell.execute_reply.started":"2022-05-04T08:17:13.685739Z","shell.execute_reply":"2022-05-04T08:17:14.104327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model fitting \n\nhistory = model.fit(X, y, epochs=400, batch_size=128, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:17:14.106842Z","iopub.execute_input":"2022-05-04T08:17:14.107096Z","iopub.status.idle":"2022-05-04T10:01:39.742616Z","shell.execute_reply.started":"2022-05-04T08:17:14.107057Z","shell.execute_reply":"2022-05-04T10:01:39.740812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving our model for later use\n\nmodel.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-05-04T10:01:39.745977Z","iopub.execute_input":"2022-05-04T10:01:39.746367Z","iopub.status.idle":"2022-05-04T10:01:40.069978Z","shell.execute_reply.started":"2022-05-04T10:01:39.746304Z","shell.execute_reply":"2022-05-04T10:01:40.069331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluation of the model","metadata":{}},{"cell_type":"code","source":"# Plotting the accuracy of the model over the epochs\n\nplt.figure(figsize=(15,5))\nplt.plot(history.history['accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T10:01:40.071437Z","iopub.execute_input":"2022-05-04T10:01:40.071913Z","iopub.status.idle":"2022-05-04T10:01:40.323781Z","shell.execute_reply.started":"2022-05-04T10:01:40.071880Z","shell.execute_reply":"2022-05-04T10:01:40.322808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the loss of the model over the epochs\n\nplt.figure(figsize=(15,5))\nplt.plot(history.history['loss'])\nplt.title('Model loss')\nplt.ylabel('loss')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T10:01:40.325054Z","iopub.execute_input":"2022-05-04T10:01:40.325293Z","iopub.status.idle":"2022-05-04T10:01:40.536230Z","shell.execute_reply.started":"2022-05-04T10:01:40.325263Z","shell.execute_reply":"2022-05-04T10:01:40.535608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inference","metadata":{}},{"cell_type":"code","source":"test = os.listdir(\"../input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2022-05-04T10:01:40.537610Z","iopub.execute_input":"2022-05-04T10:01:40.538012Z","iopub.status.idle":"2022-05-04T10:01:40.565210Z","shell.execute_reply.started":"2022-05-04T10:01:40.537975Z","shell.execute_reply":"2022-05-04T10:01:40.564097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''\n#test_df=test_df.head(n=250)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T10:01:40.566423Z","iopub.execute_input":"2022-05-04T10:01:40.566973Z","iopub.status.idle":"2022-05-04T10:01:40.590101Z","shell.execute_reply.started":"2022-05-04T10:01:40.566938Z","shell.execute_reply":"2022-05-04T10:01:40.589238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=5000\nbatch_start = 0\nbatch_end = batch_size\nL = len(test_df)\n\nwhile batch_start < L:\n    limit = min(batch_end, L)\n    test_df_batch = test_df.iloc[batch_start:limit]\n    print(type(test_df_batch))\n    X = Loading_Images(test_df_batch, test_df_batch.shape[0], \"test_images\")\n    X /= 255\n    predictions = model.predict(np.array(X), verbose=1)\n    for i, pred in enumerate(predictions):\n        p=pred.argsort()[-5:][::-1]\n        idx=-1\n        s=''\n        s1=''\n        s2=''\n        for x in p:\n            idx=idx+1\n            if pred[x]>0.6:\n                s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n            else:\n                s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n        s= s1 + ' new_individual' + s2\n        s = s.strip(' ')\n        test_df.loc[ batch_start + i, 'predictions'] = s\n    batch_start += batch_size   \n    batch_end += batch_size\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-04T10:01:40.591431Z","iopub.execute_input":"2022-05-04T10:01:40.591704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating submission.csv and printing first five rows\n\ntest_df.to_csv('submission.csv',index=False)\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}