{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.metrics import categorical_accuracy, top_k_categorical_accuracy, categorical_crossentropy\nfrom keras.models import Model\nfrom keras.applications import MobileNet\nfrom keras.optimizers import Adam\nimport seaborn as sns\n\nimport keras.backend as K\nfrom keras.models import Sequential\n\n\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"","_uuid":"","trusted":true},"cell_type":"code","source":"import pandas as pd\ntrain = pd.read_csv(\"../input/humpback-whale-identification/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = image.load_img('../input/humpback-whale-identification/train'+\"/\"+'07b1a8065.jpg', target_size=(800,800,3))\nplt.imshow(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_arr = image.img_to_array(img)\nimg_2 = plt.imshow(img_arr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_arr = preprocess_input(img_arr)\nimg_3= plt.imshow(x_arr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df= train.iloc[0:15000,:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[\"Id\"].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\ncounted = train.groupby(\"Id\").count().rename(columns={\"Image\":\"image_count\"})\ncounted.loc[counted[\"image_count\"] > 80,'image_count'] = 80\nplt.figure(figsize=(20,14))\nsns.countplot(data=counted, x=\"image_count\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\nfig = plt.figure(figsize = (20, 15))\nfor idx, img_name in enumerate(train[train['Id'] == 'new_whale']['Image'][:12]):\n    y = fig.add_subplot(3, 4, idx+1)\n    img = cv2.imread(os.path.join(\"../input/humpback-whale-identification/train\",img_name))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    y.imshow(img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prepareImages(data, m, dataset):\n    print(\"Preparing images\")\n    X_train = np.zeros((m, 100, 100, 3))\n    count = 0\n    \n    for fig in data['Image']:\n        #load images into images of size 100x100x3\n        img = image.load_img(\"../input/humpback-whale-identification/\"+dataset+\"/\"+fig, target_size=(100, 100, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n\n        X_train[count] = x\n        if (count%500 == 0):\n            print(\"Processing image: \", count+1, \", \", fig)\n        count += 1\n    \n    return X_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    # print(integer_encoded)\n\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    # print(onehot_encoded)\n\n    y = onehot_encoded\n    # print(y.shape)\n    return y, label_encoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = prepareImages(train_df, train_df.shape[0], \"train\")\nX= X/255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#print(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y, label_encoder = prepare_labels(train_df['Id'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32, (7, 7), strides = (1, 1), name = 'conv0', input_shape = (100, 100, 3)))\n\nmodel.add(BatchNormalization(axis = 3, name = 'bn0'))\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2), name='max_pool'))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1), name=\"conv1\"))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3), name='avg_pool'))\n\nmodel.add(Flatten())\nmodel.add(Dense(500, activation=\"relu\", name='rl'))\nmodel.add(Dropout(0.8))\nmodel.add(Dense(y.shape[1], activation='softmax', name='sm'))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(X, y, epochs=15, batch_size=100, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.5,\n    zoom_range=0.5,\n    horizontal_flip=True,\n    rotation_range=180,\n    featurewise_center=True,\n    width_shift_range=0.5,\n    height_shift_range=0.5)\n\ntrain_datagen.fit(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_Augment = model.fit_generator(train_datagen.flow(X, y, batch_size=100), epochs=15, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['accuracy'],'k--', label='Accuracy')\nplt.plot(history_Augment.history['accuracy'],label='Augment Accuracy')\nplt.legend()\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['loss'],'k--', label='Loss')\nplt.plot(history_Augment.history['loss'], label='Augment Loss')\nplt.legend()\nplt.title('Model loss')\nplt.ylabel('categorical_crossentropy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def top_5_accuracy(y_true, y_pred):\n    return top_k_categorical_accuracy(y_true, y_pred, k=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_Mobile = MobileNet(input_shape=(100, 100, 3), alpha=1., weights=None, classes=3934)\nmodel_Mobile.compile(optimizer=Adam(lr=0.002), loss='categorical_crossentropy',\n              metrics=[categorical_crossentropy, categorical_accuracy, top_5_accuracy])\nprint(model_Mobile.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_Mobile = model_Mobile.fit(X, y, epochs=15, batch_size=100, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_Mobile_Augment = model_Mobile.fit_generator(train_datagen.flow(X, y, batch_size=100), epochs=15, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history_Mobile.history['categorical_accuracy'],'k--', label='Mobile Accuracy')\nplt.plot(history_Mobile_Augment.history['categorical_accuracy'], label='Augment Mobile Accuracy')\nplt.legend()\nplt.title('Model categorical accuracy')\nplt.ylabel('categorical accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history_Mobile.history['top_5_accuracy'],'k--', label='Mobile Accuracy top 5')\nplt.plot(history_Mobile_Augment.history['top_5_accuracy'],label='Augment Mobile Accuracy top 5')\nplt.legend()\nplt.title('Model categorical accuracy')\nplt.ylabel('categorical accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history_Mobile.history['loss'],'k--', label='Mobile loss')\nplt.plot(history_Mobile_Augment.history['loss'], label = 'Augment Mobile loss')\nplt.title('Model loss')\nplt.legend()\nplt.ylabel('categorical_crossentropy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications.resnet50 import ResNet50\nfrom keras.metrics import categorical_accuracy, top_k_categorical_accuracy, categorical_crossentropy\nfrom keras.optimizers import Adam\n\ndef top_5_accuracy(y_true, y_pred):\n    return top_k_categorical_accuracy(y_true, y_pred, k=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def pre_model():\n    base_model = ResNet50(input_shape=(100, 100, 3), weights=None, classes=3934)\n    base_model.compile(optimizer=Adam(lr=0.002), loss='categorical_crossentropy', metrics=[categorical_crossentropy, categorical_accuracy, top_5_accuracy])\n    return base_model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_res = pre_model()\nmodel_res.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ReduceLROnPlateau","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', mode='min', restore_best_weights=False)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=3)\n\ncallback = [reduce_lr]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_res = model_res.fit(X, y, epochs=15, batch_size=128, verbose=1, validation_split=0.1, callbacks=callback)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_res_Augment = model_res.fit_generator(train_datagen.flow(X, y, batch_size=100), epochs=15, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history_res_Augment.history['categorical_accuracy'], label='Augment Residual Accuracy')\nplt.plot(history_res.history['categorical_accuracy'],'k--', label='Residual Accuracy')\nplt.legend()\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history_res_Augment.history['loss'], label='Augment Residual Loss')\nplt.plot(history_res.history['loss'],'k--', label='Residual Loss')\nplt.legend()\nplt.title('Model loss')\nplt.ylabel('categorical_crossentropy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['accuracy'], label='CNN Accuracy')\nplt.plot(history_Augment.history['accuracy'], label='Augment CNN Accuracy')\nplt.plot(history_Mobile.history['categorical_accuracy'], label='Mobile Accuracy')\nplt.plot(history_Mobile_Augment.history['categorical_accuracy'], label='Augment Mobile Accuracy')\nplt.plot(history_res_Augment.history['categorical_accuracy'], label='Augment Residual Accuracy')\nplt.plot(history_res.history['categorical_accuracy'], label='Residual Accuracy')\nplt.legend()\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['loss'], label='Loss')\nplt.plot(history_Augment.history['loss'], label='Augment Loss')\nplt.plot(history_Mobile.history['loss'], label='Mobile loss')\nplt.plot(history_Mobile_Augment.history['loss'], label = 'Augment Mobile loss')   \nplt.plot(history_res_Augment.history['loss'], label='Augment Residual Loss')\nplt.plot(history_res.history['loss'], label='Residual Loss')\nplt.legend()\nplt.title('Model loss')\nplt.ylabel('categorical_crossentropy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}