{"cells":[{"metadata":{"_uuid":"6aa4f4c74c16b5af88fbf82d8538f1b55291338b"},"cell_type":"markdown","source":"This kernel is to test applicability of a CNN approach to whale testing. \nA straightforward CNN approach is taken, and the proper testing done.\n\nResults are miserable, to say the least. Either I made some kind of a very stupid mistake, or the approach simply does not work: the network memorizes patterns, but does not generalize.\n\nTo avoid all possible problems I could think of, I converted everything (pandas) to numpy, and did all data processing in steps i could inspect. No use.\n\nI would appreciate suggestions, especially if I have made mistakes. Note: I CAN use pre-trained networks. It is just that I want to make \"plain\" CNNs work."},{"metadata":{"trusted":true,"_uuid":"0d9c73ad23e6c2eae3028255ee00c3254fe66401"},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import LabelBinarizer\n\nimport keras\nfrom keras import layers\nfrom keras import regularizers\nfrom keras.preprocessing import image\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.models import Model\nfrom keras.callbacks import ModelCheckpoint\n\nfrom tqdm import tqdm\n\nimport keras.backend as K\nfrom keras.models import Sequential\n\nfrom PIL import Image\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)\n\nwarnings.filterwarnings(\"ignore\")\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\nnp.random.seed(7)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a35cd8c45b2898cebc3af06837de2f11a90459f"},"cell_type":"code","source":"IMAGE_SIZE = 128","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cea35de3530cc898be5b85063b84e875401d092"},"cell_type":"code","source":"os.listdir(\"../input/\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4883e25b5e8ec325cee5e1396a3afb3fe6d6e0de"},"cell_type":"code","source":"# We load the training data, and divide it into two parts: \n# marked properly vs new_whales. The later should not be used in training.\n\ndf = pd.read_csv(\"../input/train.csv\") #.set_index('Image')\nnew_whale_df = df[df.Id == \"new_whale\"] # only new_whale dataset\ntrain_df = df[~(df.Id == \"new_whale\")] # no new_whale dataset, used for training\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33e81455a9a01a4494e4109c3c7dade08b3baaeb"},"cell_type":"code","source":"print(len(df), len(train_df), len(new_whale_df))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"505dbc11fa3f589584454676c4e397a4c998d481"},"cell_type":"code","source":"# We have enough ids with more than one image corresponding to them.\ntrain_df.Id.value_counts().head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"471ad919fa1189797ae19d9f7f207f9695b0b955"},"cell_type":"code","source":"# The idea is to take images (only of those ids that have multiple images, from the\n# top of the list ordered by num. of entities) and move them to the validation list.\n# This way we create validation list without reducing number of labels in a training list.\n\n# First, we create a list of unique ids:\n\ntrain_df_sorted = train_df.groupby('Id')['Image'].nunique().sort_values(\n    ascending=False).reset_index(name='count')\ntrain_df_sorted = train_df_sorted[:2000]\ntrain_df_sorted.head()\n\n# Second (later) we are going to use this list to move some items to validation set.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37d70f67d56f268f0d0adcea1b95d35e6201a01f"},"cell_type":"code","source":"# Get one-hot encoded labels along with an encoder.\n# Note that we can use same encoder on training and validation sets.\n\ndef prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    print(integer_encoded)\n\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n\n    y = onehot_encoded\n    \n    return y, label_encoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d48aab63cf4dd7c5b7be62143f1af32a42f1610d"},"cell_type":"code","source":"# Prepare labels. The call to prepare_labels(y) creates an array\n# of one-hot encoded labels. Note that ALL unique labels must be present,\n# but as we are going to only move to a validation set items that have more than\n# one image (so the rest of images stays in a training set), this requirement\n# is satisfied. \ntrain_labels, label_encoder = prepare_labels(train_df['Id'])\n\nprint(len(train_labels))\n\n#for i in range(10):\n#    print(train_labels[i])\n\n# OR:\n#lb_style = LabelBinarizer()\n#lb_results = lb_style.fit_transform(np.array(train_df['Id']))\n#pd.DataFrame(lb_results, columns=lb_style.classes_).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60e041540aea6d54b8cbd80d7980cb37967c27cc"},"cell_type":"code","source":"# Break train_df into train_df, valid_df and test_df.\n# Earlier, we have created train_df_sorted, containing unique\n# labels along with count. Its size is 2000 elements.\n# Now we are going to move first (having largest count) elements from train_df\n# to valid_df and test_df, 1000 elements to each.\n# In the same time, we will remove these records from train_df \n# Note that we have to move labels as well.\n\n# test_df is not used in this particular kernel, as I have encountered problems \n# (see charts below) during training itself.\n\n#valid_df = pd.DataFrame(columns=train_df.columns)\n#test_df = pd.DataFrame(columns=train_df.columns)\n\narr_train = []\narr_valid = []\narr_test = []\n\ntrain_df = train_df.reset_index(drop=True)\n\nvalid_labels = []\ntest_labels = []\n\narrToDelete = []\nfor i in tqdm(range(len(train_df_sorted))):\n    idx = train_df[train_df['Id'] == train_df_sorted['Id'][i]].index[0]\n    \n    if(i%2):\n        arr_valid.append([train_df['Image'][idx], train_df['Id'][idx]])\n        valid_labels.append(train_labels[idx])\n    else:\n        arr_test.append([train_df['Image'][idx], train_df['Id'][idx]])\n        test_labels.append(train_labels[idx])\n        \n    arrToDelete.append(idx)\n\nfor idx, row in train_df.iterrows():\n    arr_train.append([row['Image'], row['Id']])\n    \nnp_train = np.array(arr_train)\nnp_valid = np.array(arr_valid)\nnp_test = np.array(arr_test)\n    \n# If the next two lines are commented in order to keep valid and test data in train_df, \n# loss and valid_loss, acc and valid_acc change in synch. If uncommented,\n# the network gets heavily overtrained.\n# In other words, by NOT removing the valid data from training set, we are going\n# to see the network converging nicely, so it can learn to memorize the examples it\n# saw already. But if the two lines are not commented out (which is the way it should be,\n# as we need the network to be able to work with images it never saw), the network\n# fails to converge (on a validation set).\ntrain_labels = np.delete(train_labels, arrToDelete, 0)\nnp_train = np.delete(np_train, arrToDelete, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26b2ae6a4373e0d0f36189bb36d9f1ffdbb59d36"},"cell_type":"code","source":"print(len(arrToDelete), len(np_train), len(np_valid), len(np_test),\n     len(train_labels), len(valid_labels))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"743378ac74968f98d92e0e56f12b6a2d19371f89"},"cell_type":"code","source":"# Load images for model.fit to use\ndef prepareImages(data, m, dataset):\n    print(\"Preparing images\")\n    X_train = np.zeros((m, IMAGE_SIZE, IMAGE_SIZE, 3))\n    count = 0\n    \n    for fig in data[:, 0]:\n        #load images into images of size IMAGE_SIZExIMAGE_SIZEx3\n        img = image.load_img(\"../input/\"+dataset+\"/\"+fig, \n            target_size=(IMAGE_SIZE, IMAGE_SIZE, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        X_train[count] = x\n        if (count%500 == 0):\n            print(\"Processing image: \", count+1, \", \", fig)\n        count += 1\n    \n    return X_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4fa08ed97cf8b44a3ddd0ad43e927e3d5dde58eb"},"cell_type":"code","source":"# Prepare images. The call to prepareImages() creates an array\n# of size (m, IMAGE_SIZE, IMAGE_SIZE, 3) for a provided dataframe m\n# No X /= 255 as prepare_image does it in preprocess_input(x) call\nx_train = prepareImages(np_train, len(np_train), \"train\")\nx_valid = prepareImages(np_valid, len(np_valid), \"train\")\nx_test = prepareImages(np_test, len(np_test), \"train\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b7ee977d53732ae57edd4eaf07c183fd76b5732"},"cell_type":"code","source":"print(len(x_train), len(x_valid), len(valid_labels))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7500260e5b2d87d31c123b03c21cb8745d61e6c5"},"cell_type":"code","source":"# Now we want to add 9664 / 13697 * 1000 = 705 new_whale records to test_df\n# Note that this part is not used in this particular kernel, as the network \n# failed to learn, so it newer got as far as testing.\n#np_test = np.append(new_whale_df[:705])\n#print(len(test_df))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9ef51c414310fd7d4bc08b28017eac19018cd6b"},"cell_type":"code","source":"# Just out of curiocity, let's take a look at our pictures.\nfig = plt.figure(figsize=(15, 5))\ntrain_imgs = os.listdir(\"../input/train\")\nfor idx, img in enumerate(np.random.choice(train_df.Image, 8)):\n    ax = fig.add_subplot(2, 4, idx+1, xticks=[], yticks=[])\n    im = Image.open(\"../input/train/\" + img)\n   \n    plt.imshow(im)\n    lab = train_df.loc[train_df.Image == img, 'Id'].values[0]\n    ax.set_title(f'Label: {lab}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60ae068446f3b628e2e2fcfb262467835bae2c7b"},"cell_type":"code","source":"# This callback function will save the best model at the end of each epoch, \n# based on the best val_loss obtained.\n# Note that Keras does not attomatically use the best model, so we have to\n# load this model later on, or we will end up with the LAST model, not the\n# best one.\n\nbest_weights_filepath = \"best_weights.txt\"\n\nsaveBestModel = ModelCheckpoint(best_weights_filepath, monitor=\"val_loss\",\n    save_best_only=True, save_weights_only=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7af799d186a1b97b6aa325d7d576a1fb55a6c5d"},"cell_type":"code","source":"# Create the network itself. Here is a simple configuration, but I have\n# tried quite a few different ones, with no success: poor generalization.\n\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), strides = (1, 1), name = 'conv0', \n    input_shape = (IMAGE_SIZE, IMAGE_SIZE, 3),\n    kernel_regularizer=regularizers.l2(0.01),\n    activity_regularizer=regularizers.l1(0.01)))\nmodel.add(BatchNormalization(axis = 3))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(64, (3, 3), strides = (1, 1),\n    kernel_regularizer=regularizers.l2(0.01),\n    activity_regularizer=regularizers.l1(0.01)))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.3))\n\nmodel.add(Flatten())\nmodel.add(Dense(500, activation=\"relu\", name='rl',\n    kernel_regularizer=regularizers.l2(0.01),\n    activity_regularizer=regularizers.l1(0.01)))\nmodel.add(Dropout(0.3))\n\nmodel.add(Dense(train_labels.shape[1], activation='softmax', name='sm',\n    kernel_regularizer=regularizers.l2(0.01),\n    activity_regularizer=regularizers.l1(0.01)))\n\nmodel.compile(loss=keras.losses.categorical_crossentropy,\n    optimizer=\"adam\", metrics=['accuracy'])\n    #keras.optimizers.Adam(lr=0.001)\n    # keras.optimizers.SGD(lr=0.01, nesterov=True)\n\n#model = Sequential()\n#model.add(Conv2D(32, kernel_size=(3, 3),\n#    activation='relu',\n#    input_shape=(IMAGE_SIZE, IMAGE_SIZE, 3)))\n#model.add(Conv2D(64, (3, 3), activation='relu'))\n#model.add(MaxPooling2D(pool_size=(2, 2)))\n#model.add(Dropout(0.25))\n#model.add(Flatten())\n#model.add(Dense(128, activation='relu'))\n#model.add(Dropout(0.5))\n#model.add(Dense(train_labels.shape[1], activation='softmax'))\n#model.compile(loss=keras.losses.categorical_crossentropy,\n#    optimizer=keras.optimizers.RMSprop(lr=0.0001),\n#    metrics=['accuracy'])    \nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"874988e98e229fe98d498821412858a9388e28ee"},"cell_type":"code","source":"# This ImageDataGenerator tries its best to make images different\n# between epochs, but we still see the network memorizing samples,\n# not generalizing (good results on training set, poor on validation).\n\n# Note: I have tried all the commented out transformations: it does not improve the result\n\ndatagen = ImageDataGenerator(\n    #featurewise_center=True\n    #,featurewise_std_normalization=True\n    #,rotation_range=10\n    #,width_shift_range=0.2\n    #,height_shift_range=0.2\n    #,zoom_range=0.2\n    )\n\ndatagen.fit(x_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"395f6bd47fe4b259ea33f2e8b27259ad60401cd1"},"cell_type":"code","source":"# Make sure that previous \"best network\" is deleted.\nif(os.path.isfile(best_weights_filepath)):\n    os.remove(best_weights_filepath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64102f29f89b42f57ad8960e61b3a6049d478f6b"},"cell_type":"code","source":"# Fit the model using training and validation sets.\nhistory = model.fit_generator(datagen.flow(x_train, train_labels, \n    batch_size=100), steps_per_epoch=len(x_train) / 100 - 1, epochs=40,\n    validation_data=([x_valid], [valid_labels]),\n    callbacks=[saveBestModel])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab854bf881b5455855f5a626908ab919f97161f2"},"cell_type":"code","source":"# Load the \"best model\" we saved during the training\nmodel.load_weights(best_weights_filepath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bca48a1d0963cbf70685b75431435cef9499895"},"cell_type":"code","source":"# Create a chart for accuracy for training and validation sets\nplt.plot(history.history['acc'])\nplt.plot(history.history['val_acc'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b752b81f3d1d4e64791d6a4323abbb24b994ada3"},"cell_type":"code","source":"# Create a chart for loss for training and validation sets\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1d569b39df8932b61ca00b6f10b2f678a875acab"},"cell_type":"markdown","source":"The rest of the kernel is not used. The point of this little research can be seen from the charts: CNN \nis not a good tool for the task.\nOnce again, if this is due to my mistake, please let me know."},{"metadata":{"trusted":true,"_uuid":"af0763f43ace2a8086bb73a4d0e88d9212ef6f30"},"cell_type":"code","source":"# For x_test and test_labels, do predicting\n#predictions = model.predict(x_test, batch_size=200)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26cf35013dc8be64ed1dc25c663b080ab00683c3"},"cell_type":"code","source":"#for i, pred in enumerate(predictions):\n#    arrResult = pred[pred.argsort()[-5:][::-1]]\n#    arrResultLabels = label_encoder.inverse_transform(pred.argsort()[-5:][::-1])\n#    strResult = \"\"\n#    for j in range(4):\n##        if(arrResult[j] < 0.50):\n##            strResult = strResult + \"new_whale \"\n##        else:\n#         strResult = strResult + arrResultLabels[j] + \" \"    \n#    \n#    test_df.loc[i, 'Id'] = strResult\n#    print(strResult)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72ed8198f519f7b1ae3efbc688933c78d8cdd0e4"},"cell_type":"code","source":"#col = ['Image']\n#test_df = pd.DataFrame(test, columns=col)\n#test_df['Id'] = ''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4fdd89bfe79305d92e26ac1c2cf22662e459b219"},"cell_type":"code","source":"##X = prepareImages(valid_df, valid_df.shape[0], \"train\")\n##predictions = model.predict(np.array(X), verbose=1)\n##for i, pred in enumerate(predictions):\n##    print(label_encoder.inverse_transform(y[i].argsort()[-5:][::-1]))\n##    print(pred.argsort()[-5:][::-1])\n##    print(pred[pred.argsort()[-5:][::-1]])\n##    print(label_encoder.inverse_transform(pred.argsort()[-5:][::-1]))\n##    print(\"==================\")    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52262195fc0b8755cff78bf8c98e6116d50f79af"},"cell_type":"code","source":"#test_df = np.array(test_df)\n\n#X = prepareImages(test_df, test_df.shape[0], \"test\")\n##X /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88c8d8ff98fbdb1df4218abb6bd51889e855a6fb"},"cell_type":"code","source":"#predictions = model.predict(np.array(X), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66f0bdde31b8c7847916268aa82d9a1bdc9c0658"},"cell_type":"code","source":"#for i, pred in enumerate(predictions):\n#    arrResult = pred[pred.argsort()[-5:][::-1]]\n#    arrResultLabels = label_encoder.inverse_transform(pred.argsort()[-5:][::-1])\n#    strResult = \"\"\n#    for j in range(4):\n#        if(arrResult[j] < 0.90):\n#            strResult = strResult + \"new_whale \"\n#        else:\n#            strResult = strResult + arrResultLabels[j] + \" \"    \n#    \n#    test_df.loc[i, 'Id'] = strResult","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"09d7c1eb9b554e4e580b0c3c7eb609c15636892d"},"cell_type":"code","source":"#test_df.head(10)\n#test_df.to_csv('submission_public1.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}