{"cells":[{"metadata":{"_uuid":"da035fe58e548e8b1b7e8e89725b9e6bc745aa7b"},"cell_type":"markdown","source":"# Humpback Whale Identification\n\n## CNN (Keras) with `new_whale` threeshold\n\nNotebook adapted from https://www.kaggle.com/pestipeti/keras-cnn-starter"},{"metadata":{"_uuid":"0d9c73ad23e6c2eae3028255ee00c3254fe66401","trusted":true},"cell_type":"code","source":"import gc\nimport os\n\nimport numpy as np\nimport pandas as pd\nimport progressbar\n\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.models import Model\n\nimport keras.backend as K\nfrom keras.models import Sequential\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f46b24dbba74f22833cac6140e60348b15a8e047","trusted":true},"cell_type":"code","source":"def prepareImages(data, m, dataset):\n    #print(\"Preparing images\")\n    #print(m)\n    X_train = np.zeros((m, 128, 128, 3))\n    count = 0\n    \n    for fig in progressbar.progressbar(data['Image']):\n        # load images into images of size 128x128x3\n        img = image.load_img(\"../input/\"+dataset+\"/\"+fig, target_size=(128, 128, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        X_train[count] = x\n        count += 1\n    \n    return X_train","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6587a101b58af064af0f9c60a1070c6c8f52d45f","trusted":false},"cell_type":"code","source":"def prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    #print(integer_encoded)\n\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    #print(onehot_encoded)\n\n    y = onehot_encoded\n    #print(y.shape)\n    return y, label_encoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"75d42ee62d9e5ebaa325ce29db71df7e0cf8be64"},"cell_type":"code","source":"train = os.listdir(\"../input/train/\")\nprint(len(train))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"46a8839e13a14eb8d16ea6823de9927ea63d5001","trusted":false},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntrain.Id.value_counts().head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4898c21ffa928476f10f867e5ee17206cb6d354b"},"cell_type":"markdown","source":"---"},{"metadata":{"trusted":false,"_uuid":"9ef2118c8d8cbb8a1b2a1a79896f863962b1cec7"},"cell_type":"code","source":"# From https://www.kaggle.com/suicaokhoailang/removing-class-new-whale-is-a-good-idea\nif not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    train_df = train[train['Id'] != 'new_whale']\n    train_df.Id.value_counts().head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7d25b2b8efb50ec07f61d98133ed8779e7629fb8"},"cell_type":"markdown","source":"Maybe a better idea would be to still integrate `new_whale` data, but rather as a vector `np.zeros(...)` (with `softmax` activation) than a separate class."},{"metadata":{"_uuid":"4afe4128a0cd6859848c8a80686208082d647c39","trusted":false},"cell_type":"code","source":"if not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    X = prepareImages(train_df, train_df.shape[0], \"train\")\n    X /= 255","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"675924f8863aef27cf90dc668e0a68cd609dfc1c","trusted":false},"cell_type":"code","source":"if not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    y, label_encoder = prepare_labels(train_df['Id'])\n    y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7409cfbd0b7b63a0f2c3add2fee31d8ee89b7d40"},"cell_type":"code","source":"# Free Memory! Free Memory!\nif not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    del train_df\n    gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3c24bba33860fa8dc132eb5ed026b53292dbbd22"},"cell_type":"markdown","source":"## Train our model"},{"metadata":{"_uuid":"e7af799d186a1b97b6aa325d7d576a1fb55a6c5d","trusted":false},"cell_type":"code","source":"if not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    model = Sequential()\n\n    model.add(Conv2D(32, (7, 7), strides = (1, 1), name = 'conv0', input_shape = (128, 128, 3)))\n\n    model.add(BatchNormalization(axis = 3, name = 'bn0'))\n    model.add(Activation('relu'))\n\n    model.add(MaxPooling2D((2, 2), name='max_pool'))\n    model.add(Conv2D(64, (3, 3), strides = (1,1), name=\"conv1\"))\n    model.add(Activation('relu'))\n    model.add(AveragePooling2D((3, 3), name='avg_pool'))\n\n    model.add(Flatten())\n    model.add(Dense(500, activation=\"relu\", name='rl'))\n    model.add(Dropout(0.8))\n    model.add(Dense(y.shape[1], activation='softmax', name='sm'))\n\n    model.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\n    model.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"169f45e150c3a584e0f655a8eda523e0675da63a","trusted":false},"cell_type":"code","source":"if not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    history = model.fit(X, y, epochs=100, batch_size=100, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a5bb81e7eafb571fc50d372fb99628934141671d"},"cell_type":"code","source":"# Free Memory! Free Memory!\nif not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    del X, y\n    gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7bca48a1d0963cbf70685b75431435cef9499895","trusted":false},"cell_type":"code","source":"if not os.path.isfile(\"keras-cnn-starter-without-new-whales.json\"):\n    plt.plot(history.history['acc'])\n    plt.title('Model accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"317846168cd6e018a254a085b59b35c84ea843bc"},"cell_type":"markdown","source":"## Serialize Keras model to `.json`\n\nFrom https://keras.io/models/about-keras-models/"},{"metadata":{"trusted":false,"_uuid":"b007bf5331a93df04f357c78b647f4c386b4d91f"},"cell_type":"code","source":"if not os.path.isfile(\"../input/keras-cnn-starter-without-new-whales.json\"):\n    with open(\"../input/keras-cnn-starter-without-new-whales.json\", \"w\") as f:\n        f.write(model.to_json())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3d8487695e3579bd923f8aeddda2cbf7c4c39928"},"cell_type":"markdown","source":"---"},{"metadata":{"trusted":false,"_uuid":"38dba3472c29430e5ad3582c278a9c0b442bf159"},"cell_type":"code","source":"from keras.models import model_from_json\n\nwith open('../input/keras-cnn-starter-without-new-whales.json', 'r') as f:\n    model = model_from_json(f.read())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"51fb136aa17748f00aa59ae4b538363992e74975"},"cell_type":"markdown","source":"## `MAP@5` Score computation function\n\nFrom https://www.kaggle.com/pestipeti/explanation-of-map5-scoring-metric"},{"metadata":{"trusted":false,"_uuid":"0b9149b9037d74de8c7e1713a35c71db5791b434"},"cell_type":"code","source":"def map_per_image(label, predictions):\n    \"\"\"Computes the precision score of one image.\n\n    Parameters\n    ----------\n    label : string\n            The true label of the image\n    predictions : list\n            A list of predicted elements (order does matter, 5 predictions allowed per image)\n\n    Returns\n    -------\n    score : double\n    \"\"\"    \n    try:\n        return 1 / (predictions[:5].index(label) + 1)\n    except ValueError:\n        return 0.0","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"74d1a9ad5734dd9c2ccefca46d92491922fc7935"},"cell_type":"code","source":"def map_per_set(labels, predictions):\n    \"\"\"Computes the average over multiple images.\n\n    Parameters\n    ----------\n    labels : list\n             A list of the true labels. (Only one true label per images allowed!)\n    predictions : list of list\n             A list of predicted elements (order does matter, 5 predictions allowed per image)\n\n    Returns\n    -------\n    score : double\n    \"\"\"\n    return np.mean([map_per_image(l, p) for l,p in zip(labels, predictions)])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7495d3bfd2ff0a324b8321302338a9af7089f208"},"cell_type":"markdown","source":"## Fit `new_whale` threeshold"},{"metadata":{"_uuid":"52262195fc0b8755cff78bf8c98e6116d50f79af","trusted":false},"cell_type":"code","source":"X = prepareImages(train, train.shape[0], \"train\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ab9910e2c449a2e8a43e34a441c271cac77d8094"},"cell_type":"code","source":"y, label_encoder = prepare_labels(train['Id'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"88c8d8ff98fbdb1df4218abb6bd51889e855a6fb","trusted":false},"cell_type":"code","source":"predictions_encoded = model.predict(np.array(X), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"bb7f1718c465effa287d1b88d917e86d9b050644"},"cell_type":"code","source":"# Free Memory! Free Memory!\ndel X, y\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8a4c496cfc98e0f48604e5dc7dc17a63ce7b12bf"},"cell_type":"markdown","source":"---"},{"metadata":{"trusted":false,"_uuid":"33e243992072155edf8b72f8de5e6910b0a7ac8e"},"cell_type":"code","source":"best_pred = max(predictions_encoded.flatten())\nbest_pred # TODO ???!!! should not be < THREESHOLD","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ee72c0dd5666ea7d547ea61515e084b8f7a7bea6"},"cell_type":"code","source":"worst_pred = min(predictions_encoded.flatten())\nworst_pred","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c126836fa2ed0bf5ae726d5c10a5c33b8401c11a"},"cell_type":"markdown","source":"---"},{"metadata":{"trusted":false,"_uuid":"528ad6ebdbf9c73d5309b885b1b194b1869ea931"},"cell_type":"code","source":"# Function that's assign class \"new_whale\" to encoders with low threeshold\ndef get_top5(treeshold, pred):\n    args5 = pred.argsort()[-5:][::-1]\n    classes5 = [i for i in label_encoder.inverse_transform(args5)]\n    for i, t in enumerate(args5):\n        if pred[t] < treeshold:\n            for j in range(i + 1, 5):\n                classes5[j] = classes5[j - 1]\n            classes5[i] = \"new_whale\"\n            break\n    return classes5","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"66f0bdde31b8c7847916268aa82d9a1bdc9c0658","trusted":false},"cell_type":"code","source":"X_ = []\ny_ = []\n\ndef get_score(treeshold):\n    print(\"get_score(%s) = \" % treeshold, end=\"\")\n    predictions = []\n    for i, pred in enumerate(predictions_encoded):\n        predictions.append(get_top5(treeshold, pred))\n    result = map_per_set(train['Id'].values, predictions)\n    print(result)\n    X_.append(treeshold)\n    y_.append(result)\n    return result","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"10f3d998a222d5e30972c918bd084a2422bf72e6"},"cell_type":"markdown","source":"---"},{"metadata":{"trusted":false,"_uuid":"ca2982af7aa0dd481f866b27cf14d09e9f1fecdb"},"cell_type":"code","source":"# From https://www.scipy-lectures.org/advanced/mathematical_optimization/#getting-started-1d-optimization\nfrom scipy import optimize","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"046f618e94e3afad02475ebd5e62055fe49c8b40"},"cell_type":"code","source":"# \"new_whale\" threeshold -> will converge to optimal split threeshold\nresult = optimize.minimize_scalar(lambda x: - get_score(x), bounds=(worst_pred, best_pred), method='bounded')  # -SCORE (we try to minimize the function)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f754b3dd17a3a6e99afd09bd9d14702f957c671b"},"cell_type":"code","source":"plt.plot(X_, y_, '-')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"66f0bdde31b8c7847916268aa82d9a1bdc9c0658","trusted":false},"cell_type":"code","source":"new_whale_treeshold = result.x\nnew_whale_treeshold # == best_pred :(","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"867be3bb3d10d29a2c2a3b7fb1a29add36241de6"},"cell_type":"markdown","source":"---"},{"metadata":{"trusted":false,"_uuid":"905dda751b07e74c84b907bedfe720b442883988"},"cell_type":"code","source":"train_new_whale = train[train['Id'] == 'new_whale']\ntrain_new_whale.Id.value_counts().head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"5866a6c82e6b4dc8cef87aaff21b1d8793e5f377"},"cell_type":"code","source":"X = prepareImages(train_new_whale, train_new_whale.shape[0], \"train\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"3817e2c320cfb89993ed5b4da89f0e98fb2c4f9f"},"cell_type":"code","source":"predictions_encoded_new_whale = model.predict(np.array(X), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"5d02d5ce9d785830f3a7138f5d5c64a4aee1c9b8"},"cell_type":"code","source":"np.mean(predictions_encoded_new_whale.flatten())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"66f0bdde31b8c7847916268aa82d9a1bdc9c0658","trusted":false},"cell_type":"code","source":"# Free Memory! Free Memory!\ndel train, train_new_whale\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9db60ab2d0d6769e69f8c9dea6857b2595ca246c"},"cell_type":"markdown","source":"## Predict labels of test dataset"},{"metadata":{"_uuid":"debe961c93b72bef151d9aad3ca2cb500ee00aaa","trusted":false},"cell_type":"code","source":"test = os.listdir(\"../input/test/\")\nprint(len(test))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"72ed8198f519f7b1ae3efbc688933c78d8cdd0e4","trusted":false},"cell_type":"code","source":"col = ['Image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['Id'] = ''","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"52262195fc0b8755cff78bf8c98e6116d50f79af","trusted":false},"cell_type":"code","source":"X = prepareImages(test_df, test_df.shape[0], \"test\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"88c8d8ff98fbdb1df4218abb6bd51889e855a6fb","trusted":false},"cell_type":"code","source":"predictions_encoded = model.predict(np.array(X), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"66f0bdde31b8c7847916268aa82d9a1bdc9c0658","trusted":false},"cell_type":"code","source":"for i, pred in enumerate(predictions_encoded):\n    test_df.loc[i, 'Id'] = ' '.join(get_top5(new_whale_treeshold, pred))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"09d7c1eb9b554e4e580b0c3c7eb609c15636892d","trusted":false},"cell_type":"code","source":"test_df.head(10)\ntest_df.to_csv('keras-cnn-with-new-whale-threeshold.csv', index=False) #> Score = 0.286","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"09d7c1eb9b554e4e580b0c3c7eb609c15636892d","trusted":false},"cell_type":"code","source":"# Free Memory! Free Memory!\ndel test_df, X\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"308d54e439a55dbf8604d6cacc2398c4b1dfb26c"},"cell_type":"code","source":"!kaggle competitions submit -c humpback-whale-identification -f \"keras-cnn-with-new-whale-threeshold.csv\" -m \"CNN with Keras with new_whale (threeshold = 0.00022381447120760044)\"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}