{"cells":[{"metadata":{"_uuid":"e211f3f996c4321273bb9d6a77108da9be479199"},"cell_type":"markdown","source":"# Humpback whale identification using transfer learning\n\nTransfer learning allows us to use the knowledge of pretrained models for prediction on new dataset.\nIn this notebook, I have used ResNet50 architecture with imagenet weights. The architecture is then extended by  adding some more layers to base architecture.\n\nThe notebooks that I had referred are:\n- https://www.kaggle.com/whatvermawhat/resnet50-128x128\n- https://www.kaggle.com/sukhadj/humpback-whale-identification\n- https://www.kaggle.com/winstonvan/python-keras-resnet50-for-cancer\n\nThough the performance of the model on the given dataset is pretty bad, it is my attempt to simulate the transfer learning."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nimport matplotlib.pyplot as plt\n\nfrom keras.applications import ResNet50\nfrom keras.layers import Dense,Dropout,Flatten,BatchNormalization\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom keras.metrics import top_k_categorical_accuracy\n\n\nfrom sklearn.preprocessing import OneHotEncoder,LabelEncoder\n\nfrom keras.preprocessing import image\nfrom tensorflow.python.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.python.keras.preprocessing.image import ImageDataGenerator\n\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/resnet50/\"))\n\nfrom glob import glob\nfrom skimage.io import imread\n\nresnet_weights_path = '../input/resnet50/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5'","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_dir = \"../input/humpback-whale-identification/train/\"\ntest_dir = \"../input/humpback-whale-identification/test/\"\n\nsample_submission = pd.read_csv(\"../input/humpback-whale-identification/sample_submission.csv\")\n\n# train.csv\ntrain_df = pd.read_csv(\"../input/humpback-whale-identification/train.csv\")\n\n# psudo test.csv\ntest_df = pd.DataFrame(sample_submission[\"Image\"])\ntest_df['Id'] = ''\n\nprint(\"train.csv shape = \"+str(train_df.shape))\nprint(\"test.csv shape = \"+str(test_df.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"254a353cfa2b49f1b1e1d2fa8cf351e9d443bf24"},"cell_type":"code","source":"# unique ids - also includes \"new values\" \nids = train_df[\"Id\"]\nids.value_counts().shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d597dfb79a0cde478fde3eca188025c1d7f1e239"},"cell_type":"code","source":"num_classes = 5005","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a101e276c819543e4de0559e5f35f29528db8cc"},"cell_type":"code","source":"#image_size = 224\ntrain_data_gen = ImageDataGenerator(preprocess_input)\ntest_data_gen = ImageDataGenerator(preprocess_input)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"579267eaf5b52733a44295f1296ca92792d4773c"},"cell_type":"code","source":"train_generator = train_data_gen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=train_dir,\n    x_col='Image',\n    y_col='Id',\n    has_ext=True,\n    shuffle=True\n    )\n\ntest_generator = test_data_gen.flow_from_dataframe(\n    dataframe=test_df,\n    directory=test_dir,\n    x_col='Image',\n    y_col='Id',\n    has_ext=True,\n)\n\ntest_samples = test_generator.filenames","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"014e73e444929038b6f78575ead1268396a4618f"},"cell_type":"code","source":"# Build the model\n\nbase_model = ResNet50(include_top=False, pooling='avg', weights=resnet_weights_path)\n\nset_trainable = False\n\nfor layer in base_model.layers:\n    if layer.name == 'res5b_branch2a':\n        set_trainable = True\n    if set_trainable:\n        layer.trainable = True\n    else:\n        layer.trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef3384fea02e4a0be7d270d166d8dc99d0670913"},"cell_type":"code","source":"# ref - https://stats.stackexchange.com/questions/156471/imagenet-what-is-top-1-and-top-5-error-rate\ndef top_5_accuracy(y_true, y_pred):\n    return top_k_categorical_accuracy(y_true, y_pred, k=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7c14d4813c39ecfc82062a7529a59464860f6d8"},"cell_type":"code","source":"def build_model(base_model):\n    model = Sequential()\n    model.add(base_model)\n    #model.add(Flatten())\n    model.add(BatchNormalization(momentum=0.1, epsilon=1e-6))\n    model.add(Dropout(rate=0.5))\n    model.add(Dense(units=4096,activation='relu'))\n    model.add(Dropout(rate=0.5))\n    model.add(Dense(units=num_classes,activation='softmax'))\n    \n    return model\n\nmodel = build_model(base_model)\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5a5aa0e8ae73fc65bb7c054db18693cb30dae5a"},"cell_type":"code","source":"#adam = Adam()\nmodel.compile(optimizer='sgd',loss=['categorical_crossentropy'],metrics=[top_5_accuracy])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"355f2e9c51fbf03e76a62d1949b726fa8f75ee9d","_kg_hide-input":false,"_kg_hide-output":false},"cell_type":"code","source":"history = model.fit_generator(train_generator,steps_per_epoch=50,epochs=25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7cae0f04200d589fe680a5f3adb45fa4ec07688","scrolled":true},"cell_type":"code","source":"test_generator.reset() #?\n\npredictions = model.predict_generator(generator=test_generator,verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e1e436be093bd8c4c6f4f6286964d7a3e704851"},"cell_type":"code","source":"unique_labels = np.unique(ids)\n\nlabels_dict = dict()\nlabels_list = []\nfor i in range(len(unique_labels)):\n    labels_dict[unique_labels[i]] = i\n    labels_list.append(unique_labels[i])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"364aa4e23cf3ae2330e99aa475a5b74c64d64f7a"},"cell_type":"code","source":"best_th = 0.38\n\npreds_t = np.concatenate([np.zeros((predictions.shape[0],1))+best_th, predictions],axis=1)\nnp.save(\"preds.npy\",preds_t)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a2dc6539a5442e88f74f6dd0f4cf08003bd575a"},"cell_type":"code","source":"sample_df = pd.read_csv(\"../input/humpback-whale-identification/sample_submission.csv\")\nsample_list = list(sample_df.Image)\nlabels_list = [\"new_whale\"]+labels_list\npred_list = [[labels_list[i] for i in p.argsort()[-5:][::-1]] for p in preds_t]\npred_dic = dict((key, value) for (key, value) in zip(test_samples,pred_list))\npred_list_cor = [' '.join(pred_dic[id]) for id in sample_list]\ndf = pd.DataFrame({'Image':sample_list,'Id': pred_list_cor})\ndf.to_csv('submission.csv', header=True, index=False)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0b02981d21f14063607b4bc7c14af87b740c7bf"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}