{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport gc\nprint(os.listdir(\"../input\"))\nimport numpy as np \nimport pandas as pd\nimport time\ngc.collect()\n# Any results you write to the current directory are saved as output.\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"14b73bb9cb63c8229b5a2fc5b86c2e12eb363cba"},"cell_type":"code","source":"!ls ../input/resnet50/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e00ac1e5e481531a6b438f697dc86d96877c7a9"},"cell_type":"code","source":"!pwd\n!cp ../input/resnet50/* /tmp/.keras/models/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"366143a8d1a42bc66ceb006f2a98a09bf7133632"},"cell_type":"code","source":"!ls ~/.keras/models/","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/airbus-ship-detection/train_ship_segmentations_v2.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"efd2ea308ca04f91749ada2dbf67f62b447b5d18","scrolled":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b13464005e7ab0af595b62cc1162d19015dff6e"},"cell_type":"code","source":"list(train.columns.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf7fd68627ba9fb3e48f7e1b80a50cf8e5eba8d0"},"cell_type":"code","source":"train['exist_ship'] = train['EncodedPixels'].fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54665ce582a6c29499e5f81ab7547f723e95eed7"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06e55fcf5061e5d02dd81dffb0a75f63f6359b28","scrolled":true},"cell_type":"code","source":"train['exist_ship'] != 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0938180caa3d59ebdf3290fda7673502750fcac"},"cell_type":"code","source":"train.loc[train['exist_ship'] != 0 , 'exist_ship'] = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"759b355ad0730b12f4f097dbacae23677974a4e6"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f1f7626b0c64fac65b29609bb6adf7e32015533"},"cell_type":"code","source":"del train['EncodedPixels']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2fd2bf07e24695fd4cef7b4351c3def2ac213c9"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"617d3eb4b674814c1c1102c27c79cff363278bf4"},"cell_type":"code","source":"print(len(train['ImageId']))\nprint(train['ImageId'].value_counts().shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7943ba7be3b0bc580a4fe131416f42ff58c875fd"},"cell_type":"code","source":"train_gp = train.groupby(['ImageId']).sum().reset_index()\ntrain_gp.loc[train_gp['exist_ship']>0,'exist_ship']=1\n\ntrain_sample = train_gp.sample(5000)\ntest_sample = train_gp.sample(1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"399ad229cab10b52df239c4246fad61d0cef57e4"},"cell_type":"code","source":"print(train_gp['exist_ship'].value_counts())\nprint(train_sample['exist_ship'].value_counts())\nprint(test_sample['exist_ship'].value_counts())\nprint (train_sample.shape)\nprint (test_sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc1e684fa398a89fa776099349e91a43c2a81324"},"cell_type":"code","source":"from keras.utils import np_utils\nimport numpy as np\nfrom glob import glob\n\nTrain_path = '../input/airbus-ship-detection/train_v2/'\nTest_path = '../input/airbus-ship-detection/test_v2/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5327125d83fe2fef0f30233ef5be4df5da17e96e"},"cell_type":"code","source":"%%time\ntraining_img_data = []\ntarget_data = []\nfrom PIL import Image\ndata = np.empty((len(train_sample['ImageId']),256, 256,3), dtype=np.uint8)\ndata_target = np.empty((len(train_sample['ImageId'])), dtype=np.uint8)\nimage_name_list = os.listdir(Train_path)\nindex = 0\nfor image_name in image_name_list:\n    if image_name in list(train_sample['ImageId']):\n        imageA = Image.open(Train_path+image_name).resize((256,256)).convert('RGB')\n        data[index]=imageA\n        data_target[index]=train_sample[train_gp['ImageId'].str.contains(image_name)]['exist_ship'].iloc[0]\n        index+=1\n        \nprint(data.shape)\nprint(data_target.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b54585ce32988fbe39b0a7dc2c8939ad2662098d"},"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\ntargets =data_target.reshape(len(data_target),-1)\nenc = OneHotEncoder()\nenc.fit(targets)\ntargets = enc.transform(targets).toarray()\nprint(targets.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9beb014ffb2c155df1990cc756fbccfbd3d6cca8"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_val, y_train, y_val = train_test_split(data,targets, test_size = 0.2)\nx_train.shape, x_val.shape, y_train.shape, y_val.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05f48ae3370b6deb0ed8455bf9065cebad138ef0"},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nimg_gen = ImageDataGenerator(\n    rescale=1./255,\n    zca_whitening = False,\n    rotation_range = 90,\n    width_shift_range = 0.2,\n    height_shift_range = 0.2,\n    brightness_range = [0.5, 1.5],\n    shear_range = 0.2,\n    zoom_range = 0.2,\n    horizontal_flip = True,\n    vertical_flip = True\n    \n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bef4b7dbc4300aa68b120657bd723a4a95aa5a47"},"cell_type":"code","source":"#https://www.kaggle.com/yassinealouini/f2-score-per-epoch-in-keras\n\nimport numpy as np \nimport pandas as pd \nfrom keras.callbacks import Callback\nfrom sklearn.metrics import fbeta_score\nfrom keras.layers import Dense\nfrom keras.models import Sequential\nfrom keras.utils.test_utils import get_test_data\n\n\n\n\"\"\" F2 metric implementation for Keras models. Inspired from this Medium\narticle: https://medium.com/@thongonary/how-to-compute-f1-score-for-each-epoch-in-keras-a1acd17715a2\nBefore we start, you might ask: this is a classic metric, isn't it already \nimplemented in Keras? \nThe answer is: it used to be. It has been removed since. Why?\nWell, since metrics are computed per batch, this metric was confusing \n(should be computed globally over all the samples rather than over a mini-batch).\nFor more details, check this: https://github.com/keras-team/keras/issues/5794.\nIn this short code example, the F2 metric will only be called at the end of \neach epoch making it more useful (and correct).\n\"\"\"\n\n# Notice that since this competition has an unbalanced positive class\n# (fewer ), a beta of 2 is used (thus the F2 score). This favors recall\n# (i.e. capacity of the network to find positive classes). \n\n# Some default constants\n\nSTART = 0.5\nEND = 0.95\nSTEP = 0.05\nN_STEPS = int((END - START) / STEP) + 2\nDEFAULT_THRESHOLDS = np.linspace(START, END, N_STEPS)\nDEFAULT_BETA = 1\nDEFAULT_LOGS = {}\nFBETA_METRIC_NAME = \"val_fbeta\"\n\n# Some unit test constants\ninput_dim = 2\nnum_hidden = 4\nnum_classes = 2\nbatch_size = 5\ntrain_samples = 20\ntest_samples = 20\nSEED = 42\nTEST_BETA = 2\nEPOCHS = 5\n\n\n\n\n# Notice that this callback only works with Keras 2.0.0\n\n\nclass FBetaMetricCallback(Callback):\n\n    def __init__(self, beta=DEFAULT_BETA, thresholds=DEFAULT_THRESHOLDS):\n        self.beta = beta\n        self.thresholds = thresholds\n        # Will be initialized when the training starts\n        self.val_fbeta = None\n\n    def on_train_begin(self, logs=DEFAULT_LOGS):\n        \"\"\" This is where the validation Fbeta\n        validation scores will be saved during training: one value per\n        epoch.\n        \"\"\"\n        self.val_fbeta = []\n\n    def _score_per_threshold(self, predictions, targets, threshold):\n        \"\"\" Compute the Fbeta score per threshold.\n        \"\"\"\n        # Notice that here I am using the sklearn fbeta_score function.\n        # You can read more about it here:\n        # http://scikit-learn.org/stable/modules/generated/sklearn.metrics.fbeta_score.html\n        thresholded_predictions = (predictions > threshold).astype(int)\n        return fbeta_score(targets, thresholded_predictions, beta=self.beta, average='micro')\n\n    def on_epoch_end(self, epoch, logs=DEFAULT_LOGS):\n        val_predictions = self.model.predict(self.validation_data[0])\n        val_targets = self.validation_data[1]\n        _val_fbeta = np.mean([self._score_per_threshold(val_predictions,\n                                                        val_targets, threshold)\n                              for threshold in self.thresholds])\n        self.val_fbeta.append(_val_fbeta)\n        print(\"Current F{} metric is: {}\".format(str(self.beta), str(_val_fbeta)))\n        return\n\n    def on_train_end(self, logs=DEFAULT_LOGS):\n        \"\"\" Assign the validation Fbeta computed metric to the History object.\n        \"\"\"\n        self.model.history.history[FBETA_METRIC_NAME] = self.val_fbeta\n\n\"\"\"\nHere is how to use this metric: \nCreate a model and add the FBetaMetricCallback callback (with beta set to 2).\nf2_metric_callback = FBetaMetricCallback(beta=2)\ncallbacks = [f2_metric_callback]\nhistory = model.fit(X_train, Y_train, validation_data=(X_val, Y_val),\n                    nb_epoch=10, batch_size=64, callbacks=callbacks)\nprint(history.history.val_fbeta)\n\"\"\"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5511ab3de12eee3c4adc5895217eafd5c6f726d8"},"cell_type":"code","source":"from keras.applications.resnet50 import ResNet50\nimg_width, img_height = 256, 256\nmodel = ResNet50(weights = 'imagenet', include_top=False, input_shape = (img_width, img_height, 3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd45c25406eac0035d0f71b16931a844aef3ee85"},"cell_type":"code","source":"from keras.layers import Dropout, Flatten, Dense, GlobalAveragePooling2D\nfrom keras.models import Sequential, Model \nfor layer in model.layers:\n    layer.trainable = False\n\nx = model.output\nx = Flatten()(x)\nx = Dense(1024, activation=\"relu\")(x)\nx = Dropout(0.5)(x)\nx = Dense(1024, activation=\"relu\")(x)\npredictions = Dense(2, activation=\"softmax\")(x)\n\n# creating the final model \nmodel_final = Model(input = model.input, output = predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02830863070e114fb64ac499f9b86ecdbf45fbdf"},"cell_type":"code","source":"from keras import optimizers\nepochs = 10\nlrate = 0.001\n\nmodel_final.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nmodel_final.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0c562ddb9491247d763536e680aa834c172da33","scrolled":true},"cell_type":"code","source":"fbeta_metric_callback = FBetaMetricCallback(beta=2)\nhistory = model_final.fit_generator(img_gen.flow(x_train, y_train, batch_size = 16),steps_per_epoch = len(x_train)/16, validation_data = (x_val,y_val), epochs = epochs, callbacks=[fbeta_metric_callback] )\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"503c038e54505c1507607f75c6da4ccf2d085731"},"cell_type":"code","source":"#gc.collect()\nhistory.history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"124c95be2ab468b598814c68252e6d2164a88065"},"cell_type":"code","source":"from matplotlib import pyplot\n\npyplot.plot(history.history['acc'])\npyplot.plot(history.history['val_acc'])\npyplot.plot(history.history['val_fbeta'])\npyplot.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54862d7af690ff3c5a2ee7cba80824597b88e582"},"cell_type":"code","source":"pyplot.plot(history.history['loss'])\npyplot.plot(history.history['val_loss'])\npyplot.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"141bb153bdffad81a41991544d762a1b30e1f76f"},"cell_type":"code","source":"train_predict_sample = train_gp.sample(2000)\nprint(train_predict_sample['exist_ship'].value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"496ef6af3439f2b0567537f7613812e918dee4d8"},"cell_type":"code","source":"\nimage_test_name_list = os.listdir(Train_path)\nnumber_sample = 20000\ndata_test = np.empty((number_sample,256, 256,3), dtype=np.uint8)\ntest_name = []\nindex = 0\nfor image_name in image_test_name_list:\n    imageA = Image.open(Train_path+image_name).resize((256,256)).convert('RGB')\n    test_name.append(image_name)\n    data_test[index]=imageA\n    index+=1\n    if number_sample == index:\n        break\nprint (data_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c964220cf41982e7eb8e9e71ddb8ef7c35370e01"},"cell_type":"code","source":"len(data_test)\nlen(test_name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf7b732e32a197fc22140a6aed0972082224d07c"},"cell_type":"code","source":"gc.collect()\nresult = model_final.predict(data_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94a7d29b438441d2e2ddaddbe00fd7d1bf353331"},"cell_type":"code","source":"!rm -f Have_ship_or_not.csv\n!ls -alrt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a5f58262d8fc0a5a07a27134129b24497be7b5a"},"cell_type":"code","source":"result_list={\n    \"ImageId\": test_name,\n    \"Have_ship\":np.argmax(result,axis=1)\n}\nresult_pd = pd.DataFrame(result_list)\nresult_pd.to_csv('Have_ship_or_not.csv',index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce9c0f9c7d7bcef1b95f95928d5cc88721a96eb2"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}