{"cells":[{"metadata":{"_uuid":"da035fe58e548e8b1b7e8e89725b9e6bc745aa7b"},"cell_type":"markdown","source":"# Humpback Whale Identification - CNN with Keras\nThis kernel is based on [Anezka Kolaceke](https://www.kaggle.com/anezka)'s awesome work: [CNN with Keras for Humpback Whale ID](https://www.kaggle.com/anezka/cnn-with-keras-for-humpback-whale-id)"},{"metadata":{"trusted":true,"_uuid":"0d9c73ad23e6c2eae3028255ee00c3254fe66401"},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\nimport gc\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.models import Model\n\nimport keras.backend as K\nfrom keras.models import Sequential\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)\n\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cea35de3530cc898be5b85063b84e875401d092"},"cell_type":"code","source":"os.listdir(\"../input/\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46a8839e13a14eb8d16ea6823de9927ea63d5001"},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntrain_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SIZE = 100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f46b24dbba74f22833cac6140e60348b15a8e047"},"cell_type":"code","source":"def prepareImages(data, m, dataset):\n    print(\"Preparing images\")\n    X_train = np.zeros((m, SIZE, SIZE,3))\n    count = 0\n    threshold = 100\n    for fig in data['Image']:\n\n        img = image.load_img(\"../input/\"+dataset+\"/\"+fig, target_size=(SIZE, SIZE,3))\n        \n        #img_data = np.asarray(img)\n        #ret, img_thresh = cv2.threshold(img_data, threshold, 255, cv2.THRESH_BINARY)\n        #kernel = np.ones((2,2),np.uint8)\n        \n        #x = cv2.morphologyEx(img_thresh, cv2.MORPH_BLACKHAT, kernel)\n        \n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        \n        X_train[count] = x\n        if (count%1000 == 0):\n            print(\"Processing image: \", count+1, \", \", fig)\n        count += 1\n    \n    return X_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6587a101b58af064af0f9c60a1070c6c8f52d45f"},"cell_type":"code","source":"def prepare_labels(y):\n    for i in range(len(y)):\n        if y[i] == \"new_whale\":\n            y[i] = '1'\n        else:\n            y[i] = '0'\n    values = np.array(y)\n    \n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    # print(integer_encoded)\n\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    # print(onehot_encoded)\n\n    y = onehot_encoded\n    # print(y.shape)\n    return y, label_encoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4afe4128a0cd6859848c8a80686208082d647c39"},"cell_type":"code","source":"X = prepareImages(train_df, train_df.shape[0], \"train\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#import tensorflow as tf\n#from keras import backend\n\n#rgb = tf.image.grayscale_to_rgb(X[1])\n#array = backend.get_session().run(rgb)\n#plt.imshow(array)\n#plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"675924f8863aef27cf90dc668e0a68cd609dfc1c"},"cell_type":"code","source":"y, label_encoder = prepare_labels(train_df['Id'])\ntrain_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"14d243b19023e830b636bea16679e13bc40deae6"},"cell_type":"code","source":"y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7af799d186a1b97b6aa325d7d576a1fb55a6c5d"},"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(32, (7, 7), strides = (1, 1), name = 'conv0', input_shape = (SIZE, SIZE, 3)))\n\nmodel.add(BatchNormalization(axis = 3, name = 'bn0'))\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2), name='max_pool'))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1), name=\"conv1\"))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3), name='avg_pool'))\n\nmodel.add(Flatten())\nmodel.add(Dense(500, activation=\"relu\", name='rl'))\nmodel.add(Dropout(0.8))\nmodel.add(Dense(2, activation='softmax', name='sm'))\n\nmodel.compile(loss='binary_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"169f45e150c3a584e0f655a8eda523e0675da63a"},"cell_type":"code","source":"history = model.fit(X, y, epochs=15, batch_size=100, verbose=1)\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bca48a1d0963cbf70685b75431435cef9499895"},"cell_type":"code","source":"plt.plot(history.history['acc'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"debe961c93b72bef151d9aad3ca2cb500ee00aaa"},"cell_type":"code","source":"test = os.listdir(\"../input/test/\")\nprint(len(test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72ed8198f519f7b1ae3efbc688933c78d8cdd0e4"},"cell_type":"code","source":"col = ['Image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['Id'] = ''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52262195fc0b8755cff78bf8c98e6116d50f79af"},"cell_type":"code","source":"X = prepareImages(test_df, test_df.shape[0], \"test\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#import tensorflow as tf\n#from keras import backend\n\n#rgb = tf.image.grayscale_to_rgb(X[2])\n#array = backend.get_session().run(rgb)\n#plt.imshow(array)\n#plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88c8d8ff98fbdb1df4218abb6bd51889e855a6fb"},"cell_type":"code","source":"predictions = model.predict(np.array(X), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66f0bdde31b8c7847916268aa82d9a1bdc9c0658"},"cell_type":"code","source":"print(predictions)\nfor i, pred in enumerate(predictions):\n    if pred[0] > pred[1]:\n        ans = 'w_f48451c'\n    else:\n        ans = 'new_whale'\n    test_df.loc[i, 'Id'] = ans","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"09d7c1eb9b554e4e580b0c3c7eb609c15636892d"},"cell_type":"code","source":"test_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}