{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3586748f6073f4777fa2d34350659f4ab44ab4e"},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.pyplot import imshow\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom keras.models import Model\n\nimport keras.backend as K\nfrom keras.models import Sequential\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97bd94ad3a19b34a573df38ed3ae17ac9c961925"},"cell_type":"code","source":"Train = '../input/train/'\nTest = '../input/test/'\nLabels = '../input/train.csv'\nSample = '../input/sample_submission.csv'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0e657931ca7c0f8969ebe94ed3df9579809ebd6b"},"cell_type":"markdown","source":"## Labels\nLets check what different types of labels do we have"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df1 = pd.read_csv(Labels)\ndf1.Id.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b517f20263c25f5644768c176fc540cdbdac239"},"cell_type":"code","source":"train_names = list(f[:36] for f in os.listdir(Train))\ntest_names = list(f[:36] for f in os.listdir(Test))\nprint('Training Data Length - {}'.format(len(train_names)))\nprint('Test Data Length - {}'.format(len(test_names)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"52c3bb3dde341f70016f32dabc5a1fda189e5904"},"cell_type":"markdown","source":"## Training Images\nNow let's check some images"},{"metadata":{"trusted":true,"_uuid":"f0440876a1c9785857f03c2b88336a642984e817"},"cell_type":"code","source":"i=34\n# We can check what is the Id of the whale and mark it while plotting the Image\ntitle = df1[df1['Image']==train_names[i]]\nimg = mpimg.imread(os.path.join(Train,train_names[i]))\nlabel =  'Whale Id -' + title['Id'].to_string().split('   ')[1]\nplt.title(label)\nplt.imshow(img)\nprint(title)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3918143d8a280296ca006cc48d2ecf24d1b78394"},"cell_type":"code","source":"def prepareImages(data, m, dataset):\n    print(\"Preparing images\")\n    X_train = np.zeros((m, 100, 100, 3))\n    count = 0\n    \n    for fig in data['Image']:\n        #load images into images of size 100x100x3\n        img = image.load_img(\"../input/\"+dataset+\"/\"+fig, target_size=(100, 100, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n\n        X_train[count] = x\n        count += 1\n    \n    return X_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7ae3b375abf8ee041d22bdcde5837936e9273d7"},"cell_type":"code","source":"def prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    # print(integer_encoded)\n\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    # print(onehot_encoded)\n\n    y = onehot_encoded\n    # print(y.shape)\n    return y, label_encoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23bde5e64329abffe134189954a891ee0e32e4be"},"cell_type":"code","source":"X = prepareImages(df1, df1.shape[0], \"train\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8c718a66c24d531ff8e5b90a9e13afba1a9ef6f"},"cell_type":"code","source":"y, label_encoder = prepare_labels(df1['Id'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a8de319fedeae351970d67b8d99029000305466c","scrolled":true},"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(32, (7, 7), strides = (1, 1), name = 'conv0', input_shape = (100, 100, 3)))\nmodel.add(BatchNormalization(axis = 3, name = 'bn0'))\nmodel.add(Activation('relu'))\n\nmodel.add(Conv2D(64, (5, 5), strides = (1, 1), name = 'conv1'))\nmodel.add(BatchNormalization(axis = 3, name = 'bn1'))\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2), name='max_pool'))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1), name=\"conv2\"))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3), name='avg_pool'))\n\nmodel.add(Flatten())\nmodel.add(Dense(500, activation=\"relu\", name='rl'))\nmodel.add(Dropout(0.8))\nmodel.add(Dense(y.shape[1], activation='softmax', name='sm'))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33c066b81c4ae9f4af9c2416e2db829cdf03f63d"},"cell_type":"code","source":"from keras.callbacks import EarlyStopping\n# define early stopping callback\nearlystop = EarlyStopping(monitor='val_acc', min_delta=0.0001, patience=5, verbose=1, mode='auto')\ncallbacks = [earlystop]\nhistory = model.fit(X, y, epochs=100, batch_size=150, verbose=1,validation_split=0.20,callbacks=callbacks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33afc38acde0f10188f09b10563d5dafaf9d7a00","scrolled":true},"cell_type":"code","source":"col = ['Image']\ntest_df = pd.DataFrame(test_names, columns=col)\ntest_df['Id'] = ''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af49dd0c4c01668fe65a7bb24cb258c9e71efc90"},"cell_type":"code","source":"X = prepareImages(test_df, test_df.shape[0], \"test\")\nX /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"047f743fe079fb579dff36781901cd975fa14e98"},"cell_type":"code","source":"predictions = model.predict(np.array(X), verbose=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99d089c099f550a3321bb0753146193f35131457"},"cell_type":"code","source":"for i, pred in enumerate(predictions):\n    test_df.loc[i, 'Id'] = ' '.join(label_encoder.inverse_transform(pred.argsort()[-5:][::-1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1f4f4bf1976017e71c7afbc80185c32ba1fbdda"},"cell_type":"code","source":"test_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d02d562faff6d13b8a7730e92374e1cfca03e805"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}