{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport warnings\n# filter warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:27:53.285211Z","iopub.execute_input":"2022-05-26T14:27:53.285463Z","iopub.status.idle":"2022-05-26T14:27:53.305323Z","shell.execute_reply.started":"2022-05-26T14:27:53.285412Z","shell.execute_reply":"2022-05-26T14:27:53.304721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nprint(os.listdir(\"../input\"))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:27:53.306789Z","iopub.execute_input":"2022-05-26T14:27:53.307404Z","iopub.status.idle":"2022-05-26T14:27:53.312915Z","shell.execute_reply.started":"2022-05-26T14:27:53.307350Z","shell.execute_reply":"2022-05-26T14:27:53.312170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:27:53.314296Z","iopub.execute_input":"2022-05-26T14:27:53.314986Z","iopub.status.idle":"2022-05-26T14:27:53.372403Z","shell.execute_reply.started":"2022-05-26T14:27:53.314893Z","shell.execute_reply":"2022-05-26T14:27:53.371745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:27:53.373753Z","iopub.execute_input":"2022-05-26T14:27:53.373997Z","iopub.status.idle":"2022-05-26T14:27:53.396490Z","shell.execute_reply.started":"2022-05-26T14:27:53.373951Z","shell.execute_reply":"2022-05-26T14:27:53.395606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train[\"Id\"]\n# Drop the 'Id' column\nxtrain = train.drop(labels = [\"Id\"], axis = 1)\ny_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:27:53.398042Z","iopub.execute_input":"2022-05-26T14:27:53.398518Z","iopub.status.idle":"2022-05-26T14:27:53.410546Z","shell.execute_reply.started":"2022-05-26T14:27:53.398320Z","shell.execute_reply":"2022-05-26T14:27:53.409734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\ndef prepareImages(train, shape, path):\n    \n    x_train = np.zeros((shape, 100, 100, 3))\n    count = 0\n    \n    for fig in train['Image']:\n        \n        #load images into images of size 100x100x3\n        img = image.load_img(\"../input/\"+path+\"/\"+fig, target_size=(100, 100, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n\n        x_train[count] = x\n        if (count%500 == 0):\n            print(\"Processing image: \", count+1, \", \", fig)\n        count += 1\n    \n    return x_train\n\nx_train = prepareImages(train, train.shape[0], \"train\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:27:53.411791Z","iopub.execute_input":"2022-05-26T14:27:53.412261Z","iopub.status.idle":"2022-05-26T14:35:29.617897Z","shell.execute_reply.started":"2022-05-26T14:27:53.412214Z","shell.execute_reply":"2022-05-26T14:35:29.617026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = x_train / 255.0 \n# rescaling the dataset \n# dividing an image by 255 simply rescales the image from 0-255 to 0-1. \n# (Converting it to float from int makes computation convenient too) \nprint(\"xtrain shape: \",x_train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:29.619184Z","iopub.execute_input":"2022-05-26T14:35:29.619449Z","iopub.status.idle":"2022-05-26T14:35:34.044247Z","shell.execute_reply.started":"2022-05-26T14:35:29.619405Z","shell.execute_reply":"2022-05-26T14:35:34.043270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(x_train[0][:,:,0], cmap=\"gray\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:34.079691Z","iopub.execute_input":"2022-05-26T14:35:34.080119Z","iopub.status.idle":"2022-05-26T14:35:34.337171Z","shell.execute_reply.started":"2022-05-26T14:35:34.080069Z","shell.execute_reply":"2022-05-26T14:35:34.336418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ny_train = label_encoder.fit_transform(y_train)\ny_train[0:10] ","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:34.338633Z","iopub.execute_input":"2022-05-26T14:35:34.340238Z","iopub.status.idle":"2022-05-26T14:35:35.288450Z","shell.execute_reply.started":"2022-05-26T14:35:34.340185Z","shell.execute_reply":"2022-05-26T14:35:35.287620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:35.290390Z","iopub.execute_input":"2022-05-26T14:35:35.291213Z","iopub.status.idle":"2022-05-26T14:35:35.296981Z","shell.execute_reply.started":"2022-05-26T14:35:35.291160Z","shell.execute_reply":"2022-05-26T14:35:35.296193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils.np_utils import to_categorical\ny_train = to_categorical(y_train, num_classes = 5005)\ny_train","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:35.298755Z","iopub.execute_input":"2022-05-26T14:35:35.299675Z","iopub.status.idle":"2022-05-26T14:35:35.431435Z","shell.execute_reply.started":"2022-05-26T14:35:35.299624Z","shell.execute_reply":"2022-05-26T14:35:35.430654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:35.433117Z","iopub.execute_input":"2022-05-26T14:35:35.433913Z","iopub.status.idle":"2022-05-26T14:35:35.439635Z","shell.execute_reply.started":"2022-05-26T14:35:35.433861Z","shell.execute_reply":"2022-05-26T14:35:35.438713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils.np_utils import to_categorical # convert to one-hot-encoding\nfrom keras.models import Sequential # to create a cnn model\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D\nfrom tensorflow.keras.optimizers import SGD\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ReduceLROnPlateau\n\nmodel = Sequential()\n\nmodel.add(Conv2D(filters = 16, kernel_size = (3,3), padding = 'Same', activation = 'relu', input_shape = (100,100,3)))\nmodel.add(Conv2D(filters = 16, kernel_size = (3,3), padding = 'Same', activation = 'relu'))\nmodel.add(MaxPool2D(pool_size = (2,2)))\nmodel.add(BatchNormalization())\n\nmodel.add(Conv2D(filters = 32, kernel_size = (3,3), padding = 'Same', activation = 'relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = (3,3), padding = 'Same', activation = 'relu'))\nmodel.add(MaxPool2D(pool_size = (2,2), strides=(2,2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3), padding = 'Same', activation = 'relu'))\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3), padding = 'Same', activation = 'relu'))\nmodel.add(MaxPool2D(pool_size = (2,2), strides=(2,2)))\nmodel.add(BatchNormalization())\n\n# fully connected\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = 'relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dense(y_train.shape[1], activation = \"softmax\"))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:35.441261Z","iopub.execute_input":"2022-05-26T14:35:35.442127Z","iopub.status.idle":"2022-05-26T14:35:37.505926Z","shell.execute_reply.started":"2022-05-26T14:35:35.442075Z","shell.execute_reply":"2022-05-26T14:35:37.505042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install keras --upgrade ","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:37.507036Z","iopub.execute_input":"2022-05-26T14:35:37.507317Z","iopub.status.idle":"2022-05-26T14:35:37.514018Z","shell.execute_reply.started":"2022-05-26T14:35:37.507270Z","shell.execute_reply":"2022-05-26T14:35:37.513216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.models import load_model\nfrom keras.models import Sequential\n\nmodel.compile(loss='categorical_crossentropy', metrics=['accuracy'], optimizer=\"adam\")\n\nepochs = 50 \nbatch_size = 64\nprint(\"compiled\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:37.515965Z","iopub.execute_input":"2022-05-26T14:35:37.516233Z","iopub.status.idle":"2022-05-26T14:35:37.584293Z","shell.execute_reply.started":"2022-05-26T14:35:37.516177Z","shell.execute_reply":"2022-05-26T14:35:37.583569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n        featurewise_center=False,  # set input mean to 0 over the dataset\n        samplewise_center=False,  # set each sample mean to 0\n        featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        samplewise_std_normalization=False,  # divide each input by its std\n        zca_whitening=False,  # apply ZCA whitening\n        rotation_range=10,  # randomly rotate images in the range (degrees, 0 to 180)\n        zoom_range = 0.1, # Randomly zoom image \n        width_shift_range=0.1,  # randomly shift images horizontally (fraction of total width)\n        height_shift_range=0.1,  # randomly shift images vertically (fraction of total height)\n        horizontal_flip=False,  # randomly flip images horizontally\n        vertical_flip=False)  # randomly flip images vertically\n\n\ndatagen.fit(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:37.585846Z","iopub.execute_input":"2022-05-26T14:35:37.586625Z","iopub.status.idle":"2022-05-26T14:35:42.307220Z","shell.execute_reply.started":"2022-05-26T14:35:37.586575Z","shell.execute_reply":"2022-05-26T14:35:42.306353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x_train, y_train,epochs=50 ) ","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:42.311526Z","iopub.execute_input":"2022-05-26T14:35:42.312000Z","iopub.status.idle":"2022-05-26T14:48:29.418683Z","shell.execute_reply.started":"2022-05-26T14:35:42.311950Z","shell.execute_reply":"2022-05-26T14:48:29.417667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'], color='r', label=\"Train Loss\")\nplt.title(\"Train Loss\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:48:29.420034Z","iopub.execute_input":"2022-05-26T14:48:29.420382Z","iopub.status.idle":"2022-05-26T14:48:29.600384Z","shell.execute_reply.started":"2022-05-26T14:48:29.420304Z","shell.execute_reply":"2022-05-26T14:48:29.599625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['acc'], color='g', label=\"Train Accuracy\")\nplt.title(\"Train Accuracy\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:48:29.601959Z","iopub.execute_input":"2022-05-26T14:48:29.602680Z","iopub.status.idle":"2022-05-26T14:48:29.793019Z","shell.execute_reply.started":"2022-05-26T14:48:29.602629Z","shell.execute_reply":"2022-05-26T14:48:29.792314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train accuracy of the model: ',history.history['acc'][-1])\nprint('Train loss of the model: ',history.history['loss'][-1])","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:48:29.794129Z","iopub.execute_input":"2022-05-26T14:48:29.794540Z","iopub.status.idle":"2022-05-26T14:48:29.801733Z","shell.execute_reply.started":"2022-05-26T14:48:29.794478Z","shell.execute_reply":"2022-05-26T14:48:29.800981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = os.listdir(\"../input/test/\")\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:48:29.803599Z","iopub.execute_input":"2022-05-26T14:48:29.804760Z","iopub.status.idle":"2022-05-26T14:48:30.375092Z","shell.execute_reply.started":"2022-05-26T14:48:29.804709Z","shell.execute_reply":"2022-05-26T14:48:30.374274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['Image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['Id'] = ''\n\nX = prepareImages(test_df, test_df.shape[0], \"test\")\nX /= 255","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:48:30.376281Z","iopub.execute_input":"2022-05-26T14:48:30.376754Z","iopub.status.idle":"2022-05-26T14:50:55.506944Z","shell.execute_reply.started":"2022-05-26T14:48:30.376693Z","shell.execute_reply":"2022-05-26T14:50:55.502977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(np.array(X), verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:50:55.515571Z","iopub.execute_input":"2022-05-26T14:50:55.515914Z","iopub.status.idle":"2022-05-26T14:50:59.306365Z","shell.execute_reply.started":"2022-05-26T14:50:55.515815Z","shell.execute_reply":"2022-05-26T14:50:59.305579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:50:59.307652Z","iopub.execute_input":"2022-05-26T14:50:59.308160Z","iopub.status.idle":"2022-05-26T14:50:59.313855Z","shell.execute_reply.started":"2022-05-26T14:50:59.308101Z","shell.execute_reply":"2022-05-26T14:50:59.313046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, pred in enumerate(predictions):\n    test_df.loc[i, 'Id'] = ' '.join(label_encoder.inverse_transform(pred.argsort()[-5:][::-1]))\ntest_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:50:59.314953Z","iopub.execute_input":"2022-05-26T14:50:59.315431Z","iopub.status.idle":"2022-05-26T14:51:08.984390Z","shell.execute_reply.started":"2022-05-26T14:50:59.315381Z","shell.execute_reply":"2022-05-26T14:51:08.983635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:51:08.985998Z","iopub.execute_input":"2022-05-26T14:51:08.986795Z","iopub.status.idle":"2022-05-26T14:51:08.991336Z","shell.execute_reply.started":"2022-05-26T14:51:08.986742Z","shell.execute_reply":"2022-05-26T14:51:08.990288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\nimport pandas as pd\nimport numpy as np\nimport base64\n\n# function that takes in a dataframe and creates a text link to  \n# download it (will only work for files < 2MB or so)\ndef create_download_link(df, title = \"Download CSV file\", filename = \"submission.csv\"):  \n    csv = df.to_csv(index=False)\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\n# create a random sample dataframe\ndf = pd.DataFrame(np.random.randn(50, 4), columns=list('ABCD'))\n\n# create a link to download the dataframe\ncreate_download_link(test_df)\n","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:51:08.992863Z","iopub.execute_input":"2022-05-26T14:51:08.993288Z","iopub.status.idle":"2022-05-26T14:51:09.072517Z","shell.execute_reply.started":"2022-05-26T14:51:08.993110Z","shell.execute_reply":"2022-05-26T14:51:09.071582Z"},"trusted":true},"execution_count":null,"outputs":[]}]}