{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import cv2\nimport pandas as pd\nimport numpy as np\nimport os\nimport json\nfrom tqdm import tqdm, tqdm_notebook\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Flatten, Activation\nfrom keras.layers import Dropout\nfrom keras.layers.convolutional import Conv2D, MaxPooling2D\nfrom keras.utils import np_utils\nfrom keras.optimizers import SGD\nfrom keras.preprocessing import image\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import VGG16\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1fe6abb19a6cf918f1075d7d4fba0b3e76313679"},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ad053b05f5f45513d91899c14930fd7a8383d8f"},"cell_type":"code","source":"train_df['category_id'] = train_df['category_id'].astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da84678b09bad4f39389b677dd1c8ab3a074d001"},"cell_type":"code","source":"batch_size=32\nimg_size = 32\nnb_epochs = 10","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f80f333a5d4f3254f4c776a554fb25c62bf01704"},"cell_type":"code","source":"%%time\ntrain_datagen = ImageDataGenerator(rescale=1./255, validation_split=0.25)\ntrain_generator = train_datagen.flow_from_dataframe(\n        dataframe = train_df,        \n        directory = '../input/train_images',\n        x_col = 'file_name', y_col = 'category_id',\n        target_size=(img_size,img_size),\n        batch_size=batch_size,\n        class_mode='categorical',\n        subset='training')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f18e35b317106c34b7f31bed2343c98d3f798cd"},"cell_type":"code","source":"%%time\nvalidation_generator  = train_datagen.flow_from_dataframe(\n        dataframe = train_df,        \n        directory = '../input/train_images',\n        x_col = 'file_name', y_col = 'category_id',\n        target_size=(img_size,img_size),\n        batch_size=batch_size,\n        class_mode='categorical',\n        subset='validation')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a1210dbb8bb7bdaea2ea2d12691d476aa2dc436"},"cell_type":"code","source":"set(train_generator.class_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bf01f22eb05a1572800fa75f0072c7ce13f5bd4"},"cell_type":"code","source":"nb_classes = 14","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b102dcac7bc4709e10568bfcd71f201d6825a728"},"cell_type":"code","source":"vgg16_net = VGG16(weights='imagenet', \n                  include_top=False, \n                  input_shape=(img_size, img_size, 3))\nvgg16_net.trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d28e5d6167963f6a0f5e59ee060bb1f5fc04a28"},"cell_type":"code","source":"model = Sequential()\nmodel.add(vgg16_net)\nmodel.add(Flatten())\nmodel.add(Dense(256))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(nb_classes, activation='softmax'))\n\nopt = SGD(lr=0.01, decay=1e-6, momentum=0.9, nesterov=True)\nmodel.compile(loss='categorical_crossentropy',\n              optimizer=opt,\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"349067b6716ada48e8e66d3cd6b6b144e2e4ee92"},"cell_type":"code","source":"%%time\n# Train model\nhistory = model.fit_generator(\n            train_generator,\n#             steps_per_epoch = train_generator.samples // batch_size,\n            steps_per_epoch = 100,\n            validation_data = validation_generator, \n#             validation_steps = validation_generator.samples // batch_size,\n            validation_steps = 50,\n            epochs = nb_epochs,\n            verbose=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c39d421999bbb6c7e669fd8fa54a22863756af4b"},"cell_type":"code","source":"with open('history.json', 'w') as f:\n    json.dump(history.history, f)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['acc', 'val_acc']].plot()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af48ebd16dc02e9b0312c3df7540fabeb355ae55"},"cell_type":"markdown","source":"### Prediction"},{"metadata":{"trusted":true,"_uuid":"f998a911ad3510c76b87fb443e8373a33c6ea93f"},"cell_type":"code","source":"test_df = pd.read_csv('../input/test.csv')\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"86bdd65b8a85b3af8927a790da5ecfbc5b4129fd"},"cell_type":"code","source":"%%time\ntest_datagen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_datagen.flow_from_dataframe(\n        dataframe = test_df,        \n        directory = '../input/test_images',\n        x_col = 'file_name', y_col = None,\n        target_size = (img_size,img_size),\n        batch_size = 1,\n        shuffle = False,\n        class_mode = None\n        )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19da81c8f26570fea05d5df64c27e04b31544ecf"},"cell_type":"code","source":"%%time\ntest_generator.reset()\npredict = model.predict_generator(test_generator, steps = len(test_generator.filenames))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"747435fd7e8c4ab822cb903eb6c832443ff1045b"},"cell_type":"code","source":"len(predict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d436bcfc8166ca849f764736a519be3c16fc8c45"},"cell_type":"code","source":"predicted_class_indices=np.argmax(predict,axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6c94ec399dcee4071703cfc848c697f9a717ac7"},"cell_type":"code","source":"labels = (train_generator.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\npredictions = [labels[k] for k in predicted_class_indices]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca63e7ee9e37769033bf9b9c9066345f39359206"},"cell_type":"code","source":"sam_sub_df = pd.read_csv('../input/sample_submission.csv')\nprint(sam_sub_df.shape)\nsam_sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4e154140fb5f864c1c0d7daef471f2378c2474c0"},"cell_type":"code","source":"filenames=test_generator.filenames\nresults=pd.DataFrame({\"Id\":filenames,\n                      \"Predicted\":predictions})\nresults['Id'] = results['Id'].map(lambda x: str(x)[:-4])\nresults.to_csv(\"results.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc031fa6af4fc8f089910ae36cb25232f3f2640d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}