{"cells":[{"metadata":{},"cell_type":"markdown","source":"# First Sight of Dataset & GPU Check"},{"metadata":{"_cell_guid":"f67b9393-8ea1-4e23-b856-2ce149cfe421","_execution_state":"idle","_uuid":"72334cb006d02a4bcfc2a2fe622524eba824c6f8","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns #visualization\n%matplotlib inline\n\nnp.random.seed(42)\n\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nimport itertools\n\nfrom keras.utils.np_utils import to_categorical\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.optimizers import RMSprop, Adam\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ReduceLROnPlateau, EarlyStopping\n\nsns.set(style='white', context='notebook', palette='pastel')\n\nIMSIZE = 224\nBATCH_SIZE = 16","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nfrom keras.backend.tensorflow_backend import set_session\n\n#os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"-1\" #hide GPUs, to apply changes restart Kernel\n\n# do not change the following lines\nconfig = tf.ConfigProto()\nconfig.gpu_options.allow_growth = True #allows dynamic memory alloc growth on GPUs\nsess = tf.Session(config=config)\nset_session(sess) #Keras always uses global TF-Session \"sess\", so this line is not obligatory\n# do not change the above lines","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.python.client import device_lib\n\ndef get_available_gpus():\n    local_device_protos = device_lib.list_local_devices()\n    print(local_device_protos)\n    return [x.name for x in local_device_protos if x.device_type == \"GPU\"]\n\nnum_gpu = len(get_available_gpus())\nprint(\"Number of available GPUs: {}\".format(num_gpu))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5e51d00e-62fd-4141-bf73-50ac4f2da7d0","_execution_state":"idle","_uuid":"84bbd5ab8d7895bd430d5ecfe2f7ddf77baa7b74","trusted":true},"cell_type":"code","source":"print(os.listdir(\"../input/dat18seefood\"))\n\nbase_path = \"../input/dat18seefood/\"\ntrain_path = \"../input/dat18seefood/train/\"\ntest_path = \"../input/dat18seefood/test/\"\nextension = \".jpg\"\n\ntrain_ids_all = pd.read_csv(base_path+\"train.csv\")\ntest_ids_all = pd.read_csv(base_path+\"test.csv\")\n\nlabelnames = pd.read_csv(base_path+\"labelnames.csv\")\n\ndef plot_diag_hist(dataframe, title='NoTitle'):\n    f, ax = plt.subplots(figsize=(15, 4))\n    ax = sns.countplot(x=\"label\", data=dataframe, palette=\"GnBu_d\")\n    sns.despine()\n    plt.title(title)\n    plt.show()\n\nplot_diag_hist(train_ids_all, title=\"Labels Training Data\")\n\nprint(\"Shape of Training Data: {}\".format(train_ids_all.shape))\nprint(\"Shape of Test Data: {}\\n\".format(test_ids_all.shape))\n\ndef get_full_path_train(idcode):\n    return \"{}{}{}\".format(train_path,idcode,extension)\n\ndef get_full_path_test(idcode):\n    return \"{}{}{}\".format(test_path,idcode,extension)\n\n\ntrain_ids_all[\"path\"] = train_ids_all[\"id_code\"].apply(lambda x: get_full_path_train(x))\ntest_ids_all[\"path\"] = test_ids_all[\"id_code\"].apply(lambda x: get_full_path_test(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labelnames.at[0,\"labelname\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ids_all.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ids_all.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\n\ndef load_image(image_path):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (IMSIZE, IMSIZE))\n    return img\n\ndef load_images_as_tensor(image_path, dtype=np.uint8):\n    data = load_image(image_path).reshape((IMSIZE*IMSIZE,1))\n    return data.flatten()\n\ndef show_image(image_path, figsize=None, title=None):\n    image = load_image(image_path)\n    if figsize is not None:\n        fig = plt.figure(figsize=figsize)\n    if image.ndim == 1:\n        plt.imshow(np.reshape(image, (IMSIZE,-1)),cmap='gray')\n    elif image.ndim == 2:\n        plt.imshow(image,cmap='gray')\n    elif image.ndim == 3:\n        if image.shape[2] == 1:\n            image = image[:,:,0]\n            plt.imshow(image,cmap='gray')\n        elif image.shape[2] == 3:\n            plt.imshow(image)\n        else:\n            print(\"Invalid image dimension\")\n    if title is not None:\n        plt.title(title)\n        \ndef show_image_tensor(image, figsize=None, title=None):\n    if figsize is not None:\n        fig = plt.figure(figsize=figsize)\n    if image.ndim == 1:\n        plt.imshow(np.reshape(image, (IMSIZE,-1)),cmap='gray')\n    elif image.ndim == 2:\n        plt.imshow(image,cmap='gray')\n    elif image.ndim == 3:\n        if image.shape[2] == 1:\n            image = image[:,:,0]\n            plt.imshow(image,cmap='gray')\n        elif image.shape[2] == 3:\n            plt.imshow(image)\n        else:\n            print(\"Invalid image dimension\")\n    if title is not None:\n        plt.title(title)\n        \ndef show_Nimages(image_filenames, classifications, scale=1):\n    N=len(image_filenames)\n    fig = plt.figure(figsize=(25/scale, 16/scale))\n    for i in range(N):\n        ax = fig.add_subplot(1, N, i + 1, xticks=[], yticks=[])\n        show_image(image_filenames[i], title=\"C:{}\".format(classifications[i]))\n        \ndef show_Nrandomimages(N=10):\n    indices = (np.random.rand(N)*train_ids_all.shape[0]).astype(int)\n    show_Nimages(train_ids_all[\"path\"][indices].values, train_ids_all[\"label\"][indices].values)\n    \ndef show_Nimages_of_class(classification=0, N=10):\n    print(\"{} images of class {} = {}\".format(N, classification, labelnames.at[classification,\"labelname\"]))\n    indices = train_ids_all[train_ids_all[\"label\"] == classification].sample(N).index\n    show_Nimages(train_ids_all[\"path\"][indices].values, train_ids_all[\"label\"][indices].values)\n    \ndef show_Nerrorimages(imgs, pred, true, delta_prob=[], scale=1):\n    N=len(imgs)\n    fig = plt.figure(figsize=(25/scale, 16/scale))\n    for i in range(N):\n        ax = fig.add_subplot(1, N, i + 1, xticks=[], yticks=[])\n        if (delta_prob!=[]):\n            show_image_tensor(imgs[i], title=\"P:{} T:{} d:{:.2f}\".format(pred[i], true[i], delta_prob[i]))\n        else:\n            show_image_tensor(imgs[i], title=\"P:{} T:{}\".format(pred[i], true[i]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_index = 2477\nshow_image(train_ids_all[\"path\"][test_index], title=\"Class = {}\".format(train_ids_all[\"label\"][test_index]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_Nrandomimages(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_Nimages_of_class(classification=66)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Split Training and Validation Set"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ids_all[:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#train_ids_all_working, train_ids_all_notused = train_test_split(train_ids_all, test_size=0.75, random_state=42, stratify=train_ids_all[['label']])\ntrain_ids_all_working = train_ids_all","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, validation_df = train_test_split(train_ids_all_working, test_size=0.1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Data Generator (From Memory & From Hard Disk)"},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_y = train_df[\"label\"].values\nvalidation_y = validation_df[\"label\"].values\n\n# Encode labels to one hot vectors\nprint(train_y.shape)\ntrain_y_cat = to_categorical(train_y, num_classes = 101)\nprint(train_y_cat.shape)\n\nprint(validation_y.shape)\nvalidation_y_cat = to_categorical(validation_y, num_classes = 101)\nprint(validation_y_cat.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Common Functions"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_nice_confusion_matrix(y_true, y_pred):\n    cm = confusion_matrix(y_true, y_pred)\n    fig, ax = plt.subplots(figsize=(18,18))\n    sns.heatmap(cm, annot=True, fmt='d', linewidths=.5,  cbar=False, ax=ax, cmap=plt.cm.copper)\n    plt.ylabel('true label')\n    plt.xlabel('predicted label')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# NN Definition"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications import DenseNet121, InceptionV3, ResNet50, VGG16\nfrom keras.layers import GlobalAveragePooling2D\n\nreduceLR = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.5,\n    patience=7,\n    min_lr=1e-6,\n    verbose=1,\n    mode='min'\n)\n\nearlyStopping = EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    verbose=1,\n    mode='min',\n    restore_best_weights=True\n)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a4c55409-6a65-400a-b5e8-a1dc535429c0","_execution_state":"idle","_uuid":"420c704367b397b8255fefe9d882b35ac8929b95","trusted":true},"cell_type":"code","source":"# Define the optimizer\nmy_optimizer = RMSprop(lr=1e-4)\n#my_optimizer = Adam(lr=1e-5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# NN Training & Visualization"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_training_history(history):\n    history_df = pd.DataFrame(history.history)\n    f = plt.figure(figsize=(25,5))\n    ax = f.add_subplot(121)\n    ax.plot(history_df[\"loss\"], label=\"loss\")\n    ax.plot(history_df[\"val_loss\"], label = \"val_loss\")\n    ax.legend()\n    ax = f.add_subplot(122)\n    ax.plot(history_df[\"acc\"], label=\"acc\")\n    ax.plot(history_df[\"val_acc\"], label=\"val_acc\")\n    ax.legend()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Evaluate on Validation Data"},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Predict Test Data & Submit"},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"#define test generator \"test_data_generator\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"test_y_pred_proba = model.predict_generator(test_data_generator, \n                                            steps=np.ceil(float(test_ids_all.shape[0]) / float(BATCH_SIZE)),\n                                            verbose=1)\n\ntest_y_pred = np.argmax(test_y_pred_proba, axis=1)\nnn_results = pd.Series(test_y_pred,name=\"label\")\nsubmission = pd.concat([test_ids_all[\"id_code\"],nn_results], axis = 1)\n\nsubmission.to_csv(\"food_submission_simple_model.csv\",index=False)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.3"},"latex_envs":{"LaTeX_envs_menu_present":true,"autoclose":false,"autocomplete":true,"bibliofile":"biblio.bib","cite_by":"apalike","current_citInitial":1,"eqLabelWithNumbers":true,"eqNumInitial":1,"hotkeys":{"equation":"Ctrl-E","itemize":"Ctrl-I"},"labels_anchors":false,"latex_user_defs":false,"report_style_numbering":false,"user_envs_cfg":false},"toc":{"base_numbering":1,"nav_menu":{},"number_sections":true,"sideBar":true,"skip_h1_title":false,"title_cell":"Table of Contents","title_sidebar":"Contents","toc_cell":false,"toc_position":{},"toc_section_display":true,"toc_window_display":true}},"nbformat":4,"nbformat_minor":1}