{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nprint(os.listdir(\"../input\"))\nfrom keras.applications import ResNet50\nfrom keras.models import Model\nfrom keras.layers import Dense, GlobalAveragePooling2D, Dropout, Input\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport keras\nimport csv\nimport gc\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = \"/kaggle/input/aptos2019-blindness-detection/train.csv\"\ntest_csv = \"/kaggle/input/aptos2019-blindness-detection/test.csv\"\ntrain_dir = \"/kaggle/input/aptos2019-blindness-detection/train_images/\"\ntest_dir = \"/kaggle/input/aptos2019-blindness-detection/test_images/\"\nsize = 256,256 # input image size","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(train_csv)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def df_train_test_split_preprocess(df):\n    image_ids = df[\"id_code\"].values.tolist()\n    labels = df[\"diagnosis\"].values.tolist()\n    for i in range(len(image_ids)):\n        imgname = image_ids[i]\n        newname = str(imgname) + \".png\"\n        image_ids[i] = newname\n    xtrain, xval, ytrain, yval = train_test_split(image_ids, labels, test_size = 0.15)\n    df_train = pd.DataFrame({\"id_code\":xtrain, \"diagnosis\":ytrain})\n    df_val = pd.DataFrame({\"id_code\":xval, \"diagnosis\":yval})\n    df_train[\"diagnosis\"] = df_train[\"diagnosis\"].astype('str')\n    df_val[\"diagnosis\"] = df_val[\"diagnosis\"].astype('str')\n    print(\"Length of Training Data :\",len(df_train))\n    print(\"Length of Validation Data :\",len(df_val))\n    return df_train, df_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train, df_val = df_train_test_split_preprocess(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_cropped_image(image):\n    img = cv2.blur(image,(2,2))\n    slice1Copy = np.uint8(img)\n    canny = cv2.Canny(slice1Copy, 0, 50)\n    pts = np.argwhere(canny>0)\n    y1,x1 = pts.min(axis=0)\n    y2,x2 = pts.max(axis=0)\n    cropped_img = img[y1:y2, x1:x2]\n    cropped_img = cv2.resize(cropped_img, size)\n    cropped_img = cropped_img.astype(\"float32\")*(1.)/255\n    return np.array(cropped_img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_aug = ImageDataGenerator(#rescale=1./255,\n                               horizontal_flip = True,\n                               zoom_range = 0.25,\n                               vertical_flip = True,\n                               preprocessing_function = get_cropped_image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator = train_aug.flow_from_dataframe(dataframe = df_train,\n                                               directory = train_dir,\n                                               x_col = \"id_code\",\n                                               y_col = \"diagnosis\",\n                                               batch_size = 16, \n                                               class_mode = \"categorical\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_aug = ImageDataGenerator(rescale=1./255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"validation_generator = val_aug.flow_from_dataframe(dataframe = df_val,\n                                                    directory = train_dir,\n                                                    x_col = \"id_code\",\n                                                    y_col = \"diagnosis\",\n                                                    target_size = size,\n                                                    batch_size = 16,\n                                                    class_mode = \"categorical\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_layer = Input(shape = (256,256,3))\nbase_model = ResNet50(include_top = False, input_tensor = input_layer, weights = \"../input/resnet50/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5\")\nx = GlobalAveragePooling2D()(base_model.output)\nx = Dropout(0.5)(x)\nout = Dense(5, activation = 'softmax')(x)\n\nmodel = Model(inputs = input_layer, outputs = out)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"optimizer = keras.optimizers.Adam(lr=2e-4)\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience = 8, restore_best_weights=True)\nrlrop = ReduceLROnPlateau(monitor='val_loss', mode='min', patience = 3, factor = 0.5, min_lr=1e-6)\n    \ncallback_list = [es, rlrop]\n\nmodel.compile(optimizer = optimizer, loss = \"categorical_crossentropy\", metrics = [\"accuracy\"]) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit_generator(generator = train_generator, steps_per_epoch = len(train_generator), epochs = 20, validation_data = validation_generator, validation_steps = len(validation_generator), callbacks = callback_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df_orig = pd.read_csv(test_csv)\n\ndef process_test_df(test_df):\n    test_ids = test_df[\"id_code\"].values.tolist()\n    for i in range(len(test_ids)):\n        imgname = test_ids[i]\n        newname = str(imgname) + \".png\"\n        test_ids[i] = newname\n    test_df[\"id_code\"] = test_ids\n    return test_df\n\ntest_df = process_test_df(test_df_orig)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_aug = ImageDataGenerator(rescale = 1./255)\n\ntest_generator = test_aug.flow_from_dataframe(dataframe = test_df, \n                                              directory = test_dir,\n                                              x_col = \"id_code\",\n                                              batch_size = 1,\n                                              target_size = (256,256),\n                                              shuffle = False,\n                                              class_mode = None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predprobs = model.predict_generator(test_generator, steps=len(test_generator))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = []\nfor i in predprobs:\n    predictions.append(np.argmax(i)) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df_orig[\"diagnosis\"] = predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df_orig.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}