{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Ignore  the warnings\nimport warnings\nwarnings.filterwarnings('always')\nwarnings.filterwarnings('ignore')\n\n# data visualisation and manipulation\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib import style\nimport seaborn as sns\n \n#configure\n# sets matplotlib to inline and displays graphs below the corressponding cell.\n%matplotlib inline  \nstyle.use('fivethirtyeight')\nsns.set(style='whitegrid',color_codes=True)\n\n#model selection\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,confusion_matrix,roc_curve,roc_auc_score\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.preprocessing import LabelEncoder\n\n#preprocess.\nfrom keras.preprocessing.image import ImageDataGenerator\n\n#dl libraraies\nfrom keras import backend as K\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.optimizers import Adam,SGD,Adagrad,Adadelta,RMSprop\nfrom keras.utils import to_categorical\nfrom keras.utils.vis_utils import model_to_dot\nfrom keras.utils.vis_utils import plot_model\n\n# specifically for cnn\nfrom keras.applications.inception_v3 import InceptionV3, preprocess_input\nfrom keras.layers import Dropout, Flatten,Activation\nfrom keras.layers import Conv2D, MaxPooling2D, BatchNormalization,GlobalAveragePooling2D\nfrom keras.callbacks import ModelCheckpoint,EarlyStopping,TensorBoard,CSVLogger,ReduceLROnPlateau,LearningRateScheduler\n    \nimport tensorflow as tf\nimport random as rn\n\n# specifically for manipulating zipped images and getting numpy arrays of pixel values of images.\nimport cv2                  \nimport numpy as np  \nfrom tqdm import tqdm\nimport os                   \nfrom random import shuffle  \nfrom zipfile import ZipFile\nfrom PIL import Image\n\nprint(os.listdir(\"../input\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-06T04:47:04.015816Z","iopub.execute_input":"2022-06-06T04:47:04.016193Z","iopub.status.idle":"2022-06-06T04:47:06.715921Z","shell.execute_reply.started":"2022-06-06T04:47:04.016136Z","shell.execute_reply":"2022-06-06T04:47:06.714753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=[]\nZ=[]\nIMG_SIZE=150\nTRAIN_DIR='../input/aptos2019-blindness-detection/train_images'","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2022-06-06T04:47:06.719407Z","iopub.execute_input":"2022-06-06T04:47:06.719737Z","iopub.status.idle":"2022-06-06T04:47:06.724724Z","shell.execute_reply.started":"2022-06-06T04:47:06.719679Z","shell.execute_reply":"2022-06-06T04:47:06.723605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_train_data(label,path):\n    img = cv2.imread(path,cv2.IMREAD_COLOR)\n    img = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n\n    X.append(np.array(img))\n    Z.append(str(label))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:47:06.726518Z","iopub.execute_input":"2022-06-06T04:47:06.727120Z","iopub.status.idle":"2022-06-06T04:47:06.738452Z","shell.execute_reply.started":"2022-06-06T04:47:06.727066Z","shell.execute_reply":"2022-06-06T04:47:06.737550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:47:06.739987Z","iopub.execute_input":"2022-06-06T04:47:06.740563Z","iopub.status.idle":"2022-06-06T04:47:06.792688Z","shell.execute_reply.started":"2022-06-06T04:47:06.740360Z","shell.execute_reply":"2022-06-06T04:47:06.791683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df['id_code']\ny = df['diagnosis']","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:47:06.795916Z","iopub.execute_input":"2022-06-06T04:47:06.796244Z","iopub.status.idle":"2022-06-06T04:47:06.801075Z","shell.execute_reply.started":"2022-06-06T04:47:06.796180Z","shell.execute_reply":"2022-06-06T04:47:06.799855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for id_code,diagnosis in tqdm(zip(x,y)):\n    path = os.path.join(TRAIN_DIR,'{}.png'.format(id_code))\n    make_train_data(diagnosis,path)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:47:06.804283Z","iopub.execute_input":"2022-06-06T04:47:06.804903Z","iopub.status.idle":"2022-06-06T04:54:11.102724Z","shell.execute_reply.started":"2022-06-06T04:47:06.804839Z","shell.execute_reply":"2022-06-06T04:54:11.101738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check some image\nfig,ax=plt.subplots(5,2)\nfig.set_size_inches(15,15)\nfor i in range(5):\n    for j in range (2):\n        l=rn.randint(0,len(Z))\n        ax[i,j].imshow(X[l])\n        ax[i,j].set_title(Z[l])\n        \nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:11.104427Z","iopub.execute_input":"2022-06-06T04:54:11.104894Z","iopub.status.idle":"2022-06-06T04:54:13.888584Z","shell.execute_reply.started":"2022-06-06T04:54:11.104849Z","shell.execute_reply":"2022-06-06T04:54:13.887770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y=to_categorical(Z)\nX=np.array(X)\nX=X/255","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:13.889842Z","iopub.execute_input":"2022-06-06T04:54:13.890310Z","iopub.status.idle":"2022-06-06T04:54:16.289213Z","shell.execute_reply.started":"2022-06-06T04:54:13.890259Z","shell.execute_reply":"2022-06-06T04:54:16.287820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_valid,y_train,y_valid = train_test_split(X,Y,test_size=0.2,random_state=42)\ndel X\ndel Y\ndel Z","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:16.291053Z","iopub.execute_input":"2022-06-06T04:54:16.291440Z","iopub.status.idle":"2022-06-06T04:54:18.561712Z","shell.execute_reply.started":"2022-06-06T04:54:16.291380Z","shell.execute_reply":"2022-06-06T04:54:18.560669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Data Augmentation\naugs_gen = ImageDataGenerator(\n        featurewise_center=False,  \n        samplewise_center=False, \n        featurewise_std_normalization=False,  \n        samplewise_std_normalization=False,  \n        zca_whitening=False,  \n        rotation_range=10,  \n        zoom_range = 0.1, \n        width_shift_range=0.2,  \n        height_shift_range=0.2, \n        horizontal_flip=True,  \n        vertical_flip=False) \n\naugs_gen.fit(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:18.563364Z","iopub.execute_input":"2022-06-06T04:54:18.563653Z","iopub.status.idle":"2022-06-06T04:54:20.217193Z","shell.execute_reply.started":"2022-06-06T04:54:18.563608Z","shell.execute_reply":"2022-06-06T04:54:20.216008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # modelling starts using a CNN.\n\nmodel = Sequential()\nmodel.add(Conv2D(filters = 32, kernel_size = (5,5),padding = 'Same',activation ='relu', input_shape = (150,150,3)))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\n\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n \n\nmodel.add(Conv2D(filters =96, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n\nmodel.add(Conv2D(filters = 96, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dense(5, activation = \"softmax\"))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:20.219190Z","iopub.execute_input":"2022-06-06T04:54:20.219735Z","iopub.status.idle":"2022-06-06T04:54:20.416899Z","shell.execute_reply.started":"2022-06-06T04:54:20.219658Z","shell.execute_reply":"2022-06-06T04:54:20.415778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set callbacks\ncheckpoint = ModelCheckpoint(\n    './base.model',\n    monitor='val_loss',\n    verbose=1,\n    save_best_only=True,\n    mode='min',\n    save_weights_only=False,\n    period=1\n)\nearlystop = EarlyStopping(\n    monitor='val_loss',\n    min_delta=0.001,\n    patience=30,\n    verbose=1,\n    mode='auto'\n)\ntensorboard = TensorBoard(\n    log_dir = './logs',\n    histogram_freq=0,\n    batch_size=16,\n    write_graph=True,\n    write_grads=True,\n    write_images=False,\n)\n\ncsvlogger = CSVLogger(\n    filename= \"training_csv.log\",\n    separator = \",\",\n    append = False\n)\n\nreduce = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.1,\n    patience=5,\n    min_lr=1e-6,\n    verbose=1, \n    mode='auto'\n)\n\ncallbacks = [checkpoint,tensorboard,csvlogger,reduce]","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:20.419345Z","iopub.execute_input":"2022-06-06T04:54:20.419924Z","iopub.status.idle":"2022-06-06T04:54:22.990829Z","shell.execute_reply.started":"2022-06-06T04:54:20.419864Z","shell.execute_reply":"2022-06-06T04:54:22.989035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=64\nepochs=20","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:22.992875Z","iopub.execute_input":"2022-06-06T04:54:22.993185Z","iopub.status.idle":"2022-06-06T04:54:23.007309Z","shell.execute_reply.started":"2022-06-06T04:54:22.993144Z","shell.execute_reply":"2022-06-06T04:54:23.005814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=Adam(lr=0.001),loss='categorical_crossentropy',metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:23.009094Z","iopub.execute_input":"2022-06-06T04:54:23.009461Z","iopub.status.idle":"2022-06-06T04:54:23.077635Z","shell.execute_reply.started":"2022-06-06T04:54:23.009403Z","shell.execute_reply":"2022-06-06T04:54:23.076085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"History = model.fit_generator(augs_gen.flow(x_train,y_train, batch_size=batch_size),\n                              epochs = epochs, validation_data = (x_valid,y_valid),\n                              verbose = 1, steps_per_epoch=x_train.shape[0] // batch_size,\n                              callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T04:54:23.079847Z","iopub.execute_input":"2022-06-06T04:54:23.080336Z","iopub.status.idle":"2022-06-06T05:49:09.350774Z","shell.execute_reply.started":"2022-06-06T04:54:23.080273Z","shell.execute_reply":"2022-06-06T05:49:09.347456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:49:09.356476Z","iopub.execute_input":"2022-06-06T05:49:09.357160Z","iopub.status.idle":"2022-06-06T05:49:09.402046Z","shell.execute_reply.started":"2022-06-06T05:49:09.357104Z","shell.execute_reply":"2022-06-06T05:49:09.401093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = test_df['id_code']","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:49:09.403246Z","iopub.execute_input":"2022-06-06T05:49:09.403637Z","iopub.status.idle":"2022-06-06T05:49:09.409262Z","shell.execute_reply.started":"2022-06-06T05:49:09.403597Z","shell.execute_reply":"2022-06-06T05:49:09.407892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_X = []\ndef make_test_data(path):\n    img = cv2.imread(path,cv2.IMREAD_COLOR)\n    img = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n\n    TEST_X.append(np.array(img))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:49:09.410743Z","iopub.execute_input":"2022-06-06T05:49:09.411236Z","iopub.status.idle":"2022-06-06T05:49:09.430469Z","shell.execute_reply.started":"2022-06-06T05:49:09.411169Z","shell.execute_reply":"2022-06-06T05:49:09.429518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_DIR='../input/aptos2019-blindness-detection/test_images'\nfor id_code in tqdm(x):\n    path = os.path.join(TEST_DIR,'{}.png'.format(id_code))\n    make_test_data(path)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:49:09.431984Z","iopub.execute_input":"2022-06-06T05:49:09.432566Z","iopub.status.idle":"2022-06-06T05:50:37.379437Z","shell.execute_reply.started":"2022-06-06T05:49:09.432470Z","shell.execute_reply":"2022-06-06T05:50:37.378476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_X=np.array(TEST_X)\nTEST_X=TEST_X/255\npred=model.predict(TEST_X)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:50:37.381364Z","iopub.execute_input":"2022-06-06T05:50:37.381687Z","iopub.status.idle":"2022-06-06T05:51:31.613976Z","shell.execute_reply.started":"2022-06-06T05:50:37.381631Z","shell.execute_reply":"2022-06-06T05:51:31.612955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=np.argmax(pred,axis=1)\npred","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:51:31.615947Z","iopub.execute_input":"2022-06-06T05:51:31.616392Z","iopub.status.idle":"2022-06-06T05:51:31.626311Z","shell.execute_reply.started":"2022-06-06T05:51:31.616306Z","shell.execute_reply":"2022-06-06T05:51:31.624831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv('../input/aptos2019-blindness-detection/sample_submission.csv')\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:51:31.627992Z","iopub.execute_input":"2022-06-06T05:51:31.628568Z","iopub.status.idle":"2022-06-06T05:51:31.667495Z","shell.execute_reply.started":"2022-06-06T05:51:31.628512Z","shell.execute_reply":"2022-06-06T05:51:31.666005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.diagnosis = pred\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:51:31.669146Z","iopub.execute_input":"2022-06-06T05:51:31.669688Z","iopub.status.idle":"2022-06-06T05:51:31.689558Z","shell.execute_reply.started":"2022-06-06T05:51:31.669636Z","shell.execute_reply":"2022-06-06T05:51:31.687878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:51:31.692349Z","iopub.execute_input":"2022-06-06T05:51:31.692842Z","iopub.status.idle":"2022-06-06T05:51:32.578808Z","shell.execute_reply.started":"2022-06-06T05:51:31.692759Z","shell.execute_reply":"2022-06-06T05:51:32.577590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# serialize weights to HDF5\nmodel.save_weights(\"model.h5\")\nprint(\"Saved model to disk\")","metadata":{"execution":{"iopub.status.busy":"2022-06-06T06:46:42.224933Z","iopub.execute_input":"2022-06-06T06:46:42.225673Z","iopub.status.idle":"2022-06-06T06:46:42.319953Z","shell.execute_reply.started":"2022-06-06T06:46:42.225607Z","shell.execute_reply":"2022-06-06T06:46:42.318030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Accuracy and Loss graphs\nimport matplotlib.pyplot as plt\nplt.plot(History.history['acc'])\nplt.plot(History.history['val_acc'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()\n\n\nplt.plot(History.history['loss'])\nplt.plot(History.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T05:51:32.641734Z","iopub.execute_input":"2022-06-06T05:51:32.643302Z","iopub.status.idle":"2022-06-06T05:51:33.275734Z","shell.execute_reply.started":"2022-06-06T05:51:32.642113Z","shell.execute_reply":"2022-06-06T05:51:33.274690Z"},"trusted":true},"execution_count":null,"outputs":[]}]}