{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nfrom typing import Tuple\nfrom sklearn.model_selection import train_test_split\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nfrom tqdm import tqdm\nimport os\nimport matplotlib.pyplot as plt\n\n\nimport tensorflow.keras as keras\nfrom tensorflow.keras.layers import Input, Flatten, Dropout, Conv2D, MaxPool2D, Dense, BatchNormalization, LeakyReLU\nfrom tensorflow.keras.optimizers import Adamax\nfrom tensorflow.keras.activations import relu, softmax\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n\n\n\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.applications.vgg16 import preprocess_input","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-14T16:15:02.260266Z","iopub.execute_input":"2022-09-14T16:15:02.260748Z","iopub.status.idle":"2022-09-14T16:15:02.272075Z","shell.execute_reply.started":"2022-09-14T16:15:02.260716Z","shell.execute_reply":"2022-09-14T16:15:02.270635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size_for_resize=(100,100)\nimg_size_for_model=(100, 100, 3)\nepochs=70\nbatch_size=64","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:53:16.837031Z","iopub.execute_input":"2022-09-14T15:53:16.837459Z","iopub.status.idle":"2022-09-14T15:53:16.844815Z","shell.execute_reply.started":"2022-09-14T15:53:16.837426Z","shell.execute_reply":"2022-09-14T15:53:16.843157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"read csv files","metadata":{}},{"cell_type":"code","source":"def read_files(file_name:str)->pd.DataFrame():\n    return pd.read_csv(f'/kaggle/input/aptos2019-blindness-detection/{file_name}.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:08:39.842823Z","iopub.execute_input":"2022-09-14T15:08:39.843302Z","iopub.status.idle":"2022-09-14T15:08:39.851521Z","shell.execute_reply.started":"2022-09-14T15:08:39.843271Z","shell.execute_reply":"2022-09-14T15:08:39.849764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = read_files('train')\ndf_test = read_files('test')\ndf_submission = read_files('sample_submission')","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:08:40.029927Z","iopub.execute_input":"2022-09-14T15:08:40.030349Z","iopub.status.idle":"2022-09-14T15:08:40.053022Z","shell.execute_reply.started":"2022-09-14T15:08:40.030316Z","shell.execute_reply":"2022-09-14T15:08:40.051751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"read and classify images","metadata":{}},{"cell_type":"code","source":"def collect_data(data_type:str, df)->Tuple[list, list]:\n    X=[]\n    y=[]\n    for dirname, _, filenames in os.walk('/kaggle/input'):\n        if  data_type in dirname:\n            for filename in tqdm(filenames):\n                if 'csv' not in filename:\n                    image = cv2.imread(os.path.join(dirname, filename))\n                    image = cv2.resize(image, img_size_for_resize)\n                    X.append(image)\n                    y.append(df.loc[df['id_code']==filename[:-4], 'diagnosis'].values)\n    return X, y","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:08:42.112990Z","iopub.execute_input":"2022-09-14T15:08:42.113459Z","iopub.status.idle":"2022-09-14T15:08:42.125765Z","shell.execute_reply.started":"2022-09-14T15:08:42.113425Z","shell.execute_reply":"2022-09-14T15:08:42.121986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = collect_data('train', df_train)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:08:43.914918Z","iopub.execute_input":"2022-09-14T15:08:43.915415Z","iopub.status.idle":"2022-09-14T15:15:28.971104Z","shell.execute_reply.started":"2022-09-14T15:08:43.915365Z","shell.execute_reply":"2022-09-14T15:15:28.969613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array([image for image in X]).reshape(-1, 100, 100, 3)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:15:28.974941Z","iopub.execute_input":"2022-09-14T15:15:28.975860Z","iopub.status.idle":"2022-09-14T15:15:29.137937Z","shell.execute_reply.started":"2022-09-14T15:15:28.975811Z","shell.execute_reply":"2022-09-14T15:15:29.136342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=[value[0] for value in y]\ny = pd.get_dummies(y)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:15:29.141433Z","iopub.execute_input":"2022-09-14T15:15:29.142316Z","iopub.status.idle":"2022-09-14T15:15:29.155654Z","shell.execute_reply.started":"2022-09-14T15:15:29.142267Z","shell.execute_reply":"2022-09-14T15:15:29.154076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"split data","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.8, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:15:37.936887Z","iopub.execute_input":"2022-09-14T15:15:37.937324Z","iopub.status.idle":"2022-09-14T15:15:38.125068Z","shell.execute_reply.started":"2022-09-14T15:15:37.937290Z","shell.execute_reply":"2022-09-14T15:15:38.123224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(X_train[152])","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:15:38.587841Z","iopub.execute_input":"2022-09-14T15:15:38.588749Z","iopub.status.idle":"2022-09-14T15:15:38.846339Z","shell.execute_reply.started":"2022-09-14T15:15:38.588685Z","shell.execute_reply":"2022-09-14T15:15:38.844657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create Model","metadata":{}},{"cell_type":"code","source":"def sequential_model(nbr_classes:int)->keras.Sequential:\n    MyInput = Input(shape=img_size_for_model)\n    X = Conv2D(16,(5,5), activation=relu)(MyInput)\n    X = MaxPool2D()(X)\n    \n    X = Conv2D(32,(2,2), activation=relu)(X)\n    X = MaxPool2D()(X)\n    \n    X = BatchNormalization()(X)\n    \n    X = Flatten()(X)\n    X = Dense(32)(X)\n    X = LeakyReLU(alpha=0.03)(X)\n    X = Dropout(.1)(X)\n    X = Dense(16)(X)\n    X = LeakyReLU(alpha=0.02)(X)\n    X = Dense(nbr_classes, activation=softmax)(X)\n    \n    model = keras.Model(inputs=MyInput, outputs=X)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:53:49.900761Z","iopub.execute_input":"2022-09-14T15:53:49.901789Z","iopub.status.idle":"2022-09-14T15:53:49.912192Z","shell.execute_reply.started":"2022-09-14T15:53:49.901753Z","shell.execute_reply":"2022-09-14T15:53:49.910351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = sequential_model(nbr_classes=df_train.diagnosis.unique().shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:55:49.852903Z","iopub.execute_input":"2022-09-14T15:55:49.853666Z","iopub.status.idle":"2022-09-14T15:55:49.962063Z","shell.execute_reply.started":"2022-09-14T15:55:49.853601Z","shell.execute_reply":"2022-09-14T15:55:49.960373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"comile and train model","metadata":{}},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor=\"val_accuracy\", patience=5)\noptimizer = Adamax(learning_rate=0.01)\nmodel.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=['accuracy'])\nhistory = model.fit(X_train,\n          y_train,\n          epochs=epochs,\n          batch_size=batch_size,\n          validation_split=.3, \n          callbacks=[early_stopping]\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:55:50.269427Z","iopub.execute_input":"2022-09-14T15:55:50.270968Z","iopub.status.idle":"2022-09-14T15:55:57.616037Z","shell.execute_reply.started":"2022-09-14T15:55:50.270900Z","shell.execute_reply":"2022-09-14T15:55:57.614571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"evaluate model","metadata":{}},{"cell_type":"code","source":"model.evaluate(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:56:12.446538Z","iopub.execute_input":"2022-09-14T15:56:12.447041Z","iopub.status.idle":"2022-09-14T15:56:32.304285Z","shell.execute_reply.started":"2022-09-14T15:56:12.447007Z","shell.execute_reply":"2022-09-14T15:56:32.302855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plots","metadata":{}},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T15:30:03.572402Z","iopub.execute_input":"2022-09-14T15:30:03.573575Z","iopub.status.idle":"2022-09-14T15:30:04.110043Z","shell.execute_reply.started":"2022-09-14T15:30:03.573503Z","shell.execute_reply":"2022-09-14T15:30:04.108567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"using VGG16","metadata":{}},{"cell_type":"code","source":"base_model = VGG16(weights=\"imagenet\", include_top=False, input_shape=X_train[0].shape)\nbase_model.trainable = False ## Not trainable weights\n\n## Preprocessing input\ntrain_ds = preprocess_input(X_train) \ntest_ds = preprocess_input(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T16:16:18.084661Z","iopub.execute_input":"2022-09-14T16:16:18.085090Z","iopub.status.idle":"2022-09-14T16:16:19.463198Z","shell.execute_reply.started":"2022-09-14T16:16:18.085046Z","shell.execute_reply":"2022-09-14T16:16:19.461690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T16:16:47.246399Z","iopub.execute_input":"2022-09-14T16:16:47.247102Z","iopub.status.idle":"2022-09-14T16:16:47.270531Z","shell.execute_reply.started":"2022-09-14T16:16:47.247051Z","shell.execute_reply":"2022-09-14T16:16:47.269040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flatten_layer = Flatten()\ndense_layer_1 = Dense(50, activation='relu')\ndense_layer_2 = Dense(20, activation='relu')\nprediction_layer = Dense(df_train.diagnosis.unique().shape[0], activation='softmax')\n\n\nmodel = keras.models.Sequential([\n    base_model,\n    flatten_layer,\n    dense_layer_1,\n    dense_layer_2,\n    prediction_layer\n])","metadata":{"execution":{"iopub.status.busy":"2022-09-14T16:21:23.020244Z","iopub.execute_input":"2022-09-14T16:21:23.021041Z","iopub.status.idle":"2022-09-14T16:21:23.213826Z","shell.execute_reply.started":"2022-09-14T16:21:23.021007Z","shell.execute_reply":"2022-09-14T16:21:23.211846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer='adam',\n    loss='categorical_crossentropy',\n    metrics=['accuracy'],\n)\n\n\nes = EarlyStopping(monitor='val_accuracy', mode='max', patience=5,  restore_best_weights=True)\n\nmodel.fit(X_train, y_train, epochs=30, validation_split=0.2, batch_size=32, callbacks=[es])\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T16:21:24.085819Z","iopub.execute_input":"2022-09-14T16:21:24.086281Z","iopub.status.idle":"2022-09-14T16:21:45.544341Z","shell.execute_reply.started":"2022-09-14T16:21:24.086232Z","shell.execute_reply":"2022-09-14T16:21:45.542699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T16:21:48.313818Z","iopub.execute_input":"2022-09-14T16:21:48.314293Z","iopub.status.idle":"2022-09-14T16:21:48.815176Z","shell.execute_reply.started":"2022-09-14T16:21:48.314259Z","shell.execute_reply":"2022-09-14T16:21:48.813785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}