{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install -y gdown","metadata":{"execution":{"iopub.status.busy":"2021-10-25T05:11:41.446773Z","iopub.execute_input":"2021-10-25T05:11:41.447371Z","iopub.status.idle":"2021-10-25T05:12:33.705532Z","shell.execute_reply.started":"2021-10-25T05:11:41.447254Z","shell.execute_reply":"2021-10-25T05:12:33.704725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!gdown --id 1hE1Sej1sZg7F4xqUvx26UerizXWiY7S6","metadata":{"execution":{"iopub.status.busy":"2021-10-25T05:12:33.708931Z","iopub.execute_input":"2021-10-25T05:12:33.709164Z","iopub.status.idle":"2021-10-25T05:12:37.324987Z","shell.execute_reply.started":"2021-10-25T05:12:33.709135Z","shell.execute_reply":"2021-10-25T05:12:37.324180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nwith zipfile.ZipFile('./train-20211023T043245Z-001.zip', 'r') as zip_ref:\n    zip_ref.extractall('./train')","metadata":{"execution":{"iopub.status.busy":"2021-10-25T05:13:19.578034Z","iopub.execute_input":"2021-10-25T05:13:19.578320Z","iopub.status.idle":"2021-10-25T05:13:23.337927Z","shell.execute_reply.started":"2021-10-25T05:13:19.578275Z","shell.execute_reply":"2021-10-25T05:13:23.337208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nЗдесь использована модель сети VGG19\n'''\nimport time\nsince = time.time()\n# pip3 install python-gdcm\n\nimport tensorflow as tf\n\nfrom tensorflow.keras import datasets, layers, models\nimport matplotlib.pyplot as plt\nimport pathlib\nimport pandas as pd, numpy as np\nimport shutil\nimport re, os\nimport cv2\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom sklearn.model_selection import StratifiedKFold\nimport random\nfrom matplotlib import pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-25T06:33:19.282060Z","iopub.execute_input":"2021-10-25T06:33:19.282340Z","iopub.status.idle":"2021-10-25T06:33:19.287752Z","shell.execute_reply.started":"2021-10-25T06:33:19.282292Z","shell.execute_reply":"2021-10-25T06:33:19.287103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining Callbacks\nfilepath = './best_weightsVGG.hdf5'\n\nearlystopping = EarlyStopping(monitor = 'val_accuracy', \n                              mode = 'max' , \n                              patience = 10,\n                              verbose = 1)\n\ncheckpoint    = ModelCheckpoint(filepath, \n                                monitor = 'val_accuracy', \n                                mode='max', \n                                save_best_only=True, \n                                verbose = 1)\n\ncallback_list = [earlystopping, checkpoint]\n\n# готовим параметры для формирования ImageDataGenerator\nbatch_size = 4\n\n# строим модель\nNUM_CLASSES = 4","metadata":{"execution":{"iopub.status.busy":"2021-10-25T06:33:34.524414Z","iopub.execute_input":"2021-10-25T06:33:34.524804Z","iopub.status.idle":"2021-10-25T06:33:34.530611Z","shell.execute_reply.started":"2021-10-25T06:33:34.524764Z","shell.execute_reply":"2021-10-25T06:33:34.529817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = tf.keras.applications.vgg19.VGG19(\n    include_top=False, weights='imagenet', \n    input_shape=(224, 224, 3),\n)\n\nx = base_model.output\n# x = tf.keras.layers.GlobalAveragePooling2D(name='avg_pool')(x)\nx = layers.Flatten()(x)\n\nx = layers.Dense(1024, activation='relu')(x)\nx = layers.Dropout(0.2)(x)\nx = layers.Dense(512, activation='relu')(x)\nx = layers.Dropout(0.2)(x)\n# x = layers.GlobalAveragePooling2D()(x)\nbase_model.trainable = True\n\nfor layer in base_model.layers:\n    if layer.name == 'block4_conv1':\n        layer.trainable = True\n    if layer.name == 'block4_conv2':\n        layer.trainable = True\n    if layer.name == 'block4_conv3':\n        layer.trainable = True\n    if layer.name == 'block4_conv4':\n        layer.trainable = True\n    if layer.name == 'block5_conv1':\n        layer.trainable = True\n    if layer.name == 'block5_conv2':\n        layer.trainable = True\n    if layer.name == 'block5_conv3':\n        layer.trainable = True\n    if layer.name == 'block5_conv4':\n        layer.trainable = True\n    layer.trainable = False\n    \npredictions = layers.Dense(NUM_CLASSES, activation='softmax')(x)\nmodel = models.Model(inputs=base_model.inputs, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2021-10-25T06:33:46.064858Z","iopub.execute_input":"2021-10-25T06:33:46.065500Z","iopub.status.idle":"2021-10-25T06:33:46.432708Z","shell.execute_reply.started":"2021-10-25T06:33:46.065462Z","shell.execute_reply":"2021-10-25T06:33:46.431940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_loss = tf.keras.metrics.Mean('training_loss', dtype=tf.float32)\ntraining_accuracy = tf.keras.metrics.CategoricalCrossentropy('training_accuracy', dtype=tf.float32)\ntest_loss = tf.keras.metrics.Mean('test_loss', dtype=tf.float32)\ntest_accuracy = tf.keras.metrics.CategoricalCrossentropy('test_accuracy', dtype=tf.float32)\nprint(model.summary())\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.00001)\n\nIMG_SIZE = (224, 224)\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n                            samplewise_center=True, \n                            samplewise_std_normalization=True, \n                            horizontal_flip = True, \n                            vertical_flip = True, \n                            height_shift_range= 0.2, \n                            width_shift_range=0.2, \n                            rotation_range=15, \n                            shear_range = 0.2,\n                            # validation_split=.20,\n                            fill_mode = 'reflect',\n                            preprocessing_function=tf.keras.applications.vgg19.preprocess_input,\n                            zoom_range=0.2)\n\n# Training with K-fold cross validation\nkf = StratifiedKFold(n_splits=5, random_state=None, shuffle=True)\ndf_large = pd.read_csv('../input/traincovid/train.csv')\nX= np.array(df_large[\"ImageInstanceUID\"])\nY= np.array(df_large[\"label_id\"])\ni = 1\nfor train_index, test_index in kf.split(X, Y):\n    print(\"Iteration = \", i)\n    i += 1\n    trainData = X[train_index]\n    testData = X[test_index]\n    ## create train, valid dataframe and thus train_gen , valid_gen for each fold-loop\n    train_df = df_large.loc[df_large[\"ImageInstanceUID\"].isin(list(trainData))]\n    valid_df = df_large.loc[df_large[\"ImageInstanceUID\"].isin(list(testData))]\n    #create model object\n    model.compile(loss='categorical_crossentropy',\n             optimizer=optimizer,\n             metrics=['accuracy'])\n    all_labels = [\"Negative for Pneumonia\", \"Typical Appearance\", \"Indeterminate Appearance\", \"Atypical Appearance\"]\n    train_generator = train_datagen.flow_from_dataframe(dataframe=train_df,\n                                        directory=\"./train/train/\",\n                                        x_col = 'ImageInstanceUID',\n                                        y_col = 'label_id',\n                                        class_mode = 'categorical',\n                                        classes = all_labels,\n                                        interpolation=\"lanczos\",\n                                        target_size = IMG_SIZE,\n                                        #  color_mode = 'rgb',\n                                        batch_size = 8)\n    validation_generator = train_datagen.flow_from_dataframe(dataframe=valid_df,\n                                        directory=\"./train/train/\",\n                                        x_col = 'ImageInstanceUID',\n                                        y_col = 'label_id',\n                                        class_mode = 'categorical',\n                                        classes = all_labels,\n                                        interpolation=\"lanczos\",\n                                        target_size = IMG_SIZE,\n                                        #  color_mode = 'rgb',\n                                        batch_size = 8)\n    model_history=model.fit(\n        train_generator,\n        # steps_per_epoch= len(trainData),\n        epochs= 100,\n        validation_data=validation_generator,\n        callbacks = callback_list,\n        verbose = 1)\n    # Displays 9 generated train_generator images \n\n    '''print('Display Random Images')\n\n    # Adjust the size of your images\n    plt.figure(figsize=(20,10))\n\n    for i in range(8):\n        num = random.randint(1,7)\n        plt.subplot(3,4, i + 1)\n\n        x,y = train_generator.__getitem__(num)\n\n        plt.imshow(x[num],cmap='gray')\n        plt.axis('off')\n\n    # Adjust subplot parameters to give specified padding\n    plt.tight_layout()\n    plt.show()'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(model_history.history['loss'])\nplt.plot(model_history.history['val_loss'])\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left', bbox_to_anchor=(1,1))\nplt.show()\n\nplt.plot(model_history.history['accuracy'])\nplt.plot(model_history.history['val_accuracy'])\nplt.title('Model ACCURACY')\nplt.ylabel('ACCURACY')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left', bbox_to_anchor=(1,1))\nplt.show()","metadata":{},"execution_count":null,"outputs":[]}]}