{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport xgboost as xgb\nimport time\nimport tensorflow as tf\nimport tensorflow.keras as keras\nfrom sklearn.preprocessing import StandardScaler\nimport math\nimport sklearn\nfrom sklearn.ensemble import RandomForestClassifier\nimport warnings\nimport eli5\nimport datatable as dt\nimport matplotlib.pyplot as plt\nimport os\nimport csv\nimport cv2\nfrom tensorflow.keras.preprocessing import image\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.models import Model\nimport tensorflow_addons as tfa\nfrom numpy import copy\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (root, dirs, files) in os.walk('../input/hpa-single-cell-image-classification/train'):\n    #image_dir=files\n    i=len(files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('../input/hpa-single-cell-image-classification/train.csv')\ndf.count()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_numpy=df.to_numpy()\nprint(train_numpy.shape)\nprint(train_numpy[0][0])\nprint(train_numpy[0][1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_numpy_red=copy(train_numpy)\nprint(train_numpy_red.shape)\n\n\nfor i in range(len(train_numpy)):\n    train_numpy_red[i]=np.array([(train_numpy[i][0]+'_red.png'), train_numpy[i][1]])\n#print(train_numpy_red.shape)\n#print(train_numpy_red[0])\n#print(train_numpy_red[1])\n\n#print(train_numpy[0])\n#print(train_numpy[1])\n\ntrain_numpy_blue=copy(train_numpy)\n#print(train_numpy_blue.shape)\n\n\nfor i in range(len(train_numpy)):\n    train_numpy_blue[i]=np.array([(train_numpy[i][0]+'_blue.png'), train_numpy[i][1]])\n#print(train_numpy_blue.shape)\n#print(train_numpy_blue[0])\n#print(train_numpy_blue[1])\n\n#print(train_numpy[0])\n#print(train_numpy[1])\n\n\ntrain_numpy_green=copy(train_numpy)\n#print(train_numpy_blue.shape)\n\n\nfor i in range(len(train_numpy)):\n    train_numpy_green[i]=np.array([(train_numpy[i][0]+'_green.png'), train_numpy[i][1]])\n#print(train_numpy_green.shape)\n#print(train_numpy_green[0])\n#print(train_numpy_green[1])\n\n#print(train_numpy[0])\n#print(train_numpy[1])\n\ntrain_numpy_yellow=copy(train_numpy)\n#print(train_numpy_yellow.shape)\n\n\nfor i in range(len(train_numpy)):\n    train_numpy_yellow[i]=np.array([(train_numpy[i][0]+'_yellow.png'), train_numpy[i][1]])\n#print(train_numpy_yellow.shape)\n#print(train_numpy_yellow[0])\n#print(train_numpy_yellow[1])\n\n#print(train_numpy[0])\n#print(train_numpy[1])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_numpy_new=copy(train_numpy_blue)\ntrain_numpy_new=np.append(train_numpy_new, train_numpy_yellow, axis=0)\ntrain_numpy_new=np.append(train_numpy_new, train_numpy_green, axis=0)\ntrain_numpy_new=np.append(train_numpy_new, train_numpy_red, axis=0)\nprint(train_numpy_new.shape)\n#print(train_numpy_new[0])\n#print(train_numpy_new[21805])\n#print(train_numpy_new[21806])\n#print(train_numpy_new[43611])\n#print(train_numpy_new[43612])\n#print(train_numpy_new[65417])\n#print(train_numpy_new[65418])\n#print(train_numpy_new[87223])\n\n\ntrain_df_new=pd.DataFrame(data=train_numpy_new, columns=['id', 'labels'])\n#print(train_df_new.count())\n#print(train_df_new.head(10))\n#print(train_df_new.tail(10))\n\ntrain_df_new['labels'] = train_df_new['labels'].apply(lambda string: string.split('|'))\n#print(train_df_new.count())\n#print(train_df_new.head(10))\n#print(train_df_new.tail(10))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prob_threshold=0.4\nimg_shape=(256, 256, 3)\nn_ohe=19\nn_h, n_w, n_c=img_shape\nbatch_size=256\nvalidation_split=0.2\n\nalpha_lr1=0.0005\nalpha_lr2=0.0001\n\nepochs_num1=1\nepochs_num2=1\n\n#Second iteration of model run\ntrainable_mid=11\nnon_trainable_end=22","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = keras.preprocessing.image.ImageDataGenerator(rescale=1/255.0,\n                                                        preprocessing_function=None,\n                                                        data_format=None,\n                                                        validation_split=validation_split)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time=time.time()\n\ntrain_data = datagen.flow_from_dataframe(\n    train_df_new,\n    directory='../input/hpa-single-cell-image-classification/train',\n    x_col=\"id\",\n    y_col= 'labels',\n    color_mode=\"rgb\",\n    target_size = (n_h, n_w),\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    seed=40,\n    subset='training')\n\nend_time=time.time()\n\nprint('Time taken is ', (end_time-start_time)/60, ' mins')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_data = datagen.flow_from_dataframe(\n    train_df_new,\n    directory='../input/hpa-single-cell-image-classification/train',\n    x_col=\"id\",\n    y_col= 'labels',\n    color_mode=\"rgb\",\n    target_size = (n_h, n_w),\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    seed=40,\n    subset='validation')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_path = '../input/keras-pretrained-models/vgg16_weights_tf_dim_ordering_tf_kernels_notop.h5'\nVGG16_MODEL = keras.applications.VGG16(weights=weights_path ,include_top=False, input_shape=(n_h, n_w, n_c))\nx=VGG16_MODEL.output\nx=GlobalAveragePooling2D()(x)\nx = Dense(1024, activation='relu')(x)\nprediction=Dense(n_ohe, activation='sigmoid')(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in VGG16_MODEL.layers:\n    layer.trainable=False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Model(inputs=VGG16_MODEL.input, outputs=prediction)\nf1_score = tfa.metrics.F1Score(num_classes=n_ohe, threshold=prob_threshold, average='micro')\nmodel.compile(tf.keras.optimizers.Adam(learning_rate=alpha_lr1) , loss='binary_crossentropy', metrics=[f1_score])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time=time.time()\n\nmodel.fit(train_data, epochs=epochs_num1, validation_data=validation_data)\n\nend_time=time.time()\n\nprint('Time taken is ', (end_time-start_time)/60., ' mins'  )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\nimport cv2\n\nsubm_array=np.array([])\n\nfor dirname, _, filenames in os.walk('../input/hpa-single-cell-image-classification/test'):\n    for filename in filenames:\n        #filename_arr.append(filename)\n        img=cv2.imread(os.path.join('../input/hpa-single-cell-image-classification/test/', filename))\n        img_shape=list(img.shape)\n        \n        subm = np.array( [[filename, img_shape[0], img_shape[1],  'a']])\n        #print(subm.shape)\n        #subm_array= np.append(subm_array, subm)\n        if i == 0:\n            subm_array=subm\n        else:\n            subm_array= np.append(subm_array, subm, axis=0)\n        i+=1\n\n#print(subm_array)\nprint(subm_array.shape)\ntest_data_size=subm_array.shape[0]\n\ntest = pd.DataFrame(subm_array, columns = ['id', 'height', 'width', 'labels'])\n\ndatagen = keras.preprocessing.image.ImageDataGenerator(rescale=1/255.0,\n                                                        preprocessing_function=None,\n                                                        data_format=None,\n                                                    )\n\n\n\n\ntest_data = datagen.flow_from_dataframe(\n    test,\n    directory='../input/hpa-single-cell-image-classification/test/',\n    x_col=\"id\",\n    y_col= 'labels',\n    color_mode=\"rgb\",\n    target_size = (n_h, n_w),\n    class_mode=\"raw\",\n    batch_size=batch_size,\n    shuffle=False,\n    seed=40\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_test=model.predict(test_data)\nprint(type(predict_test))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}