{"cells":[{"metadata":{},"cell_type":"markdown","source":"Este notebook está preparado para descargarlo y abrirlo con Google Colab. \n## Preparar dataset\n1. Entrar a esta carpeta de Google Drive: https://drive.google.com/open?id=1TnG7nC_wwSl0p3pxPfGduPUF7VlMCWmt\n2. Hacer click sobre el nombre de la carpeta y \"Agregar a Mi Unidad\""},{"metadata":{"trusted":true},"cell_type":"code","source":"from google.colab import drive\ndrive.mount('/content/gdrive',force_remount=True)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\n\nfolder = '/content/gdrive/My Drive/kaggle-histology/'\ndf=pd.read_csv(folder+\"train_labels.csv\")\ndf['id']=[str(x)+'.tif' for x in df['id']]\ndf['label']=[str(x) for x in df['label']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"example_imgfile = df['id'][0]\n\nprint(example_imgfile)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from skimage.io import imread\nfrom matplotlib import pyplot as plt\n\nimg = imread(folder + 'train/' + example_imgfile)\nplt.imshow(img)\nprint(img.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nprint('Tamaño conjunto train',len(os.listdir(folder + \"train/\")))\nprint('Tamaño conjunto test',len(os.listdir(folder + \"test/\")))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Entrenamiento"},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfrom keras_preprocessing.image import ImageDataGenerator\ndatagen=ImageDataGenerator(rescale=1./255,validation_split=0.25)\ntrain_generator=datagen.flow_from_dataframe(dataframe=df, directory=folder + \"train/\", \n                                            x_col=\"id\", y_col=\"label\", class_mode=\"binary\", \n                                            subset='training',\n                                            target_size=(50,50), batch_size=8)\nvalid_generator=datagen.flow_from_dataframe(dataframe=df, directory=folder + \"train/\", \n                                            x_col=\"id\", y_col=\"label\", class_mode=\"binary\", \n                                            subset='validation',\n                                            target_size=(50,50), batch_size=8)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D,Activation,MaxPooling2D,Dropout,Flatten,Dense\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same',\n                 input_shape=(50,50,3)))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='softmax'))\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfrom keras import optimizers\nmodel.compile(optimizer=optimizers.rmsprop(lr=0.0001),loss=\"binary_crossentropy\", metrics=[\"accuracy\"])\nSTEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\n\nhist = model.fit_generator(generator=train_generator,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID,\n                    epochs=1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Test"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_datagen=ImageDataGenerator(rescale=1./255)\ntest_generator=datagen.flow_from_directory(directory=folder + \"test/\", \n                                            target_size=(50,50), batch_size=1)\n\nSTEP_SIZE_TEST=test_generator.n//test_generator.batch_size\npreds = [int(x[0]) for x in model.predict_generator(generator=test_generator,steps=STEP_SIZE_TEST)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['id'] = os.listdir(folder + \"test/allclasses\")\nsubmission['label'] = preds\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv(folder + \"submission.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Para hacer un submission en la competencia, descargá el archivo CSV y seguí el kernel de \"guia-submission\""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}