{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = '../input/train/'\ntest_path = '../input/test/'\ntrain_labels_path = '../input/train_labels.csv'\nsample_path = '../input/sample_submission.csv'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"td = pd.read_csv(train_labels_path,dtype=str)\ntestd = pd.read_csv(sample_path,dtype=str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# d = td[td['id']=='f46f19fc90347d350431da5bfcf955d9c1418b43']['label']\n# # td.label.value_counts().sum()\n# print(np.int(d))\ntestd.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport os\nfrom os import path\nfrom random import shuffle\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport math\nfrom keras.utils import plot_model\nfrom keras.models import Model, Sequential\nfrom keras.layers import Input\nfrom keras.layers import Dense\nfrom keras.layers import Flatten\nfrom keras.layers import Activation\nfrom keras.layers import Dropout\nfrom keras.layers import Maximum\nfrom keras.layers import ZeroPadding2D\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.pooling import MaxPooling2D\nfrom keras.layers.merge import concatenate\nfrom keras import regularizers\nfrom keras.layers import BatchNormalization\nfrom keras.optimizers import Adam, SGD\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, ReduceLROnPlateau\nfrom keras.layers.advanced_activations import LeakyReLU\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_img_name = os.listdir(train_path)\nfor idx in tqdm(train_img_name):\n    assert(idx.split('.')[1]=='tif')\nprint('***All pics are tif format***')\n\ndef append_(img):\n    return img+'.tif'\ndef cut(im):\n    return im[:-4]\n\n# should only run once\ntd[\"id\"]=td[\"id\"].apply(append_)\ntestd[\"id\"]=testd[\"id\"].apply(append_)\n\n\ndef plot(img):\n    plt.figure(figsize=(30,5))\n    for i, j in enumerate(img):\n        plt.subplot(2,8,i+1)\n        plt.imshow(Image.open(test_path+j))\n        \nplot(testd.id[0:16])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# reference link https://medium.com/@vijayabhaskar96/tutorial-on-keras-flow-from-dataframe-1fd4493d237c\n# implement of flow_from_dataframe\ndatagen=ImageDataGenerator(rescale=1./255, validation_split = 0.25)\ntrain_generator=datagen.flow_from_dataframe(dataframe= td,\n                                            directory= train_path,\n                                            x_col=\"id\", y_col=\"label\",\n                                            subset=\"training\",\n                                            seed=42,\n                                            shuffle=True,\n                                            class_mode=\"binary\",\n                                            target_size=(32,32),\n                                            batch_size=32)\nvalid_generator=datagen.flow_from_dataframe(dataframe= td,\n                                            directory= train_path,\n                                            x_col=\"id\", y_col=\"label\",\n                                            subset=\"validation\",\n                                            seed=42,\n                                            shuffle=True,\n                                            class_mode=\"binary\",\n                                            target_size=(32,32),\n                                            batch_size=32)\n\ntestdatagen=ImageDataGenerator(rescale=1./255)\ntest_generator=testdatagen.flow_from_dataframe(dataframe= testd,\n                                            directory= test_path,\n                                            x_col=\"id\", y_col=None,\n                                            seed=42,\n                                            shuffle=False,\n                                            class_mode=None,\n                                            target_size=(32,32),\n                                            batch_size=32)\n\nSTEP_SIZE_TRAIN=math.ceil(train_generator.n/train_generator.batch_size)\nSTEP_SIZE_VALID=math.ceil(valid_generator.n/valid_generator.batch_size)\nSTEP_SIZE_TEST=math.ceil(test_generator.n/test_generator.batch_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same',\n                 input_shape=(32,32,3)))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(Adam(lr=0.0001),loss=\"binary_crossentropy\", metrics=[\"accuracy\"])\n\nmodel.summary()\nmodel.fit_generator(generator=train_generator,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID,\n                    epochs=40)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# You need to reset the test_generator before whenever you call the predict_generator. \n# This is important, if you forget to reset the test_generator you will get outputs in a weird order.\ntest_generator.reset()\n\npred=model.predict_generator(test_generator,steps=STEP_SIZE_TEST,verbose=1)\npred[pred>=0.5]=1\npred[pred<0.5]=0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results=pd.DataFrame({\"id\":pd.read_csv(sample_path,dtype=str)['id'].apply(cut),\"label\":np.squeeze(pred)})\nresults.to_csv(\"results.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import the modules we'll need\nfrom IPython.display import HTML\nimport pandas as pd\nimport numpy as np\nimport base64\n\n# function that takes in a dataframe and creates a text link to  \n# download it (will only work for files < 2MB or so)\ndef create_download_link(df, title = \"Download CSV file\", filename = \"submit.csv\"):  \n    csv = df.to_csv()\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\n# # create a random sample dataframe\ndf = pd.DataFrame(np.random.randn(50, 4), columns=list('ABCD'))\n\n# create a link to download the dataframe\ncreate_download_link(results)\n\n# ↓ ↓ ↓  Yay, download link! ↓ ↓ ↓ ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}