{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport random\nimport os\nprint(os.listdir(\"../input\"))\nfrom tqdm import tqdm\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"0e11077cb53bd8fa787ecaf5340897a325503bef"},"cell_type":"code","source":"# EDA\n\ndf =  pd.read_csv(\"../input/train_labels.csv\")\ndf.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"01cbb89f8bd438b9e47fb3b25b7750b063ddfe52"},"cell_type":"code","source":"each_label = df.groupby(\"label\").count()\neach_label = each_label.rename(columns = {\"id\" : \"count\"})\neach_label = each_label.sort_values(\"count\", ascending=False)\neach_label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"22cf79f83f113303ae6e8a637d82f91d4000d45e"},"cell_type":"code","source":"pos_df = df[df.label==1]\nneg_df = df[df.label==0]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d75ec3ffdb486ad9e26176b03d4e2b61811ef9f2"},"cell_type":"code","source":"sample_size = 65000\nrandom_images_pos = random.sample(pos_df.id.tolist(), sample_size)\nrandom_images_neg = random.sample(neg_df.id.tolist(), sample_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e9dc10b7d7f6c1ba3b326218f2ce2c4c324fdf8"},"cell_type":"code","source":"id_ls = []\nid_ls.extend(random_images_neg)\nid_ls.extend(random_images_pos)\nlen(id_ls)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a176abe81f9251d6567bdb961ac8ac0512618cd"},"cell_type":"code","source":"new_df = pd.DataFrame({\"id\":id_ls})\nnew_df = pd.merge(new_df, df, how='inner', on=['id'])\ndf = new_df.sample(frac=1)\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad8fc2376a8ad197f8cd5b7dfbf424c0dc53da8b"},"cell_type":"code","source":"import cv2\ndef load_img(i,path):\n    im = cv2.imread(path+i+\".tif\")\n    im = cv2.resize(im,(71,71))\n    return im/255\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ccd1e07e39baba2ff269cd7bd008dabda0a3c91"},"cell_type":"code","source":"# xception\nimport keras\nfrom sklearn.model_selection import train_test_split\nfrom keras.applications import Xception\nfrom keras.layers import Dense,Flatten\nfrom keras import Sequential\nfrom keras.models import Model\nfrom keras.optimizers import adam\n# Parameter\nnum_class = 2\nim_size = 256\n\nbase_model = Xception(weights='imagenet', include_top=False, input_shape=(im_size, im_size, 3))\n\n\n# Add a new top layer\nx = base_model.output\nx = Flatten()(x)\npredictions = Dense(num_class, activation='softmax')(x)\n\n# This is the model we wi`l train\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# First: train only the top layers (which were randomly initialized)\nfor layer in base_model.layers:\n    layer.trainable = False\n\nmodel.compile(adam(lr=0.00001),loss='categorical_crossentropy', \n              metrics=[\"accuracy\"])\n\ncallbacks_list = [keras.callbacks.EarlyStopping(monitor='val_acc', patience=3, verbose=1)]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"655024031ea60bfa58bd06996d9d521a6f6fbd98"},"cell_type":"code","source":"\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n# from keras.preprocessing.image import ImageDataGenerator\ntrain_datagen = ImageDataGenerator(\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest')\n\ntest_datagen=ImageDataGenerator()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51cc5761398a5d72c15c42f500441fd41358049e"},"cell_type":"code","source":"train_df,test_df = train_test_split( new_df, test_size=0.15, random_state=11)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9685dd0ccf04863f359c50e601ad34071557fb69"},"cell_type":"code","source":"train_gen = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = \"../input/train/\",\n    x_col = \"id\",\n    y_col = \"label\",\n    has_ext = False,\n    shuffle= True)\n\ntest_gen = test_datagen.flow_from_dataframe(\n    dataframe = test_df,\n    directory = \"../input/train/\",\n    x_col = \"id\",\n    y_col = \"label\",\n    has_ext = False,\n    shuffle= False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc2dd09dd6a66768d441290c704dc25b34521275"},"cell_type":"code","source":"model.fit_generator(train_gen,steps_per_epoch=30,\n                    validation_data=test_gen,\n                    validation_steps=30,\n                    epochs=25, \n                    verbose=1)\nmodel.save(\"cancerDetectionXception.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a5a85940ebbc61ff09a5401fa1dee3ffbd000e5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}