{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n# apply ignore\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# show size of train and test data\ntrain_images = os.listdir('../input/understanding_cloud_organization/train_images')\nprint(len(train_images))\ntest_images = os.listdir('../input/understanding_cloud_organization/test_images')\nprint(len(test_images))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv('../input/understanding_cloud_organization/train.csv')\n\n# Split Image_Label into ImageId and Label\nsplit = train_data['Image_Label'].str.split('_', n = 1, expand = True)\ntrain_data['id'] = split[0]\ntrain_data['label'] = split[1]\n\n# Select columns \nselected_features = [cname for cname in train_data.columns if cname not in ['Image_Label']]\n\ntrain_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# count unique labels \ntrain_data['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Dropping old Image_Label columns \ndf = train_data[['id', 'label']]\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Specify Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.python.keras.applications import ResNet50\nfrom tensorflow.python.keras.models import Sequential\nfrom tensorflow.python.keras.layers import Dense","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_classes = 4\n# imagenet easy to debug\nmy_new_model = Sequential()\n# resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5\nmy_new_model.add(ResNet50(include_top=False, pooling='avg', weights='imagenet'))\nmy_new_model.add(Dense(num_classes, activation='softmax'))\n\n# Say not to train first layer (ResNet) model. It is already trained\nmy_new_model.layers[0].trainable = False","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Compile the Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"my_new_model.compile(optimizer='sgd', \n                     loss='categorical_crossentropy', \n                     metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_new_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# for proto modelling, lets grab only part of the data\n\n# Break off validation set from training data\ntrain_df, valid_df, _, _ = train_test_split(df, df.label, train_size=0.1, \n                                                      test_size=0.1,random_state=2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Fit Model**"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimage_size = 224\ndata_generator = ImageDataGenerator(preprocess_input, validation_split=0.20)\ntrain_data_dir = '../input/understanding_cloud_organization/train_images'\n\ntrain_generator = data_generator.flow_from_dataframe(\n                                        dataframe=train_df,\n                                        directory=train_data_dir,\n                                        x_col=\"id\",\n                                        y_col=\"label\",\n                                        target_size=(image_size, image_size),\n                                        batch_size=1000,\n                                        class_mode='categorical',\n                                        subset='training')\n\nvalidation_generator = data_generator.flow_from_dataframe(\n                                        dataframe=valid_df,\n                                        directory=train_data_dir,\n                                        x_col=\"id\",\n                                        y_col=\"label\",\n                                        target_size=(image_size, image_size),\n                                        batch_size=32,\n                                        class_mode='categorical',\n                                        subset='validation')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# fit_stats below saves some statistics describing how model fitting went\n# the key role of the following line is how it changes my_new_model by fitting to data\nfit_stats = my_new_model.fit_generator(train_generator,\n                                       steps_per_epoch=2,\n                                       validation_data=validation_generator,\n                                       validation_steps=1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"** Make Predictions **<br>\nRead the file of \"test\" data. And apply model to make predictions"},{"metadata":{"trusted":true},"cell_type":"code","source":"# path to file for predictions\n#test_data_path = '../input/test.csv'\n\n# read test data\n#test_data = pd.read_csv(test_data_path)\n\n# create test_X which comes from test_data but includes only the columns used for prediction.\n# The list of columns is stored in a variable called features\n#test_X = test_data[forest_features]\n\n# make predictions used to submit. \n#test_preds = np.round(forest_model.predict(test_X)).astype(int)\n\n# The lines below shows how to save predictions in competition format\n\n#output = pd.DataFrame({'id': test_data.Id,\n#                       'label': test_preds})\n#output.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}