{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Some CNN Bois\n\nFirst, thanks for some useful notebook\n\n\nhttps://www.kaggle.com/dimakyn/multi-label-keras\n\nhttps://github.com/lmoroney/dlaicourse\n\nhttps://github.com/salmanhiro/Galaxy-Zoo-CNN\n\nThen I would like to hear some music\n\n\n[![IMAGE ALT TEXT HERE](https://img.youtube.com/vi/z9mH-OZ2B-Y/0.jpg)](https://www.youtube.com/watch?v=z9mH-OZ2B-Y)\n"},{"metadata":{},"cell_type":"markdown","source":"# Loading"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#Ugh, so long\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/imet-2020-fgvc7/train.csv')\ndf_label = pd.read_csv('/kaggle/input/imet-2020-fgvc7/labels.csv')\nsubmission = pd.read_csv('/kaggle/input/imet-2020-fgvc7/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['id'] += '.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_label.head(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['id'] += '.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head(1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Gonna make the `attribute_ids` to list\n\nThanks to [dimakyn](https://www.kaggle.com/dimakyn/multi-label-keras) idea"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train[\"attribute_ids\"] = df_train[\"attribute_ids\"].apply(lambda x:x.split())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%matplotlib inline\n\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Take a look"},{"metadata":{"trusted":true},"cell_type":"code","source":"img = mpimg.imread('/kaggle/input/imet-2020-fgvc7/train/000040d66f14ced4cdd18cd95d91800f.png')\nplt.imshow(img)\nplt.axis('Off')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras_preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Construct Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"training_datagen = ImageDataGenerator(rescale = 1./255,\n                                      rotation_range=180,\n                                      width_shift_range=0.2,\n                                      height_shift_range=0.2,\n                                      shear_range=0.2,\n                                      zoom_range=0.2,\n                                      horizontal_flip=True,\n                                      fill_mode='nearest')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255.)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator = training_datagen.flow_from_dataframe(dataframe=df_train,\n                                                       directory='/kaggle/input/imet-2020-fgvc7/train/',\n                                                       x_col='id',\n                                                       y_col='attribute_ids',\n                                                       batch_size=128,\n                                                       seed=17,\n                                                       shuffle=True,\n                                                       class_mode=\"categorical\",\n                                                       target_size=(128,128))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_generator = test_datagen.flow_from_dataframe(dataframe=submission,\n                                                       directory='/kaggle/input/imet-2020-fgvc7/test/',\n                                                       x_col='id',\n                                                       batch_size=1,\n                                                       seed=17,\n                                                       shuffle=False,\n                                                       class_mode=None,\n                                                       target_size=(128,128))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Construct Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.layers import Dense, Conv2D, MaxPooling2D, Flatten, Dropout, BatchNormalization\nfrom keras.callbacks import EarlyStopping\nfrom keras.models import Sequential","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_shape = (128, 128, 3)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(32, (3, 3), padding=\"same\",input_shape=input_shape, activation = 'relu'))\nmodel.add(BatchNormalization(axis=-1, momentum=0.99, epsilon=0.001))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(32, (3, 3), padding=\"same\", activation = 'relu'))\nmodel.add(BatchNormalization(axis=-1, momentum=0.99, epsilon=0.001))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.2))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation = 'relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Dense(3471, activation='sigmoid')) \n\nmodel.compile(optimizer = 'adam',loss=\"binary_crossentropy\",metrics=[\"accuracy\"])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training"},{"metadata":{"trusted":true},"cell_type":"code","source":"early_stopping_callback = EarlyStopping(monitor='val_loss', patience=10)\n\nmodel.fit_generator(generator = train_generator,\n                    steps_per_epoch = train_generator.n//train_generator.batch_size,\n                    callbacks = [early_stopping_callback],\n                    epochs = 3,\n                    verbose = 1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Predicting"},{"metadata":{},"cell_type":"markdown","source":"I use the predict generator as used in https://www.kaggle.com/dimakyn/multi-label-keras"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_generator.reset()\npred = model.predict_generator(test_generator,\n                               steps=test_generator.n//test_generator.batch_size,\n                               verbose=1)\n\npred_bool = (pred >0.2)\n\npredictions=[]\n\nlabels = train_generator.class_indices\n\nlabels = dict((v,k) for k,v in labels.items())\n\nfor row in pred_bool:\n    l=[]\n    for index,cls in enumerate(row):\n        if cls:\n            l.append(labels[index])\n    predictions.append(\" \".join(l))\n    \nfilenames = test_generator.filenames\n\nresults = pd.DataFrame({\"id\":filenames,\"attribute_ids\":predictions})\nresults[\"id\"] = results[\"id\"].apply(lambda x:x.split(\".\")[0])\nresults.to_csv(\"submission.csv\",index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}