{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"DATASET_DIR = '../input/understanding_cloud_organization/'\nTEST_SIZE = 0.3\nRANDOM_STATE = 1024\n\nNUM_TRAIN_SAMPLES = 5 # The number of train samples used for visualization\nNUM_VAL_SAMPLES = 5 # The number of val samples used for visualization\nCOLORS = ['b', 'g', 'r', 'm'] # Color of each class\n\nimport pandas as pd\nimport os\nimport cv2\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Polygon\nfrom matplotlib.collections import PatchCollection\n\nfrom shutil import copyfile\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm_notebook\n\ndf = pd.read_csv(os.path.join(DATASET_DIR, 'train.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['Image'] = df['Image_Label'].map(lambda x: x.split('_')[0])\ndf['HavingDefection'] = df['EncodedPixels'].map(lambda x: 0 if x is np.nan else 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.array(df['HavingDefection']).reshape(-1,4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sum(np.array(df['HavingDefection'])!=1) + sum(np.array(df['HavingDefection'])!=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.array(df['HavingDefection']).shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_col = np.array(df['Image'])\nimage_files = image_col[::4]\nall_labels = np.array(df['HavingDefection']).reshape(-1, 4)\n\nnum_img_fish = np.sum(all_labels[:, 0])\nnum_img_flower = np.sum(all_labels[:, 1])\nnum_img_gravel = np.sum(all_labels[:, 2])\nnum_img_sugar = np.sum(all_labels[:, 3])\nprint('Fish: {} images'.format(num_img_fish))\nprint('Flower: {} images'.format(num_img_flower))\nprint('Gravel: {} images'.format(num_img_gravel))\nprint('Sugar: {} images'.format(num_img_sugar)) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(image_files, all_labels, test_size=TEST_SIZE, random_state=RANDOM_STATE)\ntrain_pairs = np.array(list(zip(X_train, y_train)))\ntrain_samples = train_pairs[np.random.choice(train_pairs.shape[0], NUM_TRAIN_SAMPLES, replace=False), :]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[df['Image_Label'].str.contains(train_samples[2][0])]['EncodedPixels'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def rle2mask(mask_rle, shape=(2100, 1400)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T\n\n\ndef show_samples(samples):\n    for sample in samples:\n        fig, ax = plt.subplots(figsize=(15, 10))\n        img_path = os.path.join(DATASET_DIR, 'train_images', sample[0])\n        img = cv2.imread(img_path, 1)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        # Get annotations\n        labels = df[df['Image_Label'].str.contains(sample[0])]['EncodedPixels']\n\n        patches = []\n        for idx, rle in enumerate(labels.values):\n            if rle is not np.nan:\n                mask = rle2mask(rle)\n                contours, _ = cv2.findContours(mask, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE)\n                for contour in contours:\n                    poly_patch = Polygon(contour.reshape(-1, 2), closed=True, linewidth=2, edgecolor=COLORS[idx], facecolor=COLORS[idx], fill=True)\n                    patches.append(poly_patch)\n        p = PatchCollection(patches, match_original=True, cmap=matplotlib.cm.jet, alpha=0.3)\n\n        ax.imshow(img/255)\n        ax.set_title('{} - ({})'.format(sample[0], ', '.join(sample[1].astype(np.str))))\n        ax.add_collection(p)\n        ax.set_xticklabels([])\n        ax.set_yticklabels([])\n        plt.show()\n        \n\n        \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(image_files, all_labels, test_size=TEST_SIZE, random_state=RANDOM_STATE)\ntrain_pairs = np.array(list(zip(X_train, y_train)))\ntrain_samples = train_pairs[np.random.choice(train_pairs.shape[0], NUM_TRAIN_SAMPLES, replace=False), :]\n\nshow_samples(train_samples)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_files\nall_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['EncodedPixels']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['HavingDefection']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.columns\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(image_files, all_labels, test_size=TEST_SIZE, random_state=RANDOM_STATE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('X_train:', X_train.shape)\nprint('y_train:', y_train.shape)\nprint('X_val:', X_val.shape)\nprint('y_val:', y_val.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras_preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, Activation, Flatten, Dropout, BatchNormalization\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras import regularizers, optimizers\nimport pandas as pd\nimport numpy as np\nfrom keras.models import Sequential\n\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same',\n                 input_shape=(256,256,3)))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(4, activation='sigmoid'))\nmodel.compile(optimizers.rmsprop(lr=0.0001, decay=1e-6),loss=\"binary_crossentropy\",metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_files.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = {'Image' : image_files , 'Fish' : all_labels[:,0] , 'Flower' : all_labels[:,1] , 'Gravel' : all_labels[:,2] , 'Sugar' : all_labels[:,3] }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_new = pd.DataFrame(data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_new.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns=[\"Fish\", \"Flower\", \"Gravel\", \"Sugar\"]\ndatagen=ImageDataGenerator(rescale=1./255.)\ntest_datagen=ImageDataGenerator(rescale=1./255.)\ntrain_generator=datagen.flow_from_dataframe(dataframe=df_new[:4436],directory=\"/kaggle/input/understanding_cloud_organization/train_images\",x_col=\"Image\",y_col=columns,batch_size=32,seed=42,shuffle=True,class_mode=\"other\")\n\nvalid_generator=test_datagen.flow_from_dataframe(dataframe=df_new[4436:5546],directory=\"/kaggle/input/understanding_cloud_organization/train_images\",x_col=\"Image\",y_col=columns,batch_size=32,seed=42,shuffle=True,class_mode=\"other\")\n\n#test_generator=test_datagen.flow_from_dataframe(dataframe=df_new[4990:5546],directory=\"/kaggle/input/understanding_cloud_organization/train_images\",x_col=\"Image\",y_col=columns,batch_size=32,seed=42,shuffle=True,class_mode=\"other\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\n#STEP_SIZE_TEST=test_generator.n//test_generator.batch_size\nmodel.fit_generator(generator=train_generator,steps_per_epoch=STEP_SIZE_TRAIN,validation_data=valid_generator,validation_steps=STEP_SIZE_VALID,epochs=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#test_generator\ntest_generator.reset()\npred=model.predict_generator(test_generator,steps=STEP_SIZE_TEST,verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df1 = pd.read_csv(os.path.join(DATASET_DIR, 'val.csv'))\ndf1['Image'] = df1['Image_Label'].map(lambda x: x.split('_')[0])\ndf1['HavingDefection'] = df1['EncodedPixels'].map(lambda x: 0 if x is np.nan else 1)\nimage_col1 = np.array(df1['Image'])\nimage_files1 = image_col1[::4]\nall_labels1 = np.array(df1['HavingDefection']).reshape(-1, 4)\n\nnum_img_fish1 = np.sum(all_labels1[:, 0])\nnum_img_flower1 = np.sum(all_labels1[:, 1])\nnum_img_gravel1 = np.sum(all_labels1[:, 2])\nnum_img_sugar1 = np.sum(all_labels1[:, 3])\nprint('Fish: {} images'.format(num_img_fish))\nprint('Flower: {} images'.format(num_img_flower))\nprint('Gravel: {} images'.format(num_img_gravel))\nprint('Sugar: {} images'.format(num_img_sugar))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_pairs = np.array(list(zip(X_train, y_train)))\ntrain_samples = train_pairs[np.random.choice(train_pairs.shape[0], NUM_TRAIN_SAMPLES, replace=False), :]\nfor sample in train_samples:\n    labels = df[df['Image_Label'].str.contains(sample[0])]['EncodedPixels']\n    print(labels)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}