{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Exploratory Analysis of Satellite Cloud Data"},{"metadata":{},"cell_type":"markdown","source":"### Importing Libraries"},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install pyfpgrowth","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\nimport time\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom tqdm import tqdm_notebook\nimport pyfpgrowth as fpg\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '../input/understanding_cloud_organization/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('{}//train.csv'.format(path))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_image_path = '{}//train_images//'.format(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Training Data shape {}'.format(train.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[['Image_ID','Image_Label']] = train.Image_Label.str.split('_', expand=True) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.to_csv('train_processsed.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# creating another dataframe with redudant label entries removed\nlabelcount = train[['Image_ID', 'Image_Label', 'EncodedPixels']].groupby('Image_ID').apply(lambda x: x.dropna()['Image_Label'].values).reset_index()\nlabelcount = labelcount.rename(columns = {0: 'labels'})\nlabelcount['label_counts'] = labelcount['labels'].apply(lambda x: len(x))            ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labelcount.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_hist(df, col):\n    ax = df[col].value_counts().plot(kind = 'bar', figsize=(10,7),\n                                        fontsize=10);\n    ax.set_alpha(0.8)\n\n    # create a list to collect the plt.patches data\n    totals = []\n\n    # find the values and append to list\n    for i in ax.patches:\n        totals.append(i.get_width())\n\n    # set individual bar lables using above list\n    total = sum(totals)\n\n    # set individual bar lables using above list\n    for i in ax.patches:\n        # get_width pulls left or right; get_y pushes up or down\n        ax.text(i.get_x()+.1, i.get_height()+.5, str(i.get_height()), fontsize=15,\n    color='black')\n        \n    return ax","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = get_hist(labelcount, 'label_counts')\nax.set_title(\"Histogram of Label Counts per Image\", fontsize=18)\nax.set_xlabel(\"Number of occurrences\", fontsize=18)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = get_hist(train.dropna(), 'Image_Label')\nax.set_title(\"Histogram of Labels (with valid masks)\", fontsize=18)\nax.set_xlabel(\"Label\", fontsize=18);","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Label Associations"},{"metadata":{"trusted":true},"cell_type":"code","source":"patterns = fpg.find_frequent_patterns(labelcount['labels'], 2)\npatternsdf = pd.DataFrame({'Label Association': list(patterns.keys()), 'Occurrences': list(patterns.values())})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f = plt.figure(figsize = (15,10))\nax = patternsdf.plot(x = 'Label Association', y = 'Occurrences', kind = 'bar')\nfor i in ax.patches:\n    ax.text(i.get_x()-0.2, i.get_height()+.5, str(i.get_height()), fontsize=10, color='black')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rules = fpg.generate_association_rules(patterns, 0.3)\nrulesdf = pd.DataFrame({'Association Rules': list(rules.keys()), \n                        'Labels': [0]*len(list(rules.keys())), 'Probabilities': list(rules.values())})\nrulesdf.loc[:, 'Labels'] = rulesdf['Probabilities'].apply(lambda x: x[0][0])\nrulesdf.loc[:, 'Probabilities'] = rulesdf['Probabilities'].apply(lambda x: x[1])\nrulesdf = rulesdf.sort_values('Probabilities', ascending = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rulesdf","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image Analysis"},{"metadata":{"trusted":true},"cell_type":"code","source":"# run length encoding function\ndef rle_decode(mask,shape=(1400,2100)):\n    \n    s=mask.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts-=1\n    end=starts+lengths\n    img=np.zeros(shape[0]*shape[1],dtype=np.uint8)\n    for l,m in zip(starts,end):\n        img[l:m]=1\n    return img.reshape(shape[0],shape[1],order='F')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.loc[:, 'MaskArea'] = train['EncodedPixels'].apply(lambda x: np.sum(rle_decode(str(x))) if not pd.isna(x) else 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# distribution plots for mask areas for different labels\nf, ax = plt.subplots(1, 1, figsize = (10, 7))\nsns.distplot(train[(train['Image_Label'] == 'Fish') & \n                   (train['MaskArea'] > 0)]['MaskArea'], kde=True, hist=False, ax = ax, color = 'red')\nsns.distplot(train[(train['Image_Label'] == 'Flower') & \n                   (train['MaskArea'] > 0)]['MaskArea'], kde=True, hist=False, ax = ax, color = 'blue')\nsns.distplot(train[(train['Image_Label'] == 'Gravel') & \n                   (train['MaskArea'] > 0)]['MaskArea'], kde=True, hist=False, ax = ax, color = 'green')\nsns.distplot(train[(train['Image_Label'] == 'Sugar') & \n                   (train['MaskArea'] > 0)]['MaskArea'], kde=True, hist=False, ax = ax, color = 'black')\nax.legend(labels=['Fish', 'Flower', 'Gravel', 'Sugar'])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### References:\n\n1. https://www.kaggle.com/ekhtiar/eda-find-me-in-the-clouds"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.4"}},"nbformat":4,"nbformat_minor":1}