{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\"\"\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-01T02:26:21.368134Z","iopub.execute_input":"2021-12-01T02:26:21.368498Z","iopub.status.idle":"2021-12-01T02:26:21.375407Z","shell.execute_reply.started":"2021-12-01T02:26:21.368465Z","shell.execute_reply":"2021-12-01T02:26:21.374695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading in the necessary libraries","metadata":{}},{"cell_type":"code","source":"# Loading the dependencies\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport os\nimport random\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:13:30.851008Z","iopub.execute_input":"2021-12-01T06:13:30.851339Z","iopub.status.idle":"2021-12-01T06:13:30.856983Z","shell.execute_reply.started":"2021-12-01T06:13:30.851304Z","shell.execute_reply":"2021-12-01T06:13:30.855836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the data","metadata":{}},{"cell_type":"code","source":"# Loading csv file\ndf = pd.read_csv(\"../input/understanding_cloud_organization/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:21.391112Z","iopub.execute_input":"2021-12-01T02:26:21.391631Z","iopub.status.idle":"2021-12-01T02:26:23.145174Z","shell.execute_reply.started":"2021-12-01T02:26:21.391592Z","shell.execute_reply":"2021-12-01T02:26:23.144394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.150082Z","iopub.execute_input":"2021-12-01T02:26:23.150541Z","iopub.status.idle":"2021-12-01T02:26:23.163477Z","shell.execute_reply.started":"2021-12-01T02:26:23.150502Z","shell.execute_reply":"2021-12-01T02:26:23.161417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the data","metadata":{}},{"cell_type":"code","source":"# Exploring the data\n# But this is the total number of labels that\n# can be assigned to whole of the dataset\nprint(f\"The number of data points {len(df)}\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.164927Z","iopub.execute_input":"2021-12-01T02:26:23.165238Z","iopub.status.idle":"2021-12-01T02:26:23.170698Z","shell.execute_reply.started":"2021-12-01T02:26:23.165210Z","shell.execute_reply":"2021-12-01T02:26:23.169925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Null values in each of the columns\ndf.isna().sum().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.172089Z","iopub.execute_input":"2021-12-01T02:26:23.172670Z","iopub.status.idle":"2021-12-01T02:26:23.374645Z","shell.execute_reply.started":"2021-12-01T02:26:23.172631Z","shell.execute_reply":"2021-12-01T02:26:23.373983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Percentage of null values\nsize = [len(df)-df.EncodedPixels.count(),df.EncodedPixels.count()]\nplt.figure(figsize=(8,8))\nplt.pie(size, labels=[\"empty\",\"Non-empty\"],explode=(0,0.1), autopct=\"%1.1f\")\nplt.title(\"Null value percentage\")  ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.376266Z","iopub.execute_input":"2021-12-01T02:26:23.376860Z","iopub.status.idle":"2021-12-01T02:26:23.483206Z","shell.execute_reply.started":"2021-12-01T02:26:23.376823Z","shell.execute_reply":"2021-12-01T02:26:23.482485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Replacing the null values with the -1","metadata":{}},{"cell_type":"code","source":"# Replacing the nan values with 0s\ndf[\"EncodedPixels\"] = df['EncodedPixels'].fillna(-1)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.484426Z","iopub.execute_input":"2021-12-01T02:26:23.484898Z","iopub.status.idle":"2021-12-01T02:26:23.495244Z","shell.execute_reply.started":"2021-12-01T02:26:23.484862Z","shell.execute_reply":"2021-12-01T02:26:23.494482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating a new columns with label\ndf[\"Label\"] = df[\"Image_Label\"].apply(lambda x: x.split(\"_\")[1])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.496503Z","iopub.execute_input":"2021-12-01T02:26:23.497013Z","iopub.status.idle":"2021-12-01T02:26:23.529666Z","shell.execute_reply.started":"2021-12-01T02:26:23.496975Z","shell.execute_reply":"2021-12-01T02:26:23.529001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating an new feature with just the image names\ndf[\"Image_name\"] = df[\"Image_Label\"].apply(lambda x: x.split(\"_\")[0])\n#df.drop(\"Image_Label\",axis=1,inplace=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.532845Z","iopub.execute_input":"2021-12-01T02:26:23.533273Z","iopub.status.idle":"2021-12-01T02:26:23.562581Z","shell.execute_reply.started":"2021-12-01T02:26:23.533237Z","shell.execute_reply":"2021-12-01T02:26:23.561896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets check the number of each corresponding labels\ndef check_num(label):\n  return df[(df[\"Label\"]==label) & (df[\"EncodedPixels\"]!=-1)][\"EncodedPixels\"].count()\n\nvalues = {}\nfor i in df.Label.unique():\n  values[i] = check_num(i)\n\nprint(values)\nplt.title(\"Number of each classes\")\npd.Series(values).plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.563865Z","iopub.execute_input":"2021-12-01T02:26:23.564495Z","iopub.status.idle":"2021-12-01T02:26:23.808686Z","shell.execute_reply.started":"2021-12-01T02:26:23.564456Z","shell.execute_reply":"2021-12-01T02:26:23.807972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a dataframe of images and classes in each images\ndef dummy_var(label):\n  values = []\n  df_temp = df[df[\"Label\"]==label]\n  df_temp[\"Dummy\"] = df_temp[\"EncodedPixels\"].apply(lambda x: 1 if x!=-1 else 0)\n  return list(df_temp[\"Dummy\"])\n\ndf_images = pd.DataFrame()\ndf_images[\"Image\"] = df[\"Image_name\"].unique()\nfor i in df[\"Label\"].unique():\n  df_images[i] = dummy_var(i)\n\ndf_images.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.810236Z","iopub.execute_input":"2021-12-01T02:26:23.810758Z","iopub.status.idle":"2021-12-01T02:26:23.879526Z","shell.execute_reply.started":"2021-12-01T02:26:23.810719Z","shell.execute_reply":"2021-12-01T02:26:23.878670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of images available for us\nprint(f\"Number of images: {len(df_images)}\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.881032Z","iopub.execute_input":"2021-12-01T02:26:23.881315Z","iopub.status.idle":"2021-12-01T02:26:23.885789Z","shell.execute_reply.started":"2021-12-01T02:26:23.881271Z","shell.execute_reply":"2021-12-01T02:26:23.885092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of maps per image","metadata":{}},{"cell_type":"code","source":"# Number of detections per images\ndf_images[\"Total\"] = df_images[\"Fish\"]+df_images[\"Flower\"]+df_images[\"Gravel\"]+df_images[\"Sugar\"]\nplt.title(\"Number of labels per image\")\nsns.countplot(df_images[\"Total\"])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:23.887325Z","iopub.execute_input":"2021-12-01T02:26:23.887915Z","iopub.status.idle":"2021-12-01T02:26:24.103660Z","shell.execute_reply.started":"2021-12-01T02:26:23.887879Z","shell.execute_reply":"2021-12-01T02:26:24.102834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating a new df","metadata":{}},{"cell_type":"code","source":"# Create one column for each mask\ntrain_df = pd.pivot_table(df, index=['Image_name'], values=['EncodedPixels'], columns=['Label'], aggfunc=np.min).reset_index()\ntrain_df.columns = ['image', 'Fish_mask', 'Flower_mask', 'Gravel_mask', 'Sugar_mask']\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:24.104950Z","iopub.execute_input":"2021-12-01T02:26:24.105500Z","iopub.status.idle":"2021-12-01T02:26:25.866740Z","shell.execute_reply.started":"2021-12-01T02:26:24.105455Z","shell.execute_reply":"2021-12-01T02:26:25.865923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the segmentation masks","metadata":{}},{"cell_type":"code","source":"# dimenesions of image \nwidth = 2100\nheight = 1400","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:25.868298Z","iopub.execute_input":"2021-12-01T02:26:25.868688Z","iopub.status.idle":"2021-12-01T02:26:25.873090Z","shell.execute_reply.started":"2021-12-01T02:26:25.868653Z","shell.execute_reply":"2021-12-01T02:26:25.872193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function to decode the segmentation maps","metadata":{}},{"cell_type":"code","source":"# Function to decode the encoded pixels\ndef decode_pixels(pix, rows=2100, cols=1400,label=255):\n  # coverting the string into a list of numbers\n  rle_numbers = [int(num_string) for num_string in pix.split(' ')]\n  # Coverting them into starting index and length pairs\n  rle_pairs = np.array(rle_numbers).reshape(-1,2)\n  # Creating a blank image in form of a single row array\n  img = np.zeros(rows*cols, dtype=np.uint8)\n\n  # Setting the segmented pixels in the img\n  for ind, length in rle_pairs:\n    ind -= 1\n    img[ind:ind+length] = label\n  img = img.reshape(rows,cols)\n  img = img.T\n  return img\n","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:20:26.245004Z","iopub.execute_input":"2021-12-01T06:20:26.245491Z","iopub.status.idle":"2021-12-01T06:20:26.252452Z","shell.execute_reply.started":"2021-12-01T06:20:26.245453Z","shell.execute_reply":"2021-12-01T06:20:26.251584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Testing the function out\nseg = decode_pixels(df[\"EncodedPixels\"][4])\nseg = cv2.resize(seg, (1050,700))\nplt.imshow(seg)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:20:34.470386Z","iopub.execute_input":"2021-12-01T06:20:34.471040Z","iopub.status.idle":"2021-12-01T06:20:34.792536Z","shell.execute_reply.started":"2021-12-01T06:20:34.471001Z","shell.execute_reply":"2021-12-01T06:20:34.791672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Displaying some random segment maps","metadata":{}},{"cell_type":"code","source":"# Sample of the segment regions\nplt.figure(figsize=(15,8))\nj = 0\nfor i in range(6):\n  plt.subplot(2,3,i+1)\n  while True:\n    if df[\"EncodedPixels\"][j]!=-1:\n      break\n    j+=1\n  plt.imshow(decode_pixels(df[\"EncodedPixels\"][j]))\n  plt.title(df[\"Image_name\"][j]+\"_\"+df[\"Label\"][j])\n  j+=1\n  plt.xticks([])\n  plt.yticks([])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:26.842388Z","iopub.execute_input":"2021-12-01T02:26:26.843485Z","iopub.status.idle":"2021-12-01T02:26:29.249606Z","shell.execute_reply.started":"2021-12-01T02:26:26.843444Z","shell.execute_reply":"2021-12-01T02:26:29.248051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading and displaying satellite images","metadata":{}},{"cell_type":"code","source":"# location of img directory\nimg_dir = \"../input/understanding_cloud_organization/train_images\"","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:29.251318Z","iopub.execute_input":"2021-12-01T02:26:29.251928Z","iopub.status.idle":"2021-12-01T02:26:29.255576Z","shell.execute_reply.started":"2021-12-01T02:26:29.251889Z","shell.execute_reply":"2021-12-01T02:26:29.254847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seeing the cloumns of train.csv\ndf.columns","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:29.257052Z","iopub.execute_input":"2021-12-01T02:26:29.257714Z","iopub.status.idle":"2021-12-01T02:26:29.268777Z","shell.execute_reply.started":"2021-12-01T02:26:29.257671Z","shell.execute_reply":"2021-12-01T02:26:29.268034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Looking at any one image name\ndf[\"Image_name\"].unique()[10]","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:29.270417Z","iopub.execute_input":"2021-12-01T02:26:29.271059Z","iopub.status.idle":"2021-12-01T02:26:29.281618Z","shell.execute_reply.started":"2021-12-01T02:26:29.271020Z","shell.execute_reply":"2021-12-01T02:26:29.280769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying one image\npath = os.path.join(img_dir,df[\"Image_name\"][0])\nimg = cv2.imread(path,1)\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:29.283257Z","iopub.execute_input":"2021-12-01T02:26:29.283826Z","iopub.status.idle":"2021-12-01T02:26:30.251027Z","shell.execute_reply.started":"2021-12-01T02:26:29.283791Z","shell.execute_reply":"2021-12-01T02:26:30.250000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Displaying some random images","metadata":{}},{"cell_type":"code","source":"# Displaying sample images\nimgs = df[\"Image_name\"].unique()[:6]\nplt.figure(figsize=(15,8))\nfor i in range(len(imgs)):\n  plt.subplot(2,3,i+1)\n  path = os.path.join(img_dir,imgs[i])\n  img = cv2.imread(path,1)\n  plt.title(imgs[i])\n  plt.imshow(cv2.cvtColor(img,cv2.COLOR_BGR2RGB))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:30.252794Z","iopub.execute_input":"2021-12-01T02:26:30.253410Z","iopub.status.idle":"2021-12-01T02:26:32.745024Z","shell.execute_reply.started":"2021-12-01T02:26:30.253367Z","shell.execute_reply":"2021-12-01T02:26:32.744394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing the mask overlaid images","metadata":{}},{"cell_type":"code","source":"# Sample image with masks overlayed\nplt.figure(figsize=(15,8))\nj = 0\nfor i in range(6):\n  plt.subplot(2,3,i+1)\n  while True:\n    if df[\"EncodedPixels\"][j]!=-1:\n      break\n    j+=1\n  seg = decode_pixels(df[\"EncodedPixels\"][j])\n  path = os.path.join(img_dir,df[\"Image_name\"][j])\n  img = cv2.imread(path,0)\n  dest = cv2.addWeighted(img, 0.8, seg, 0.4, 0.0)\n  plt.imshow(dest)\n  plt.title(df[\"Image_name\"][j])\n  j+=1\n  plt.xticks([])\n  plt.yticks([])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:32.746124Z","iopub.execute_input":"2021-12-01T02:26:32.747311Z","iopub.status.idle":"2021-12-01T02:26:34.793646Z","shell.execute_reply.started":"2021-12-01T02:26:32.747267Z","shell.execute_reply":"2021-12-01T02:26:34.791218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the data","metadata":{}},{"cell_type":"code","source":"# Function to preprocess the data \nBASE_DIR = \"../input/understanding_cloud_organization/train_images\"\nlabels = list(df[\"Label\"].unique())\ndef preprocess(df):\n  data = []\n  for i in range(len(df)):\n    if i % 100 == 0:\n      print(f\"{i} completed\")\n    path = os.path.join(BASE_DIR,df[\"image\"][i])\n    img_arr = cv2.imread(path,1)\n    img_arr = cv2.resize(img_arr,(480,384))\n    channels = []\n    for j in df.columns[1:]:\n      #print(type(df[j][i]),j,i)\n      if type(df[j][i]) is not str:\n            arr = np.zeros(384*480, dtype=np.uint8)\n            arr = arr.reshape(480,384)\n            #img = img.T\n            channels.append(arr.T)\n            continue\n      arr = decode_pixels(df[j][i],label=1)\n      arr = cv2.resize(arr,(480,384))\n      channels.append(arr)\n\n    data.append([img_arr/255,np.dstack(channels)])\n  return data","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:11:45.002936Z","iopub.execute_input":"2021-12-01T06:11:45.003215Z","iopub.status.idle":"2021-12-01T06:11:45.025782Z","shell.execute_reply.started":"2021-12-01T06:11:45.003185Z","shell.execute_reply":"2021-12-01T06:11:45.024817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity check on the output\ndata = preprocess(train_df[:5])\nlen(data)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:11:48.078668Z","iopub.execute_input":"2021-12-01T06:11:48.078933Z","iopub.status.idle":"2021-12-01T06:11:48.391900Z","shell.execute_reply.started":"2021-12-01T06:11:48.078903Z","shell.execute_reply":"2021-12-01T06:11:48.391116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the shape of image\ndata[1][0].shape","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:12:01.063301Z","iopub.execute_input":"2021-12-01T06:12:01.064055Z","iopub.status.idle":"2021-12-01T06:12:01.070105Z","shell.execute_reply.started":"2021-12-01T06:12:01.064014Z","shell.execute_reply":"2021-12-01T06:12:01.069195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the shape of segmentation mask\ndata[1][1].shape","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:12:30.983219Z","iopub.execute_input":"2021-12-01T06:12:30.984067Z","iopub.status.idle":"2021-12-01T06:12:30.991667Z","shell.execute_reply.started":"2021-12-01T06:12:30.984026Z","shell.execute_reply":"2021-12-01T06:12:30.990802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating a custom generator for batch learning","metadata":{}},{"cell_type":"code","source":"img_dir = \"../input/understanding_cloud_organization/train_images\"\nmasks_dir = \"./masks\"","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:35.694941Z","iopub.execute_input":"2021-12-01T02:26:35.695476Z","iopub.status.idle":"2021-12-01T02:26:35.699570Z","shell.execute_reply.started":"2021-12-01T02:26:35.695437Z","shell.execute_reply":"2021-12-01T02:26:35.698764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# utility function for Data Generators\nBASE_DIR = \"../input/understanding_cloud_organization/train_images\"\nlabels = list(df[\"Label\"].unique())\n\ndef preprocess1(df):\n    # To store the data\n    data = []\n    # Iterating through each of the rows in the dataframe\n    for i in range(len(df)):\n        # Getting the path of the image\n        path = os.path.join(BASE_DIR,df.iloc[i][\"image\"])\n        # Reading in the image\n        img_arr = cv2.imread(path,1)\n        # Resizing it to the proper size\n        img_arr = cv2.resize(img_arr,(480,384))\n        # To store the differnt segmentation maps\n        channels = []\n        # Getting the differnt segmentation maps\n        for j in df.columns[1:]:\n          # making an empty map if the image doesn't contain a label\n          if type(df.iloc[i][j]) is not str:\n                arr = np.zeros(384*480, dtype=np.uint8)\n                arr = arr.reshape(480,384)\n                channels.append(arr.T)\n                continue\n          # Creating the segmentation map\n          arr = decode_pixels(df.iloc[i][j],label=1)\n          # Resizing it to proper size\n          arr = cv2.resize(arr,(480,384))\n          channels.append(arr)\n        # Adding to the data list as [image, output seg map]\n        data.append([img_arr/255,np.dstack(channels)])\n    # Spliting the data into input and output\n    imgs = []\n    masks = []\n    for i, j in data:\n        imgs.append(i)\n        masks.append(j)\n\n    return np.array(imgs), np.array(masks).astype(np.float)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:35.721788Z","iopub.execute_input":"2021-12-01T02:26:35.722563Z","iopub.status.idle":"2021-12-01T02:26:35.735624Z","shell.execute_reply.started":"2021-12-01T02:26:35.722526Z","shell.execute_reply":"2021-12-01T02:26:35.734995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Custom data generators","metadata":{}},{"cell_type":"code","source":"# Creating a custom Data Generator\ndef data_gen(img_folder, df, batch_size):\n    \n    c = 0 \n    n = list(df[\"image\"])\n    while True:\n        c1 = c+batch_size\n        \n        if c1 > len(df):\n            c1 = len(df)\n        imgs, masks = preprocess1(df.iloc[c:c1])\n        c = c1\n        if c1 >= len(df):\n            c = 0\n        if imgs.shape == (batch_size,384, 480, 3) and masks.shape == (batch_size,384, 480, 4):\n            yield imgs, masks\n        else:\n            continue\n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:35.737331Z","iopub.execute_input":"2021-12-01T02:26:35.737999Z","iopub.status.idle":"2021-12-01T02:26:35.748181Z","shell.execute_reply.started":"2021-12-01T02:26:35.737964Z","shell.execute_reply":"2021-12-01T02:26:35.747442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample check to see if the code is working\nthis = data_gen(img_folder=img_dir, df=train_df, batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:35.749943Z","iopub.execute_input":"2021-12-01T02:26:35.750676Z","iopub.status.idle":"2021-12-01T02:26:35.756923Z","shell.execute_reply.started":"2021-12-01T02:26:35.750584Z","shell.execute_reply":"2021-12-01T02:26:35.756176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking\nk = 0\nfor i,j in this:\n    if k == 5:\n        break\n    print(i.shape,j.shape)\n    k+=1\n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:35.758660Z","iopub.execute_input":"2021-12-01T02:26:35.759307Z","iopub.status.idle":"2021-12-01T02:26:39.050282Z","shell.execute_reply.started":"2021-12-01T02:26:35.759270Z","shell.execute_reply":"2021-12-01T02:26:39.049498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building","metadata":{}},{"cell_type":"code","source":"# Installing the segmentation_models library\n! pip install segmentation_models","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:39.052038Z","iopub.execute_input":"2021-12-01T02:26:39.052321Z","iopub.status.idle":"2021-12-01T02:26:46.256554Z","shell.execute_reply.started":"2021-12-01T02:26:39.052270Z","shell.execute_reply":"2021-12-01T02:26:46.255519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the dependencies\nimport tensorflow as tf\nimport segmentation_models as sm\nimport glob\nimport cv2\nimport os\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport keras\n\nfrom tensorflow.keras.utils import normalize\nfrom keras.metrics import MeanIoU\n\nsm.set_framework('tf.keras')\n\nsm.framework()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:46.258170Z","iopub.execute_input":"2021-12-01T02:26:46.258412Z","iopub.status.idle":"2021-12-01T02:26:46.270216Z","shell.execute_reply.started":"2021-12-01T02:26:46.258383Z","shell.execute_reply":"2021-12-01T02:26:46.269158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting up the Hyperparamters","metadata":{}},{"cell_type":"code","source":"BACKBONE = 'efficientnetb5'\nLEARNING_RATE = 0.002\nHEIGHT = 384\nWIDTH = 480\nCHANNELS = 3\nN_CLASSES = 4\nES_PATIENCE = 10\nRLROP_PATIENCE = 3\nDECAY = 0.0001\nDECAY_DROP = 0.2\nmodel_path = f'uNet_%s_%sx%s_lr{LEARNING_RATE}.h5' % (BACKBONE, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:46.272052Z","iopub.execute_input":"2021-12-01T02:26:46.272351Z","iopub.status.idle":"2021-12-01T02:26:46.278650Z","shell.execute_reply.started":"2021-12-01T02:26:46.272316Z","shell.execute_reply":"2021-12-01T02:26:46.277738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting up the optimizer\noptim = tf.keras.optimizers.Adam(LEARNING_RATE)\n\n# Setting up the metrics\nmetrics = [sm.metrics.IOUScore(threshold=0.50),sm.metrics.FScore(threshold=0.5)]\n\n# Setting up the Callbacks\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(model_path, monitor='val_loss', mode='min', save_best_only=True, save_weights_only=True)\nes = tf.keras.callbacks.EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\nrlrop = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', mode='min', patience=RLROP_PATIENCE, factor=DECAY_DROP, min_lr=1e-6, verbose=1)\n\n# Final list of call backs\ncallback_list = [checkpoint, es, rlrop]","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:46.280560Z","iopub.execute_input":"2021-12-01T02:26:46.280744Z","iopub.status.idle":"2021-12-01T02:26:46.292520Z","shell.execute_reply.started":"2021-12-01T02:26:46.280722Z","shell.execute_reply":"2021-12-01T02:26:46.291741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the model architecture","metadata":{}},{"cell_type":"code","source":"# Defining model\nmodel = sm.Unet(backbone_name=BACKBONE, \n                encoder_weights='imagenet',\n                classes=N_CLASSES,\n                activation='sigmoid', encoder_freeze=True,\n                input_shape=(HEIGHT, WIDTH, CHANNELS))\n\n# Compiling the model\nmodel.compile(optimizer=optim, loss=sm.losses.bce_dice_loss, metrics=metrics)\n\n# Model summary\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:46.293776Z","iopub.execute_input":"2021-12-01T02:26:46.295038Z","iopub.status.idle":"2021-12-01T02:26:50.558060Z","shell.execute_reply.started":"2021-12-01T02:26:46.295009Z","shell.execute_reply":"2021-12-01T02:26:50.557198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting the training and testing data","metadata":{}},{"cell_type":"code","source":"# Total number of images\nprint(f\"Total size of data {len(train_df)}\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:50.560021Z","iopub.execute_input":"2021-12-01T02:26:50.561338Z","iopub.status.idle":"2021-12-01T02:26:50.567988Z","shell.execute_reply.started":"2021-12-01T02:26:50.561292Z","shell.execute_reply":"2021-12-01T02:26:50.567030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splitting the data into train and test","metadata":{}},{"cell_type":"code","source":"# Training and Testing Data\ntrain = train_df.iloc[0:4500]\ntest = train_df.iloc[4500:5000]\nbatch_size = 8","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:50.570239Z","iopub.execute_input":"2021-12-01T02:26:50.571332Z","iopub.status.idle":"2021-12-01T02:26:50.577889Z","shell.execute_reply.started":"2021-12-01T02:26:50.571291Z","shell.execute_reply":"2021-12-01T02:26:50.576970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating the data generators for batch learning","metadata":{}},{"cell_type":"code","source":"# Data Generators\nTrain_data_generator = data_gen(img_folder=img_dir, df=train, batch_size=batch_size)\nValidation_data_generator = data_gen(img_folder=img_dir, df=test, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:50.579567Z","iopub.execute_input":"2021-12-01T02:26:50.580595Z","iopub.status.idle":"2021-12-01T02:26:50.587585Z","shell.execute_reply.started":"2021-12-01T02:26:50.580554Z","shell.execute_reply":"2021-12-01T02:26:50.586677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the model","metadata":{}},{"cell_type":"code","source":"history = model.fit(Train_data_generator,epochs=40,\n                             steps_per_epoch=(4500//batch_size),\n                             validation_data=Validation_data_generator,\n                             validation_steps=(500//batch_size),\n                             callbacks=callback_list)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T02:26:50.589426Z","iopub.execute_input":"2021-12-01T02:26:50.590619Z","iopub.status.idle":"2021-12-01T05:04:42.136711Z","shell.execute_reply.started":"2021-12-01T02:26:50.590577Z","shell.execute_reply":"2021-12-01T05:04:42.135797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluating the model on test data","metadata":{}},{"cell_type":"code","source":"res = model.evaluate(Validation_data_generator,steps=500//16)\nprint(f\"Loss:{res[0]}\")\nprint(f\"IoU:{res[1]}\")\nprint(f\"F1:{res[2]}\")   ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T06:04:23.977769Z","iopub.execute_input":"2021-12-01T06:04:23.978312Z","iopub.status.idle":"2021-12-01T06:04:43.911895Z","shell.execute_reply.started":"2021-12-01T06:04:23.978266Z","shell.execute_reply":"2021-12-01T06:04:43.911018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Saving the best Model","metadata":{}},{"cell_type":"code","source":"model.save(f\"clouds_efficientnetb5_iouscore-{str(res[1])[:5]}.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:09:27.768787Z","iopub.execute_input":"2021-12-01T05:09:27.769068Z","iopub.status.idle":"2021-12-01T05:09:29.286186Z","shell.execute_reply.started":"2021-12-01T05:09:27.769037Z","shell.execute_reply":"2021-12-01T05:09:29.285395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualising the training of model","metadata":{}},{"cell_type":"code","source":"# All the curves together\ndf_res = pd.DataFrame(history.history)\nplt.title(\"Model Performance\")\nplt.plot(df_res)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:11:44.880326Z","iopub.execute_input":"2021-12-01T05:11:44.880573Z","iopub.status.idle":"2021-12-01T05:11:45.087681Z","shell.execute_reply.started":"2021-12-01T05:11:44.880545Z","shell.execute_reply":"2021-12-01T05:11:45.086993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Looking at the columns of the result df\ndf_res.columns","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:03.335329Z","iopub.execute_input":"2021-12-01T05:05:03.335920Z","iopub.status.idle":"2021-12-01T05:05:03.343211Z","shell.execute_reply.started":"2021-12-01T05:05:03.335873Z","shell.execute_reply":"2021-12-01T05:05:03.342333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing each curve separately","metadata":{}},{"cell_type":"code","source":"# Each of the learning curves for the model has been displayed separately\ncolors = \"bgrcy\"\nplt.figure(figsize=(15,15))\nfor i in range(len(df_res.columns)):\n    plt.subplot(4,2,i+1)\n    df_res[df_res.columns[i]].plot(color=colors[random.randint(0,4)])\n    plt.title(df_res.columns[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:18:00.354195Z","iopub.execute_input":"2021-12-01T05:18:00.354735Z","iopub.status.idle":"2021-12-01T05:18:01.373739Z","shell.execute_reply.started":"2021-12-01T05:18:00.354698Z","shell.execute_reply":"2021-12-01T05:18:01.373031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Performance of model on unseen images","metadata":{}},{"cell_type":"markdown","source":"## Post processing the output from model","metadata":{}},{"cell_type":"code","source":"# Thresholding function to be applied on the output\ndef threshold(x):\n    if x>0.5:\n        return 1\n    else:\n        return 0\n\n# Making the function applicable to a numpy array\nexp =np.vectorize(threshold)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:24:16.047405Z","iopub.execute_input":"2021-12-01T05:24:16.047680Z","iopub.status.idle":"2021-12-01T05:24:16.052089Z","shell.execute_reply.started":"2021-12-01T05:24:16.047650Z","shell.execute_reply":"2021-12-01T05:24:16.051198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to compare the predicted mask to the actual mask\n# Function simply plots the actual and predicted mask of the 4 classes \n# side by side\ndef compare_masks(actual,predicted):\n    plt.figure(figsize=(15,15))\n    j = 0\n    for i in range(8):\n        plt.subplot(4,2,i+1)\n        if (i+1)%2!=0:\n            plt.title(f\"Actual-{labels[j]}\")\n            plt.imshow(actual[:,:,j])\n        else:\n            plt.title(f\"Predicted-{labels[j]}\")\n            plt.imshow(predicted[:,:,j])\n            j+=1    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:24:17.725722Z","iopub.execute_input":"2021-12-01T05:24:17.725986Z","iopub.status.idle":"2021-12-01T05:24:17.732750Z","shell.execute_reply.started":"2021-12-01T05:24:17.725956Z","shell.execute_reply":"2021-12-01T05:24:17.732028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to predict and visualise the outputs of the models\n# Simply combine the above 2 functions together\ndef predict(df):\n    data = preprocess(df)\n    output = model.predict(data[0][0][ np.newaxis, ...])\n    output = exp(output)\n    compare_masks(data[0][1],output[0])\n    return     ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:04.889852Z","iopub.execute_input":"2021-12-01T05:05:04.890646Z","iopub.status.idle":"2021-12-01T05:05:04.897028Z","shell.execute_reply.started":"2021-12-01T05:05:04.890606Z","shell.execute_reply":"2021-12-01T05:05:04.896284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to display the actual image\ndef display_img(img_name):\n    path = os.path.join(img_dir,img_name)\n    img = cv2.imread(path,1)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:29:29.323015Z","iopub.execute_input":"2021-12-01T05:29:29.323570Z","iopub.status.idle":"2021-12-01T05:29:29.327782Z","shell.execute_reply.started":"2021-12-01T05:29:29.323533Z","shell.execute_reply":"2021-12-01T05:29:29.327028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing the model on unseen data","metadata":{}},{"cell_type":"markdown","source":"### 1:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5107])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:29:39.885531Z","iopub.execute_input":"2021-12-01T05:29:39.886266Z","iopub.status.idle":"2021-12-01T05:29:40.547744Z","shell.execute_reply.started":"2021-12-01T05:29:39.886210Z","shell.execute_reply":"2021-12-01T05:29:40.547010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Performance of model on image\ntest = pd.DataFrame(train_df.iloc[5107]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:04.898742Z","iopub.execute_input":"2021-12-01T05:05:04.899558Z","iopub.status.idle":"2021-12-01T05:05:09.923621Z","shell.execute_reply.started":"2021-12-01T05:05:04.899522Z","shell.execute_reply":"2021-12-01T05:05:09.922972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5101])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:31:40.308046Z","iopub.execute_input":"2021-12-01T05:31:40.308770Z","iopub.status.idle":"2021-12-01T05:31:40.954068Z","shell.execute_reply.started":"2021-12-01T05:31:40.308731Z","shell.execute_reply":"2021-12-01T05:31:40.953459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5101]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:09.925109Z","iopub.execute_input":"2021-12-01T05:05:09.925693Z","iopub.status.idle":"2021-12-01T05:05:11.376594Z","shell.execute_reply.started":"2021-12-01T05:05:09.925655Z","shell.execute_reply":"2021-12-01T05:05:11.375201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5105])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:32:21.293067Z","iopub.execute_input":"2021-12-01T05:32:21.293418Z","iopub.status.idle":"2021-12-01T05:32:22.006836Z","shell.execute_reply.started":"2021-12-01T05:32:21.293382Z","shell.execute_reply":"2021-12-01T05:32:22.006068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5105]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:11.378290Z","iopub.execute_input":"2021-12-01T05:05:11.378900Z","iopub.status.idle":"2021-12-01T05:05:12.953632Z","shell.execute_reply.started":"2021-12-01T05:05:11.378859Z","shell.execute_reply":"2021-12-01T05:05:12.952510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import keras","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:12.955225Z","iopub.execute_input":"2021-12-01T05:05:12.955791Z","iopub.status.idle":"2021-12-01T05:05:12.959756Z","shell.execute_reply.started":"2021-12-01T05:05:12.955752Z","shell.execute_reply":"2021-12-01T05:05:12.959062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.load_weights('../input/modelforsegmentation/clouds_iouscore-0.39.h5')","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:12.961540Z","iopub.execute_input":"2021-12-01T05:05:12.962232Z","iopub.status.idle":"2021-12-01T05:05:12.969170Z","shell.execute_reply.started":"2021-12-01T05:05:12.962120Z","shell.execute_reply":"2021-12-01T05:05:12.968033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5108])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:33:07.699612Z","iopub.execute_input":"2021-12-01T05:33:07.700035Z","iopub.status.idle":"2021-12-01T05:33:08.761175Z","shell.execute_reply.started":"2021-12-01T05:33:07.699993Z","shell.execute_reply":"2021-12-01T05:33:08.760139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5108]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:12.970920Z","iopub.execute_input":"2021-12-01T05:05:12.971769Z","iopub.status.idle":"2021-12-01T05:05:14.694660Z","shell.execute_reply.started":"2021-12-01T05:05:12.971729Z","shell.execute_reply":"2021-12-01T05:05:14.693970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5117])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:33:31.937874Z","iopub.execute_input":"2021-12-01T05:33:31.938131Z","iopub.status.idle":"2021-12-01T05:33:32.616598Z","shell.execute_reply.started":"2021-12-01T05:33:31.938103Z","shell.execute_reply":"2021-12-01T05:33:32.613706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5117]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:14.696208Z","iopub.execute_input":"2021-12-01T05:05:14.696738Z","iopub.status.idle":"2021-12-01T05:05:16.536699Z","shell.execute_reply.started":"2021-12-01T05:05:14.696698Z","shell.execute_reply":"2021-12-01T05:05:16.535972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 6","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][510])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:33:47.882410Z","iopub.execute_input":"2021-12-01T05:33:47.882661Z","iopub.status.idle":"2021-12-01T05:33:48.544390Z","shell.execute_reply.started":"2021-12-01T05:33:47.882633Z","shell.execute_reply":"2021-12-01T05:33:48.542772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[510]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:33:48.545932Z","iopub.execute_input":"2021-12-01T05:33:48.547142Z","iopub.status.idle":"2021-12-01T05:33:50.270583Z","shell.execute_reply.started":"2021-12-01T05:33:48.547104Z","shell.execute_reply":"2021-12-01T05:33:50.267654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5000])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:34:12.230001Z","iopub.execute_input":"2021-12-01T05:34:12.230480Z","iopub.status.idle":"2021-12-01T05:34:12.882507Z","shell.execute_reply.started":"2021-12-01T05:34:12.230442Z","shell.execute_reply":"2021-12-01T05:34:12.881407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5000]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:18.038371Z","iopub.execute_input":"2021-12-01T05:05:18.039193Z","iopub.status.idle":"2021-12-01T05:05:19.498704Z","shell.execute_reply.started":"2021-12-01T05:05:18.039122Z","shell.execute_reply":"2021-12-01T05:05:19.498028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5111])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:34:49.721584Z","iopub.execute_input":"2021-12-01T05:34:49.721860Z","iopub.status.idle":"2021-12-01T05:34:50.368743Z","shell.execute_reply.started":"2021-12-01T05:34:49.721828Z","shell.execute_reply":"2021-12-01T05:34:50.366217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5111]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:19.500403Z","iopub.execute_input":"2021-12-01T05:05:19.500965Z","iopub.status.idle":"2021-12-01T05:05:20.954776Z","shell.execute_reply.started":"2021-12-01T05:05:19.500902Z","shell.execute_reply":"2021-12-01T05:05:20.954031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9:","metadata":{}},{"cell_type":"code","source":"# Displaying the actual image\ndisplay_img(train_df[\"image\"][5222])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:35:10.828149Z","iopub.execute_input":"2021-12-01T05:35:10.828448Z","iopub.status.idle":"2021-12-01T05:35:11.483547Z","shell.execute_reply.started":"2021-12-01T05:35:10.828419Z","shell.execute_reply":"2021-12-01T05:35:11.482176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(train_df.iloc[5222]).T.reset_index().drop(\"index\",axis=1)\npredict(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:05:20.956373Z","iopub.execute_input":"2021-12-01T05:05:20.956920Z","iopub.status.idle":"2021-12-01T05:05:22.381934Z","shell.execute_reply.started":"2021-12-01T05:05:20.956884Z","shell.execute_reply":"2021-12-01T05:05:22.381130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading in the trained model","metadata":{}},{"cell_type":"code","source":"# Importing keras\nimport keras","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:36:43.630915Z","iopub.execute_input":"2021-12-01T05:36:43.631197Z","iopub.status.idle":"2021-12-01T05:36:43.634713Z","shell.execute_reply.started":"2021-12-01T05:36:43.631165Z","shell.execute_reply":"2021-12-01T05:36:43.633826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a model with the same architecture\nloaded_model = sm.Unet(backbone_name=BACKBONE, \n                encoder_weights='imagenet',\n                classes=N_CLASSES,\n                activation='sigmoid', encoder_freeze=True,\n                input_shape=(HEIGHT, WIDTH, CHANNELS))\n\n# COmpiling the model\nloaded_model.compile(optimizer=optim, loss=sm.losses.bce_dice_loss, metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:55:12.743701Z","iopub.execute_input":"2021-12-01T05:55:12.743974Z","iopub.status.idle":"2021-12-01T05:55:16.943596Z","shell.execute_reply.started":"2021-12-01T05:55:12.743943Z","shell.execute_reply":"2021-12-01T05:55:16.942834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loaded_model.load_weights('../input/trained-model/clouds_effientnetb5_iouscore-0.417.h5')","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:55:16.945068Z","iopub.execute_input":"2021-12-01T05:55:16.945347Z","iopub.status.idle":"2021-12-01T05:55:20.236748Z","shell.execute_reply.started":"2021-12-01T05:55:16.945315Z","shell.execute_reply":"2021-12-01T05:55:20.236014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For testing\ndef predict_loaded(df):\n    data = preprocess(df)\n    output = loaded_model.predict(data[0][0][ np.newaxis, ...])\n    output = exp(output)\n    compare_masks(data[0][1],output[0])\n    return ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:55:45.799440Z","iopub.execute_input":"2021-12-01T05:55:45.800110Z","iopub.status.idle":"2021-12-01T05:55:45.804850Z","shell.execute_reply.started":"2021-12-01T05:55:45.800074Z","shell.execute_reply":"2021-12-01T05:55:45.803894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Testing the model\ntest = pd.DataFrame(train_df.iloc[5222]).T.reset_index().drop(\"index\",axis=1)\npredict_loaded(test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T05:55:46.274945Z","iopub.execute_input":"2021-12-01T05:55:46.275501Z","iopub.status.idle":"2021-12-01T05:56:24.197638Z","shell.execute_reply.started":"2021-12-01T05:55:46.275461Z","shell.execute_reply":"2021-12-01T05:56:24.196959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}