{"cells":[{"metadata":{"_uuid":"5bc595efe054fed55a6a1fe823be4fadce966168"},"cell_type":"markdown","source":">  ** I have used very small subset of train data. Just to feel for the data I have taken small data and played around with it. And here I tried to continue up the flow of whole process in  setting up train and test data.**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Import \nimport cv2\nimport pandas as pd \nimport numpy as np \nimport matplotlib\nfrom IPython.display import clear_output, Image, display\nimport PIL.Image\nimport io\nimport glob\nfrom matplotlib import pyplot as plt\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D, MaxPooling2D, Flatten\nfrom keras.optimizers import Adam\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a11bceedec7ab6c678a7f45865721a643c104f63"},"cell_type":"code","source":"# Read the train data\ndf_train = pd.read_csv(\"../input/train.csv\")\nprint (\"Total Number of Images: \" + str(df_train.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b14f7c8a97809fc45709d92074c990e602f634a3"},"cell_type":"code","source":"# check for null data \ndata_check_images = df_train.loc[df_train['Image'].isnull()]\ndata_check_id = df_train.loc[df_train['Id'].isnull()]\nprint ('Number of null entry in Images : ' + str(data_check_images.shape[0]))\nprint ('Number of null entry in Id : ' + str (data_check_id.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"158870975d145dec824e96f790d65bb5392409f5"},"cell_type":"code","source":"#Function to display image in jupyter notebook \ndef showarray(a, fmt='jpeg'):\n    a = np.uint8(np.clip(a, 0, 255))\n    f = io.BytesIO()\n    PIL.Image.fromarray(a).save(f, fmt)\n    display(Image(data=f.getvalue()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c26c9d91e368648f2aeeb8c5bc86bf6f882171c8"},"cell_type":"code","source":"# Get all the images into the list\nimages_glob = glob.glob(\"../input/train/*.jpg\")\nprint (\"Number of Train images: \" + str(len(images_glob)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"892e2c7cfc58c675a3ebe314e197f0a04e700ed6"},"cell_type":"code","source":"# Display a random image\nrandom_image = images_glob[0]\nimage_data = cv2.imread(random_image)\nshowarray(image_data)\nprint (\"Id of the image : \")\nprint (df_train['Id'].loc[df_train['Image'].apply(lambda image : image==random_image.split('/')[-1])])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30f74835d1352ac2252586b35c7dcb5b6a79e87c"},"cell_type":"code","source":"#check for sizes of the images\nfor each_image in images_glob[5:10]:\n    data_image = cv2.imread(each_image)\n    print (each_image.split('/')[-1] +\" shape is : \" + str(data_image.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e52906aa3e6020e6769070ad8cf0b5630b31b0f4"},"cell_type":"code","source":"# create a dictionary containing key as image and its value as its shape\nimage_data_id = {}\nfor each_image, each_id in zip(df_train['Image'].tolist()[:100], df_train['Id'].tolist()[:100]):\n    image_data_id[each_image] = cv2.imread(\"../input/train/\"+each_image).shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e10b07d1482f2138652efb0ed2bf68cb20baa27"},"cell_type":"code","source":"# print out the minimum resolution of the image\nprint (\"Minimum Length of an Image in whole subset of data : \" + str(np.array(list(image_data_id.values()))[:,0].min())) \nprint (\"Minimum Width of an Image in Whole Subset of data: \" + str(np.array(list(image_data_id.values()))[:,1].min())) \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"868d38570e140646502c38830d50cbb4aaa06821"},"cell_type":"code","source":"# create a dictionary with image name as key and resized image data as its value\nimage_data = {}\nfor each_image, each_id in zip(df_train['Image'].tolist()[:100], df_train['Id'].tolist()[:100]):\n    data_image = cv2.imread(\"../input/train/\"+each_image)\n    data_image = cv2.resize(data_image, (100,300))\n    image_data[each_image] = data_image\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f34a42c82a96f03bf83b479a1f8f496d24ceaf0c"},"cell_type":"code","source":"# Create a dataframe with resized data of the images\ndf_train_resized = pd.DataFrame()\ndf_train_resized['Image'] = image_data.keys()\ndf_train_resized['resized_data']=list(image_data.values())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42694c8bcf6307bd97db833313c9b4bad4ddc6b2"},"cell_type":"code","source":"print (df_train_resized.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"175d3f31f44032b9ef11da8803cfe67fe5c3744a"},"cell_type":"code","source":"# Display the resized image\nshowarray(df_train_resized.iloc[7,1])\nprint (\"Shape of the resized Image : \" + str(df_train_resized.iloc[7,1].shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa0a13bb1c6b448e948731c65be6fd702bb68862"},"cell_type":"code","source":"# resize the image to the lowest available resolution\ndf_train_resized['Labels'] = df_train[\"Id\"].tolist()[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1d7650f1f45dcb10cfbcd47d3f3b0057d24485a"},"cell_type":"code","source":"# create a target data\ntarget = pd.get_dummies(df_train_resized['Labels'])\nlabelled_whale = df_train_resized.loc[target.iloc[:,1:].any(axis=1)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1813e048de7845d558b239d60e6e03a902dc05ed"},"cell_type":"code","source":"print (labelled_whale.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18ff487a0b089a0cbd67c4ffea92ba750a4d9e8b"},"cell_type":"code","source":"# Dataframe with only images labelled as \"new_whale\"\nnew_whale = df_train_resized.drop(labelled_whale.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f86e27f742940318e5b73043d5006f9688df0d8"},"cell_type":"code","source":"# create a train data \ntrain_data = np.array(df_train_resized['resized_data'][:50].apply(lambda arr: arr.flatten()).tolist()).reshape((-1,300,100,3))\ntarget_labeled_vs_new_whale = target.iloc[:,1:].any(axis=1)[:50]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed1e1add5032e10177580234aa6ce12bd61ee611"},"cell_type":"code","source":"# create a dataframe for the distribution plot\ndata_frame_train = pd.DataFrame()\ndata_frame_train['labels'] = target_labeled_vs_new_whale.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a088fa06342fa65689eeb4543580de515120781"},"cell_type":"code","source":"# Percentage Distribution of whales with labelling and unknown whales (new_whale)\nprint (\"percentage of Lablled whales in train data : \" + str((target_labeled_vs_new_whale.mean(axis=0)) * 100))\nprint (\"Percentage of whales lablled as \\'new_whale in train data' : \" +  str(100 - ((target_labeled_vs_new_whale.mean(axis=0)) * 100)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a444c7244fe40e662fabdbcaf2198c34615f602a"},"cell_type":"code","source":"# Distribution under sample of training data \nplt.bar([0,1],data_frame_train['labels'].value_counts().tolist(), color=['r','b'], width=0.3)\nplt.xticks([0,1], ['Labelled_with_Id', 'New_whale'])\nplt.ylabel(\"Number of Images\")\nplt.title(\"Small Subset of data : 50 examples\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"285c5f3a354ed6b096bfb98f0bf43205c592e936"},"cell_type":"code","source":"# creating the validation data\nval_data = np.array(df_train_resized['resized_data'][50:].apply(lambda arr: arr.flatten()).tolist()).reshape((-1,300,100,3))\nval_target = target.iloc[:,1:].any(axis=1)[50:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a84ee75c4a0fa7e2c98178aa70cd15288ccf5440"},"cell_type":"code","source":"print (\"percentage of Lablled whales in test data : \" + str ((val_target.mean(axis=0)) * 100))\nprint (\"percentage of Lablled as \\'new_whale' : \" + str(100 - ((val_target.mean(axis=0)) * 100)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d76c35e18d9206b4c7b19ce24e3ddc3e96ccbf15"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}