{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n        break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-14T16:31:14.042774Z","iopub.execute_input":"2022-03-14T16:31:14.043372Z","iopub.status.idle":"2022-03-14T16:31:41.423983Z","shell.execute_reply.started":"2022-03-14T16:31:14.043274Z","shell.execute_reply":"2022-03-14T16:31:41.421989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Starting slowly looking into data given  ***Note* I am a noobie**","metadata":{}},{"cell_type":"code","source":"#improting essentials\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport time\nfrom tensorflow import keras \nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom IPython.display import Image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom sklearn.model_selection import train_test_split\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:41.427878Z","iopub.execute_input":"2022-03-14T16:31:41.430929Z","iopub.status.idle":"2022-03-14T16:31:43.559671Z","shell.execute_reply.started":"2022-03-14T16:31:41.430887Z","shell.execute_reply":"2022-03-14T16:31:43.558981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nlabels.head()\nimport os\nif len(os.listdir(\"/kaggle/input/happy-whale-and-dolphin/train_images/\"))==len(labels):\n  print(\"proceed Further you are good to go\")\nelse:\n  print(\"some file may be missing check again!\")","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:43.560894Z","iopub.execute_input":"2022-03-14T16:31:43.561152Z","iopub.status.idle":"2022-03-14T16:31:43.643092Z","shell.execute_reply.started":"2022-03-14T16:31:43.561117Z","shell.execute_reply":"2022-03-14T16:31:43.642309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking unique no of species in dataset\nspecies=np.array(labels['species'])\nprint(np.unique(species),len(np.unique(species)))","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:43.645095Z","iopub.execute_input":"2022-03-14T16:31:43.645344Z","iopub.status.idle":"2022-03-14T16:31:43.737067Z","shell.execute_reply.started":"2022-03-14T16:31:43.645308Z","shell.execute_reply":"2022-03-14T16:31:43.736361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cleaning some stuff in labels that we noticed in previous cells\nlabels['species']=labels['species'].apply(lambda x: 'bottlenose_dolphin' if x=='bottlenose_dolpin' else x)\nlabels['species']=labels['species'].apply(lambda x: 'killer_whale' if x=='kiler_whale' else x)\nlabels['species']=labels['species'].apply(lambda x: 'long_finned_pilot_whale' if x=='pilot_whale' else x)\nspecies=np.array(labels['species'])\nprint(np.unique(species),len(np.unique(species)))","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:43.738245Z","iopub.execute_input":"2022-03-14T16:31:43.738480Z","iopub.status.idle":"2022-03-14T16:31:43.866613Z","shell.execute_reply.started":"2022-03-14T16:31:43.738445Z","shell.execute_reply":"2022-03-14T16:31:43.865821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making a boolean matrix for a species\nunique_species=np.unique(species)\nspecies[0]==unique_species","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:43.867951Z","iopub.execute_input":"2022-03-14T16:31:43.868209Z","iopub.status.idle":"2022-03-14T16:31:43.919040Z","shell.execute_reply.started":"2022-03-14T16:31:43.868176Z","shell.execute_reply":"2022-03-14T16:31:43.918157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#like previous making boolean labels for all species  values\nboolean_labels=[temp==unique_species for temp in species]\nlen(boolean_labels)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:43.920496Z","iopub.execute_input":"2022-03-14T16:31:43.920800Z","iopub.status.idle":"2022-03-14T16:31:44.129551Z","shell.execute_reply.started":"2022-03-14T16:31:43.920764Z","shell.execute_reply":"2022-03-14T16:31:44.128832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we can easily switch from boolean values to numbers\nint_label=boolean_labels[0].astype(int)\nint_label","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:44.130802Z","iopub.execute_input":"2022-03-14T16:31:44.131065Z","iopub.status.idle":"2022-03-14T16:31:44.136443Z","shell.execute_reply.started":"2022-03-14T16:31:44.131031Z","shell.execute_reply":"2022-03-14T16:31:44.135815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Starting looking into training images**","metadata":{}},{"cell_type":"code","source":"#trying to get all images path in a variable\ndef give_image_paths(repo):\n    image_paths=labels['image']\n    image_paths=image_paths.apply(lambda x: '/kaggle/input/happy-whale-and-dolphin/'+str(repo)+'/'+str(x))\n    return np.array(image_paths)\ntrain_images=give_image_paths('train_images')\ntrain_images[0],len(train_images)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:44.137549Z","iopub.execute_input":"2022-03-14T16:31:44.138261Z","iopub.status.idle":"2022-03-14T16:31:44.184556Z","shell.execute_reply.started":"2022-03-14T16:31:44.138224Z","shell.execute_reply":"2022-03-14T16:31:44.183879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Sneak peek at the training images\ndef read_one_image(path):\n    testimg=tf.io.read_file(path)\n    testimg=tf.image.decode_jpeg(testimg,channels=3)\n    testimg=tf.image.convert_image_dtype(testimg,tf.float32)\n    testimg=tf.image.resize(testimg,size=[256,256])\n    return testimg\ni=0\nplt.figure(figsize=(15,15))\nfor i in range(25):\n    testimg=read_one_image(train_images[i])\n    ax=plt.subplot(5,5,i+1)\n    i+=1\n    plt.imshow(testimg)\n    plt.title(species[i])\n    plt.axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:33:00.972098Z","iopub.execute_input":"2022-03-14T16:33:00.972746Z","iopub.status.idle":"2022-03-14T16:33:11.194886Z","shell.execute_reply.started":"2022-03-14T16:33:00.972710Z","shell.execute_reply":"2022-03-14T16:33:11.193874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we will start experimenting on first few images\nNUM_IMAGES = 5000\nX=train_images[:NUM_IMAGES]\ny=boolean_labels[:NUM_IMAGES]\nX_train,X_val,y_train,y_val=train_test_split(X[:NUM_IMAGES],y[:NUM_IMAGES] , test_size=0.2,random_state=42)\nlen(X_train),len(y_train),len(X_val),len(y_val)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:55.393513Z","iopub.status.idle":"2022-03-14T16:31:55.394047Z","shell.execute_reply.started":"2022-03-14T16:31:55.393741Z","shell.execute_reply":"2022-03-14T16:31:55.393766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shape is as expected\nread_one_image(X[0]).shape","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:55.395680Z","iopub.status.idle":"2022-03-14T16:31:55.396632Z","shell.execute_reply.started":"2022-03-14T16:31:55.396345Z","shell.execute_reply":"2022-03-14T16:31:55.396375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets turn our data into batches**","metadata":{}},{"cell_type":"code","source":"# create a tuple of tensors\ndef get_image_label(image_path,label):\n  \"\"\"\n  takes image path anmd label and give tensor image and label tuple\n  \"\"\"\n  image=read_one_image(image_path)\n  return image,label\nbatch_size=32\n#create a function to turn data into batches\n#we are gonna turn training , validation and test batch separately\ndef create_data_batches(X,y=None,batch_size=batch_size,valid_data=False,test_data=False):\n  \"\"\"\n  create batches of data out of image (X) and label(y) pairs.\n  Shuffles the data if its training but doesnt shuffle if its validation data\n  \"\"\"\n  # if data is test data setwe know we dont have labels\n  if test_data:\n    print(\"Creating test data batches......\")\n    data=tf.data.Dataset.from_tensor_slices((tf.constant(X)))\n    data_batch=data.map(read_one_image).batch(batch_size)\n    return data_batch\n  # if the data is validatuion dataset,we dont need to shuffle it\n  elif valid_data:\n    print(\"Creating validation data batches.....\")\n    data=tf.data.Dataset.from_tensor_slices((tf.constant(X),tf.constant(y)))\n    data_batch=data.map(get_image_label).batch(batch_size)\n    return data_batch\n  # if its taring data\n  else:\n    print(\"Creating training data batches......\")\n    #turn file path and labels into tensors\n    data=tf.data.Dataset.from_tensor_slices((tf.constant(X),tf.constant(y)))\n    #shuffling data\n    data=data.shuffle(buffer_size=len(X))\n    data=data.map(get_image_label)\n    #turn trainig data into batches\n    data_batch=data.batch(batch_size)\n    return data_batch","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:55.397848Z","iopub.status.idle":"2022-03-14T16:31:55.398658Z","shell.execute_reply.started":"2022-03-14T16:31:55.398413Z","shell.execute_reply":"2022-03-14T16:31:55.398440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create training and validataion data batches\ntrain_data=create_data_batches(X_train,y_train)\nval_data=create_data_batches(X_val,y_val,valid_data=True)\ntrain_data.element_spec,val_data.element_spec","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:55.400070Z","iopub.status.idle":"2022-03-14T16:31:55.401216Z","shell.execute_reply.started":"2022-03-14T16:31:55.400973Z","shell.execute_reply":"2022-03-14T16:31:55.400998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:55.402754Z","iopub.status.idle":"2022-03-14T16:31:55.404224Z","shell.execute_reply.started":"2022-03-14T16:31:55.403982Z","shell.execute_reply":"2022-03-14T16:31:55.404021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building starts","metadata":{}},{"cell_type":"code","source":"# lets get our input shape and output shape cleared to build a model\nNUM_SIZE=256\ninput_shape=[None,NUM_SIZE,NUM_SIZE,3]\noutput_shape=len(unique_species)\ninput_shape,output_shape","metadata":{"execution":{"iopub.status.busy":"2022-03-14T16:31:55.405757Z","iopub.status.idle":"2022-03-14T16:31:55.407273Z","shell.execute_reply.started":"2022-03-14T16:31:55.407021Z","shell.execute_reply":"2022-03-14T16:31:55.407049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}