{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom matplotlib.image import imread\nfrom sklearn.model_selection import train_test_split\nimport os\nfrom matplotlib import pyplot as plt\n\nfrom PIL import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-15T14:58:29.555542Z","iopub.execute_input":"2022-04-15T14:58:29.556033Z","iopub.status.idle":"2022-04-15T14:58:34.942907Z","shell.execute_reply.started":"2022-04-15T14:58:29.555937Z","shell.execute_reply":"2022-04-15T14:58:34.942185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imags = '../input/happy-whale-and-dolphin/train_images'\ntrain_images = []\ny_train = []\ntrain_imgs = os.walk(train_imags, topdown=True)\nfor (root, dirs, files) in train_imgs:\n    for element in files:\n        train_images.append('{}{}'.format(\"../input/happy-whale-and-dolphin/train_images/\", element))\n        y_train.append(element.replace('\\'',''))","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:00:27.132332Z","iopub.execute_input":"2022-04-15T15:00:27.132952Z","iopub.status.idle":"2022-04-15T15:00:52.723272Z","shell.execute_reply.started":"2022-04-15T15:00:27.132912Z","shell.execute_reply":"2022-04-15T15:00:52.722484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ntrain_image_pixels = []\nit=1999\nfor image in (train_images):\n    if image.endswith('.jpg'):\n        train_image_pixels.append(np.asarray(keras.utils.load_img(image, grayscale=False, color_mode='rgb', target_size=(300,300,3), interpolation='nearest')))\n        #plt.imshow(imread(image), interpolation='nearest')\n        #plt.show()\n        #print(imread(image).shape)\n        gc.collect()\n        it-=1\n        if it<0:\n            break\n    else:\n        continue\n    print(it)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-04-15T15:01:00.431260Z","iopub.execute_input":"2022-04-15T15:01:00.431649Z","iopub.status.idle":"2022-04-15T15:08:20.656690Z","shell.execute_reply.started":"2022-04-15T15:01:00.431614Z","shell.execute_reply":"2022-04-15T15:08:20.655981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-04-03T20:10:08.781049Z","iopub.execute_input":"2022-04-03T20:10:08.781431Z","iopub.status.idle":"2022-04-03T20:10:08.801131Z","shell.execute_reply.started":"2022-04-03T20:10:08.781402Z","shell.execute_reply":"2022-04-03T20:10:08.800215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train_image_pixels = []\ni = 499\n\nfor image in (train_images):\n    if image.endswith('.jpg'):\n        train_image_pixels.append(np.asarray(imread(image)))\n        #plt.imshow(imread(image), interpolation='nearest')\n        #plt.show()\n        #print(imread(image).shape)\n        i-=1\n        if i<0:\n            break\n    else:\n        continue","metadata":{}},{"cell_type":"markdown","source":"train_image_pixels = []\ni = 28000\n\nfor image in (train_images):\n    if image.endswith('.jpg'):\n        imread(image)\n        #plt.imshow(imread(image), interpolation='nearest')\n        #plt.show()\n        #print(imread(image).shape)\n        i-=1\n        print(i)\n        if i<0:\n            break\n    else:\n        continue","metadata":{"execution":{"iopub.status.busy":"2022-04-03T17:55:47.107941Z","iopub.execute_input":"2022-04-03T17:55:47.108217Z","iopub.status.idle":"2022-04-03T18:05:52.506948Z","shell.execute_reply.started":"2022-04-03T17:55:47.108187Z","shell.execute_reply":"2022-04-03T18:05:52.505762Z"}}},{"cell_type":"markdown","source":"import gc\ntrain_image_pixels = []\nyour_mom=0\n\nfor i in range(28000):\n    for image in (train_images):\n        if image.endswith('.jpg'):\n            train_image_pixels.append(np.asarray(imread(image)))\n            #plt.imshow(imread(image), interpolation='nearest')\n            #plt.show()\n            #print(imread(image).shape)\n            your_mom-=1\n            print(your_mom)\n            gc.collect()\n        else:\n            continue","metadata":{"execution":{"iopub.status.busy":"2022-04-03T18:15:25.883909Z","iopub.execute_input":"2022-04-03T18:15:25.884357Z"}}},{"cell_type":"markdown","source":"import gc\ntrain_image_pixels = []\n\ndef pixel_generator(train_imagess):\n    for image in (train_imagess):\n        if image.endswith('.jpg'):\n            train_image_pixels.append(np.asarray(imread(image)))\n            #plt.imshow(imread(image), interpolation='nearest')\n            #plt.show()\n            #print(imread(image).shape)\n            gc.collect()\n        else:\n            continue\n    print(\"your_mom\")","metadata":{"execution":{"iopub.status.busy":"2022-04-03T18:47:27.613147Z","iopub.execute_input":"2022-04-03T18:47:27.613446Z","iopub.status.idle":"2022-04-03T18:47:27.619515Z","shell.execute_reply.started":"2022-04-03T18:47:27.613412Z","shell.execute_reply":"2022-04-03T18:47:27.618476Z"}}},{"cell_type":"code","source":"train_image_pixels = np.asarray(train_image_pixels)\ntrain_image_pixels.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:08:27.062379Z","iopub.execute_input":"2022-04-15T15:08:27.062623Z","iopub.status.idle":"2022-04-15T15:08:27.249530Z","shell.execute_reply.started":"2022-04-15T15:08:27.062593Z","shell.execute_reply":"2022-04-15T15:08:27.248869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(train_image_pixels[0], interpolation='nearest')","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:08:44.084610Z","iopub.execute_input":"2022-04-15T15:08:44.085173Z","iopub.status.idle":"2022-04-15T15:08:44.304693Z","shell.execute_reply.started":"2022-04-15T15:08:44.085129Z","shell.execute_reply":"2022-04-15T15:08:44.304059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(train_image_pixels[0], interpolation='nearest')\n#plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T14:24:33.59613Z","iopub.status.idle":"2022-04-03T14:24:33.596746Z","shell.execute_reply.started":"2022-04-03T14:24:33.596438Z","shell.execute_reply":"2022-04-03T14:24:33.596466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_pixels = np.reshape(train_image_pixels, (2000, 270000))\n#Needs to be fixed. I reshaped it in order to be able to turn it into a Pandas DataFrame,\n# and I need a pandas dataframe to index and match the Individual ids to data","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:08:58.506416Z","iopub.execute_input":"2022-04-15T15:08:58.506670Z","iopub.status.idle":"2022-04-15T15:08:58.511128Z","shell.execute_reply.started":"2022-04-15T15:08:58.506641Z","shell.execute_reply":"2022-04-15T15:08:58.509933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_pixels_pd = pd.DataFrame(train_image_pixels)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:09:10.685316Z","iopub.execute_input":"2022-04-15T15:09:10.685576Z","iopub.status.idle":"2022-04-15T15:09:10.690432Z","shell.execute_reply.started":"2022-04-15T15:09:10.685546Z","shell.execute_reply":"2022-04-15T15:09:10.689704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_image_pixels_pd)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:09:12.044610Z","iopub.execute_input":"2022-04-15T15:09:12.045211Z","iopub.status.idle":"2022-04-15T15:09:12.065552Z","shell.execute_reply.started":"2022-04-15T15:09:12.045170Z","shell.execute_reply":"2022-04-15T15:09:12.064869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_pixels_pd['label'] = y_train[:2000]\n#file name is matched with data","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:09:21.334164Z","iopub.execute_input":"2022-04-15T15:09:21.334428Z","iopub.status.idle":"2022-04-15T15:09:21.438662Z","shell.execute_reply.started":"2022-04-15T15:09:21.334398Z","shell.execute_reply":"2022-04-15T15:09:21.437907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:09:22.759282Z","iopub.execute_input":"2022-04-15T15:09:22.759732Z","iopub.status.idle":"2022-04-15T15:09:22.853595Z","shell.execute_reply.started":"2022-04-15T15:09:22.759696Z","shell.execute_reply":"2022-04-15T15:09:22.852720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('../input/happy-whale-and-dolphin/train.csv')\n\ny = []\n#for loop to match file name with individual_id\nfor image in train['image']:\n    for label in train_image_pixels_pd['label']:\n        if image == label:\n            y.append(((train.loc[train['image'] == label])['individual_id'])) #returning individual id\n        else:\n            continue\n\nx = train_image_pixels_pd.drop(columns='label')\n#y = train_image_pixels_pd['label']\n#y = train_image_pixels_pd['label']\n\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:09:24.625108Z","iopub.execute_input":"2022-04-15T15:09:24.625646Z","iopub.status.idle":"2022-04-15T15:10:04.088809Z","shell.execute_reply.started":"2022-04-15T15:09:24.625608Z","shell.execute_reply":"2022-04-15T15:10:04.088078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yy = []\n\nfor target in y:\n    a = str(target)\n    #y == x.replace('\\nName: species, dtype: object','')\n    #y2 == y.replace('1','')\n    #y3 == y2.replace('2','')\n    #y4 == y3.replace('3','')\n    #y5 == y4.replace('4','')\n    #y6 == y5.replace('5','')\n    #y7 == y6.replace('6','')\n    #y8 == y7.replace('7','')\n    #y9 == y8.replace('8','')\n    #y10 == y9.replace('9','')\n    #y11 == y10.replace('0','')\n    yy.append(a)\nyyy = [] \nfor target in yy:\n    yyy.append(target.replace('\\nName: individual_id, dtype: object',''))\n    \ny1 = []\nfor target in yyy:\n    y1.append(target[-12:]) #isolating individual ids\nyyy[0:5]\ny1[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:18.496775Z","iopub.execute_input":"2022-04-15T15:10:18.497496Z","iopub.status.idle":"2022-04-15T15:10:19.152124Z","shell.execute_reply.started":"2022-04-15T15:10:18.497458Z","shell.execute_reply":"2022-04-15T15:10:19.151322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1[0:5]\ny_labels = pd.DataFrame(y1)\ny_labels = np.asarray(y_labels)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:27.620370Z","iopub.execute_input":"2022-04-15T15:10:27.620844Z","iopub.status.idle":"2022-04-15T15:10:27.627681Z","shell.execute_reply.started":"2022-04-15T15:10:27.620800Z","shell.execute_reply":"2022-04-15T15:10:27.626879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_labels.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:31.863684Z","iopub.execute_input":"2022-04-15T15:10:31.864228Z","iopub.status.idle":"2022-04-15T15:10:31.869281Z","shell.execute_reply.started":"2022-04-15T15:10:31.864188Z","shell.execute_reply":"2022-04-15T15:10:31.868615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import LabelEncoder\n\nenc = OneHotEncoder(handle_unknown='ignore', sparse=False)\n\ny_labels_df = pd.DataFrame(enc.fit_transform(y_labels))\n#y_train_df = pd.DataFrame(enc.fit_transform(y_train))\n\ny_labels_df.head(10)\ny_labels = np.asarray(y_labels_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:37.190738Z","iopub.execute_input":"2022-04-15T15:10:37.191206Z","iopub.status.idle":"2022-04-15T15:10:37.214404Z","shell.execute_reply.started":"2022-04-15T15:10:37.191168Z","shell.execute_reply":"2022-04-15T15:10:37.213702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_labels_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:39.152723Z","iopub.execute_input":"2022-04-15T15:10:39.153512Z","iopub.status.idle":"2022-04-15T15:10:39.190555Z","shell.execute_reply.started":"2022-04-15T15:10:39.153465Z","shell.execute_reply":"2022-04-15T15:10:39.189847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(x, y_labels, test_size = 0.3, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:43.933318Z","iopub.execute_input":"2022-04-15T15:10:43.933887Z","iopub.status.idle":"2022-04-15T15:10:44.809940Z","shell.execute_reply.started":"2022-04-15T15:10:43.933829Z","shell.execute_reply":"2022-04-15T15:10:44.809179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape, x_val.shape, y_train.shape, y_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:10:46.510570Z","iopub.execute_input":"2022-04-15T15:10:46.511131Z","iopub.status.idle":"2022-04-15T15:10:46.516880Z","shell.execute_reply.started":"2022-04-15T15:10:46.511090Z","shell.execute_reply":"2022-04-15T15:10:46.516158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import LabelEncoder\n\nenc = OneHotEncoder(handle_unknown='ignore', sparse=False)\n\ny_val_df = pd.DataFrame(enc.fit_transform(y_val))\ny_train_df = pd.DataFrame(enc.fit_transform(y_train))\n\ny_val_df.head(10)\n#y_train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T12:49:21.527516Z","iopub.execute_input":"2022-04-12T12:49:21.528044Z","iopub.status.idle":"2022-04-12T12:49:21.588454Z","shell.execute_reply.started":"2022-04-12T12:49:21.528008Z","shell.execute_reply":"2022-04-12T12:49:21.586639Z"}}},{"cell_type":"markdown","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import LabelEncoder\n\nenc = OneHotEncoder(handle_unknown='ignore', sparse=False)\n\nlabelencoder = LabelEncoder()\ny_val = pd.DataFrame(y_val, columns = ['labels'])\ny_train = pd.DataFrame(y_train, columns = ['labels'])\n\ny_val_df = pd.DataFrame(enc.fit_transform(y_val[['labels']]))\ny_train_df = pd.DataFrame(enc.fit_transform(y_train[['labels']]))\n\ny_val_df.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T20:22:00.876557Z","iopub.execute_input":"2022-03-20T20:22:00.877284Z","iopub.status.idle":"2022-03-20T20:22:00.900935Z","shell.execute_reply.started":"2022-03-20T20:22:00.877245Z","shell.execute_reply":"2022-03-20T20:22:00.900208Z"}}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.Sequential([\n    layers.InputLayer(input_shape=(300, 300, 3)),\n    layers.Conv2D(kernel_size=(3, 3), activation='relu', filters=1),\n    layers.MaxPooling2D(2,2),\n    #layers.Dropout(0.3),\n    #layers.BatchNormalization(),\n    layers.Conv2D(kernel_size=(3, 3), activation='relu', filters=1),\n    layers.MaxPooling2D(2,2),\n    layers.Flatten(),\n    layers.Dropout(0.1),\n    layers.BatchNormalization(),\n    layers.Dense(1486, activation='softmax')\n])","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:42:11.181124Z","iopub.execute_input":"2022-04-15T15:42:11.181384Z","iopub.status.idle":"2022-04-15T15:42:11.235846Z","shell.execute_reply.started":"2022-04-15T15:42:11.181353Z","shell.execute_reply":"2022-04-15T15:42:11.235196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.01),\n    loss='categorical_crossentropy',\n    metrics='accuracy',\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:42:13.994718Z","iopub.execute_input":"2022-04-15T15:42:13.995392Z","iopub.status.idle":"2022-04-15T15:42:14.006381Z","shell.execute_reply.started":"2022-04-15T15:42:13.995352Z","shell.execute_reply":"2022-04-15T15:42:14.005605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = keras.callbacks.EarlyStopping(\n    min_delta=0.001,\n    patience=12,\n    restore_best_weights=True,\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:42:03.219133Z","iopub.execute_input":"2022-04-15T15:42:03.219399Z","iopub.status.idle":"2022-04-15T15:42:03.223951Z","shell.execute_reply.started":"2022-04-15T15:42:03.219368Z","shell.execute_reply":"2022-04-15T15:42:03.223172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, y_train_np, x_val, y_val_np = np.asarray(x_train), np.asarray(y_train), np.asarray(x_val), np.asarray(y_val)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:12:17.576655Z","iopub.execute_input":"2022-04-15T15:12:17.576938Z","iopub.status.idle":"2022-04-15T15:12:17.581084Z","shell.execute_reply.started":"2022-04-15T15:12:17.576899Z","shell.execute_reply":"2022-04-15T15:12:17.580377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape, y_train_np.shape, x_val.shape, y_val_np.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:12:19.712656Z","iopub.execute_input":"2022-04-15T15:12:19.712927Z","iopub.status.idle":"2022-04-15T15:12:19.719204Z","shell.execute_reply.started":"2022-04-15T15:12:19.712895Z","shell.execute_reply":"2022-04-15T15:12:19.718304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val = x_train.reshape(1400,300,300,3), x_val.reshape(600,300,300,3)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:12:37.393004Z","iopub.execute_input":"2022-04-15T15:12:37.393298Z","iopub.status.idle":"2022-04-15T15:12:37.401909Z","shell.execute_reply.started":"2022-04-15T15:12:37.393262Z","shell.execute_reply":"2022-04-15T15:12:37.401105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    x_train, y_train_np,\n    validation_data=(x_val, y_val_np),\n    batch_size=32,\n    epochs=125,\n    #callbacks=[early_stopping],\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:42:16.368643Z","iopub.execute_input":"2022-04-15T15:42:16.369072Z","iopub.status.idle":"2022-04-15T15:43:42.439953Z","shell.execute_reply.started":"2022-04-15T15:42:16.369032Z","shell.execute_reply":"2022-04-15T15:43:42.439252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape, x_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-07T00:33:00.904308Z","iopub.execute_input":"2022-04-07T00:33:00.904558Z","iopub.status.idle":"2022-04-07T00:33:00.911472Z","shell.execute_reply.started":"2022-04-07T00:33:00.90453Z","shell.execute_reply":"2022-04-07T00:33:00.910608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\n\nhistory_df.loc[:, ['loss', 'val_loss']].plot()\nhistory_df.loc[:, ['accuracy', 'val_accuracy']].plot()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T14:24:33.638037Z","iopub.status.idle":"2022-04-03T14:24:33.638768Z","shell.execute_reply.started":"2022-04-03T14:24:33.638435Z","shell.execute_reply":"2022-04-03T14:24:33.638466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T14:24:33.640227Z","iopub.status.idle":"2022-04-03T14:24:33.640954Z","shell.execute_reply.started":"2022-04-03T14:24:33.640638Z","shell.execute_reply":"2022-04-03T14:24:33.640667Z"},"trusted":true},"execution_count":null,"outputs":[]}]}