{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n   \nimport os\nimport tensorflow as tf\nimport sklearn\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import KFold\n\n# You can write up to 20GB to the current directory (/kaggle/wo`rking/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-05T10:02:09.315149Z","iopub.execute_input":"2022-02-05T10:02:09.315669Z","iopub.status.idle":"2022-02-05T10:02:14.158740Z","shell.execute_reply.started":"2022-02-05T10:02:09.315554Z","shell.execute_reply":"2022-02-05T10:02:14.158008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/happy-whale-and-dolphin/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:14.160593Z","iopub.execute_input":"2022-02-05T10:02:14.160805Z","iopub.status.idle":"2022-02-05T10:02:14.170817Z","shell.execute_reply.started":"2022-02-05T10:02:14.160776Z","shell.execute_reply":"2022-02-05T10:02:14.169461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.read_csv(path+'train.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:14.172589Z","iopub.execute_input":"2022-02-05T10:02:14.172831Z","iopub.status.idle":"2022-02-05T10:02:14.300611Z","shell.execute_reply.started":"2022-02-05T10:02:14.172798Z","shell.execute_reply":"2022-02-05T10:02:14.299903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = all_data.copy()\ntrain_data.image = train_data.image.apply(lambda x: '/kaggle/input/happy-whale-and-dolphin/train_images/' + x)\n\ntrain_data.species = train_data.species.apply(lambda x: 'bottlenose_dolphin' if x == 'bottlenose_dolpin' else x)\ntrain_data.species = train_data.species.apply(lambda x: 'killer_whale' if x == 'kiler_whale' else x)\n\nlbl_coder = LabelEncoder()\nlbl_coder.fit(train_data.species)\n\n\ntrain_data['species_id'] = lbl_coder.transform(train_data.species)\ntrain_data.sample(4)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:14.302506Z","iopub.execute_input":"2022-02-05T10:02:14.302699Z","iopub.status.idle":"2022-02-05T10:02:14.376244Z","shell.execute_reply.started":"2022-02-05T10:02:14.302675Z","shell.execute_reply":"2022-02-05T10:02:14.375541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds = list(KFold().split(train_data))\nfold_id = 0\n\ncsv_train = train_data.iloc[folds[fold_id][0]]\ncsv_val = train_data.iloc[folds[fold_id][1]]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:14.378082Z","iopub.execute_input":"2022-02-05T10:02:14.378453Z","iopub.status.idle":"2022-02-05T10:02:14.389351Z","shell.execute_reply.started":"2022-02-05T10:02:14.378416Z","shell.execute_reply":"2022-02-05T10:02:14.388661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(csv_train.image.apply(lambda x: x.split('.')[-1]))","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:14.390799Z","iopub.execute_input":"2022-02-05T10:02:14.392653Z","iopub.status.idle":"2022-02-05T10:02:14.427380Z","shell.execute_reply.started":"2022-02-05T10:02:14.392616Z","shell.execute_reply":"2022-02-05T10:02:14.426727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndef load_img(x):\n    img_data = tf.io.read_file(x)\n    img = tf.image.decode_jpeg(img_data, channels=3)\n    #if img.shape[-1] != 3:\n     #   img = tf.image.grayscale_to_rgb(img)\n    return img\n\ndef normalize(img):\n    img = tf.image.resize(img, (128, 128))\n    \n    img /= 255\n    return img\n\ndef encode_l(x):\n    return tf.one_hot(x, 28)\n\ndef load_base_ds(paths, labels):\n    ds_i = tf.data.Dataset.from_tensor_slices(paths)\n    ds_i = ds_i.map(load_img).map(normalize)\n    ds_l = tf.data.Dataset.from_tensor_slices(labels).map(encode_l)\n    ds = tf.data.Dataset.zip((ds_i, ds_l))\n    return ds\n    \nds_train = load_base_ds(csv_train.image, csv_train.species_id)\nds_val = load_base_ds(csv_val.image, csv_val.species_id)\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:14.428727Z","iopub.execute_input":"2022-02-05T10:02:14.428984Z","iopub.status.idle":"2022-02-05T10:02:17.108332Z","shell.execute_reply.started":"2022-02-05T10:02:14.428952Z","shell.execute_reply":"2022-02-05T10:02:17.107608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\n\nm = tf.keras.Sequential([\n    hub.KerasLayer(\"https://tfhub.dev/google/imagenet/efficientnet_v2_imagenet1k_b0/feature_vector/2\", trainable=False),\n    tf.keras.layers.Dense(28, activation='softmax'),\n])\n\nm.compile(optimizer=tf.keras.optimizers.Adam(), loss=tf.keras.losses.CategoricalCrossentropy(),\n         metrics=[tf.keras.metrics.CategoricalAccuracy()])","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:02:17.109484Z","iopub.execute_input":"2022-02-05T10:02:17.110055Z","iopub.status.idle":"2022-02-05T10:02:27.247740Z","shell.execute_reply.started":"2022-02-05T10:02:17.110009Z","shell.execute_reply":"2022-02-05T10:02:27.247028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_to_train(ds):\n    return ds.shuffle(128).batch(32).prefetch(4)\n\n_ds_train = prepare_to_train(ds_train)\n_ds_val = prepare_to_train(ds_val)\n\nm.fit(_ds_train, validation_data=_ds_val, epochs=3)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T10:20:35.102024Z","iopub.execute_input":"2022-02-05T10:20:35.102489Z","iopub.status.idle":"2022-02-05T11:41:06.947686Z","shell.execute_reply.started":"2022-02-05T10:20:35.102454Z","shell.execute_reply":"2022-02-05T11:41:06.945134Z"},"trusted":true},"execution_count":null,"outputs":[]}]}