{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Happy Whales and Dolphins 🐬","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:08.959778Z","iopub.execute_input":"2022-02-05T17:36:08.960026Z","iopub.status.idle":"2022-02-05T17:36:08.964547Z","shell.execute_reply.started":"2022-02-05T17:36:08.959997Z","shell.execute_reply":"2022-02-05T17:36:08.963754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Competition Files and Folders: {os.listdir('/kaggle/input/happy-whale-and-dolphin')}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:08.971768Z","iopub.execute_input":"2022-02-05T17:36:08.972282Z","iopub.status.idle":"2022-02-05T17:36:08.979324Z","shell.execute_reply.started":"2022-02-05T17:36:08.972253Z","shell.execute_reply":"2022-02-05T17:36:08.978653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading `train.csv` and `sample_submission.csv`.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nsamp_submission_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:08.982233Z","iopub.execute_input":"2022-02-05T17:36:08.982679Z","iopub.status.idle":"2022-02-05T17:36:09.142073Z","shell.execute_reply.started":"2022-02-05T17:36:08.982650Z","shell.execute_reply":"2022-02-05T17:36:09.141334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at `train_df` contents.","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:09.143329Z","iopub.execute_input":"2022-02-05T17:36:09.143560Z","iopub.status.idle":"2022-02-05T17:36:09.161331Z","shell.execute_reply.started":"2022-02-05T17:36:09.143528Z","shell.execute_reply":"2022-02-05T17:36:09.160401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Correcting incorrect spellings for sanity check purposes.\n`kiler_whale` → `killer_whale`\n`bottlenose_dolpin` → `bottlenose_dolphin`","metadata":{}},{"cell_type":"code","source":"train_df.loc[train_df.species == 'kiler_whale', 'species'] = 'killer_whale'\ntrain_df.loc[train_df.species == 'bottlenose_dolpin', 'species'] = 'bottlenose_dolphin'","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:09.162929Z","iopub.execute_input":"2022-02-05T17:36:09.163440Z","iopub.status.idle":"2022-02-05T17:36:09.185149Z","shell.execute_reply.started":"2022-02-05T17:36:09.163402Z","shell.execute_reply":"2022-02-05T17:36:09.184537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at `sample_submission` contents.","metadata":{}},{"cell_type":"code","source":"samp_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:09.188329Z","iopub.execute_input":"2022-02-05T17:36:09.188750Z","iopub.status.idle":"2022-02-05T17:36:09.197530Z","shell.execute_reply.started":"2022-02-05T17:36:09.188714Z","shell.execute_reply":"2022-02-05T17:36:09.196756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How many images, species, and individual IDs are present?","metadata":{}},{"cell_type":"code","source":"print(f\"Images in train index file: {train_df.image.nunique()}\")\nprint(f\"Species in train index file: {train_df.species.nunique()}\")\nprint(f\"Individual IDs in train index file: {train_df.individual_id.nunique()}\")\n\nprint(f\"Images in train images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))}\")\nprint(f\"Images in test images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/test_images'))}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:09.199062Z","iopub.execute_input":"2022-02-05T17:36:09.199794Z","iopub.status.idle":"2022-02-05T17:36:10.280591Z","shell.execute_reply.started":"2022-02-05T17:36:09.199738Z","shell.execute_reply":"2022-02-05T17:36:10.279842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Species frequency within the train dataset.","metadata":{}},{"cell_type":"code","source":"spec_freq = train_df[\"species\"].value_counts()\ndf = pd.DataFrame({'Species': spec_freq.index,\n                   'Images': spec_freq.values})\nplt.figure(figsize = (12, 6))\nplt.title('Distribution of Species Images - Train Dataset')\nsns.set_color_codes(\"deep\")\ns = sns.barplot(x = \"Species\", y=\"Images\", data=df)\ns.set_xticklabels(s.get_xticklabels(), rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:10.281866Z","iopub.execute_input":"2022-02-05T17:36:10.282121Z","iopub.status.idle":"2022-02-05T17:36:10.685665Z","shell.execute_reply.started":"2022-02-05T17:36:10.282084Z","shell.execute_reply":"2022-02-05T17:36:10.685001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualizing the Individual IDs found associated with each species.","metadata":{}},{"cell_type":"code","source":"id_freq = train_df.groupby([\"species\"])[\"individual_id\"].nunique()\ndf = pd.DataFrame({'Species': id_freq.index,\n                   'Unique ID Count': id_freq.values\n                  })\ndf = df.sort_values(['Unique ID Count'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title('Distribution of Species Individual IDs - train dataset')\nsns.set_color_codes(\"deep\")\ns = sns.barplot(x = 'Species', y=\"Unique ID Count\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:10.686767Z","iopub.execute_input":"2022-02-05T17:36:10.687238Z","iopub.status.idle":"2022-02-05T17:36:11.085822Z","shell.execute_reply.started":"2022-02-05T17:36:10.687196Z","shell.execute_reply":"2022-02-05T17:36:11.085185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking if the images listed in `train_df` are identical with those found within the list of images in `train_images`.","metadata":{}},{"cell_type":"code","source":"train_df_list = list(train_df.image.unique())\ntrain_images_list = list(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))\ndelta = set(train_df_list) & set(train_images_list) # iterable conversion\nminus = set(train_df_list) - set(train_images_list) # difference between sets\nprint(f\"Images in train dataset: {len(train_df_list)}\\nImages in train folder: {len(train_images_list)}\\nIntersection: {len(delta)}\\nDifference: {len(minus)}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:11.086872Z","iopub.execute_input":"2022-02-05T17:36:11.088315Z","iopub.status.idle":"2022-02-05T17:36:11.142676Z","shell.execute_reply.started":"2022-02-05T17:36:11.088274Z","shell.execute_reply":"2022-02-05T17:36:11.141858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All the images present in `train_df` are also present in the `train_images` folder.","metadata":{}},{"cell_type":"markdown","source":"Creating a helper function which returns the shape of an image within `train_images`","metadata":{}},{"cell_type":"code","source":"def show_image_size(file_name):\n    image = cv2.imread('/kaggle/input/happy-whale-and-dolphin/train_images/' + file_name)\n    return list(image.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:11.144110Z","iopub.execute_input":"2022-02-05T17:36:11.144377Z","iopub.status.idle":"2022-02-05T17:36:11.148369Z","shell.execute_reply.started":"2022-02-05T17:36:11.144341Z","shell.execute_reply":"2022-02-05T17:36:11.147553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using a sample size of 2500 images, let's determine the image dimensions","metadata":{}},{"cell_type":"markdown","source":"As per competition format consideration import `time`","metadata":{}},{"cell_type":"code","source":"import time\nsample_size = 2500\ntime_alpha = time.time() # start time\ntrain_sample_df = train_df.sample(sample_size)\nsample_img_func = np.stack(train_sample_df['image'].apply(show_image_size))\ndimensions_df = pd.DataFrame(sample_img_func, columns=['width', 'height', 'c_channels'])\nprint(f\"Total run time for {sample_size} images: {round(time.time()-time_alpha, 2)} sec.\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:36:11.149907Z","iopub.execute_input":"2022-02-05T17:36:11.150161Z","iopub.status.idle":"2022-02-05T17:38:49.085469Z","shell.execute_reply.started":"2022-02-05T17:36:11.150127Z","shell.execute_reply":"2022-02-05T17:38:49.084251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's see how many different image dimensions are present in just 2500 image samples.","metadata":{}},{"cell_type":"code","source":"train_img_df = pd.concat([train_sample_df, dimensions_df], axis=1, sort=False)\nprint(f\"Number of different image sizes in {2500} samples: {train_img_df.groupby(['width', 'height','c_channels']).count().shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:49.087771Z","iopub.execute_input":"2022-02-05T17:38:49.088618Z","iopub.status.idle":"2022-02-05T17:38:49.117359Z","shell.execute_reply.started":"2022-02-05T17:38:49.088543Z","shell.execute_reply":"2022-02-05T17:38:49.115963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preperation","metadata":{}},{"cell_type":"code","source":"import PIL\nimport PIL.Image\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:49.119132Z","iopub.execute_input":"2022-02-05T17:38:49.119934Z","iopub.status.idle":"2022-02-05T17:38:53.289124Z","shell.execute_reply.started":"2022-02-05T17:38:49.119859Z","shell.execute_reply":"2022-02-05T17:38:53.288411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_unique = train_df['individual_id'].unique()\nid_unique","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.293178Z","iopub.execute_input":"2022-02-05T17:38:53.293372Z","iopub.status.idle":"2022-02-05T17:38:53.305124Z","shell.execute_reply.started":"2022-02-05T17:38:53.293349Z","shell.execute_reply":"2022-02-05T17:38:53.304179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_to_index = dict((name, index) for index, name in enumerate(id_unique))","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.306749Z","iopub.execute_input":"2022-02-05T17:38:53.307089Z","iopub.status.idle":"2022-02-05T17:38:53.315875Z","shell.execute_reply.started":"2022-02-05T17:38:53.307050Z","shell.execute_reply":"2022-02-05T17:38:53.315149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id_index = [id_to_index[i] for i in train_df['individual_id']]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.317499Z","iopub.execute_input":"2022-02-05T17:38:53.318049Z","iopub.status.idle":"2022-02-05T17:38:53.332455Z","shell.execute_reply.started":"2022-02-05T17:38:53.318010Z","shell.execute_reply":"2022-02-05T17:38:53.331737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id_index[:10]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.335114Z","iopub.execute_input":"2022-02-05T17:38:53.335348Z","iopub.status.idle":"2022-02-05T17:38:53.344482Z","shell.execute_reply.started":"2022-02-05T17:38:53.335319Z","shell.execute_reply":"2022-02-05T17:38:53.343859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'] = image_id_index\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.345943Z","iopub.execute_input":"2022-02-05T17:38:53.346752Z","iopub.status.idle":"2022-02-05T17:38:53.376174Z","shell.execute_reply.started":"2022-02-05T17:38:53.346714Z","shell.execute_reply":"2022-02-05T17:38:53.375404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Collect training image paths:","metadata":{}},{"cell_type":"code","source":"train_image_paths = ['/kaggle/input/happy-whale-and-dolphin/train_images/' + img for img in train_df['image']]\ntrain_image_paths[:10]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.377388Z","iopub.execute_input":"2022-02-05T17:38:53.377689Z","iopub.status.idle":"2022-02-05T17:38:53.399906Z","shell.execute_reply.started":"2022-02-05T17:38:53.377654Z","shell.execute_reply":"2022-02-05T17:38:53.399201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Build helper function to resize images.","metadata":{}},{"cell_type":"code","source":"def image_preprocess(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [224, 224])\n    image = image / 255.0\n    return image","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.401154Z","iopub.execute_input":"2022-02-05T17:38:53.401575Z","iopub.status.idle":"2022-02-05T17:38:53.408888Z","shell.execute_reply.started":"2022-02-05T17:38:53.401528Z","shell.execute_reply":"2022-02-05T17:38:53.408119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_and_process(path):\n    image = tf.io.read_file(path)\n    return image_preprocess(image)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.410115Z","iopub.execute_input":"2022-02-05T17:38:53.410435Z","iopub.status.idle":"2022-02-05T17:38:53.417412Z","shell.execute_reply.started":"2022-02-05T17:38:53.410401Z","shell.execute_reply":"2022-02-05T17:38:53.416599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Function test:","metadata":{}},{"cell_type":"code","source":"for i in range(22):\n    temp_img_path = train_image_paths[i]\n    temp_label = image_id_index[i]\n    plt.imshow(load_and_process(temp_img_path))\n    plt.grid(False)\n    plt.title(id_unique[i] + \" (\" + train_df['species'][i] + \")\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:53.419357Z","iopub.execute_input":"2022-02-05T17:38:53.419894Z","iopub.status.idle":"2022-02-05T17:38:57.957934Z","shell.execute_reply.started":"2022-02-05T17:38:53.419857Z","shell.execute_reply":"2022-02-05T17:38:57.957090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tensorflow Dataset","metadata":{}},{"cell_type":"code","source":"paths_ds = tf.data.Dataset.from_tensor_slices(train_image_paths)\nimages_ds = paths_ds.map(load_and_process, num_parallel_calls=tf.data.experimental.AUTOTUNE)\nlabels_ds = tf.data.Dataset.from_tensor_slices(tf.cast(image_id_index, tf.int64))\nimage_labels_ds = tf.data.Dataset.zip((images_ds, labels_ds)) # Bringing together images and their id labels (image, label)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:57.958997Z","iopub.execute_input":"2022-02-05T17:38:57.959260Z","iopub.status.idle":"2022-02-05T17:38:58.241529Z","shell.execute_reply.started":"2022-02-05T17:38:57.959226Z","shell.execute_reply":"2022-02-05T17:38:58.240844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\nds = image_labels_ds.shuffle(buffer_size=1024)\nds = ds.batch(batch_size)\nds = ds.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\nds","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:58.242592Z","iopub.execute_input":"2022-02-05T17:38:58.244421Z","iopub.status.idle":"2022-02-05T17:38:58.259587Z","shell.execute_reply.started":"2022-02-05T17:38:58.244389Z","shell.execute_reply":"2022-02-05T17:38:58.258969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB0","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:58.260634Z","iopub.execute_input":"2022-02-05T17:38:58.260853Z","iopub.status.idle":"2022-02-05T17:38:59.111117Z","shell.execute_reply.started":"2022-02-05T17:38:58.260822Z","shell.execute_reply":"2022-02-05T17:38:59.110364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess_input = tf.keras.applications.efficientnet.preprocess_input\n\nbase_model = EfficientNetB0(input_shape=(224, 224, 3), include_top=False, weights='imagenet', classifier_activation='softmax')\nbase_model.trainable=True\nprediction_layer = tf.keras.layers.Dense(len(id_unique))","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:38:59.112279Z","iopub.execute_input":"2022-02-05T17:38:59.112536Z","iopub.status.idle":"2022-02-05T17:39:01.124898Z","shell.execute_reply.started":"2022-02-05T17:38:59.112501Z","shell.execute_reply":"2022-02-05T17:39:01.124210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Previous version neglected using a classifier_activation parameter. Accuracy has risen since its addition.","metadata":{}},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=(224, 224, 3))\nx = preprocess_input(inputs)\nx = base_model(x, training=False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dense(1024, activation='relu')(x)\nx = tf.keras.layers.Dense(1024, activation='relu')(x)\nx = tf.keras.layers.Dense(1024, activation='relu')(x)\nx = tf.keras.layers.Dense(1024, activation='relu')(x)\noutputs = prediction_layer(x)\n\nmodel = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:39:01.126339Z","iopub.execute_input":"2022-02-05T17:39:01.126588Z","iopub.status.idle":"2022-02-05T17:39:01.821244Z","shell.execute_reply.started":"2022-02-05T17:39:01.126545Z","shell.execute_reply":"2022-02-05T17:39:01.820491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:39:01.822522Z","iopub.execute_input":"2022-02-05T17:39:01.822987Z","iopub.status.idle":"2022-02-05T17:39:01.845063Z","shell.execute_reply.started":"2022-02-05T17:39:01.822949Z","shell.execute_reply":"2022-02-05T17:39:01.844316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:39:01.848691Z","iopub.execute_input":"2022-02-05T17:39:01.849013Z","iopub.status.idle":"2022-02-05T17:39:01.884526Z","shell.execute_reply.started":"2022-02-05T17:39:01.848946Z","shell.execute_reply":"2022-02-05T17:39:01.883902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(ds, epochs=5)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:39:01.888244Z","iopub.execute_input":"2022-02-05T17:39:01.890110Z","iopub.status.idle":"2022-02-05T19:25:31.344149Z","shell.execute_reply.started":"2022-02-05T17:39:01.890071Z","shell.execute_reply":"2022-02-05T19:25:31.342500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Taking a look at the sample_submission.csv:","metadata":{}},{"cell_type":"code","source":"samp_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:33:27.339882Z","iopub.execute_input":"2022-02-05T19:33:27.340146Z","iopub.status.idle":"2022-02-05T19:33:27.351307Z","shell.execute_reply.started":"2022-02-05T19:33:27.340116Z","shell.execute_reply":"2022-02-05T19:33:27.350606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_paths = ['/kaggle/input/happy-whale-and-dolphin/test_images/' + img for img in samp_submission_df['image']]\ntest_path_ds = tf.data.Dataset.from_tensor_slices(test_image_paths)\ntest_image_ds = test_path_ds.map(load_and_process, num_parallel_calls=tf.data.experimental.AUTOTUNE)\ntest_ds = test_image_ds.batch(32).prefetch(buffer_size=tf.data.experimental.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:33:34.163084Z","iopub.execute_input":"2022-02-05T19:33:34.163480Z","iopub.status.idle":"2022-02-05T19:33:34.273789Z","shell.execute_reply.started":"2022-02-05T19:33:34.163446Z","shell.execute_reply":"2022-02-05T19:33:34.273067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npred = model.predict(test_ds)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:33:39.702847Z","iopub.execute_input":"2022-02-05T19:33:39.703098Z","iopub.status.idle":"2022-02-05T19:44:01.713996Z","shell.execute_reply.started":"2022-02-05T19:33:39.703068Z","shell.execute_reply":"2022-02-05T19:44:01.713239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pred.argsort(axis=1)[:,::-1]\npred = pred[:,0:5]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:44:10.049365Z","iopub.execute_input":"2022-02-05T19:44:10.050033Z","iopub.status.idle":"2022-02-05T19:44:47.122637Z","shell.execute_reply.started":"2022-02-05T19:44:10.049996Z","shell.execute_reply":"2022-02-05T19:44:47.121837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_to_id = {v: k for k, v in id_to_index.items()}\npredictions = [None] * len(pred)\n\nfor i in range(len(pred)):\n    row = [None] * 5\n    \n    for j in range(5):\n        row[j] = index_to_id[pred[i][j]]\n        \n    predictions[i] = \" \".join(row)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:44:50.162780Z","iopub.execute_input":"2022-02-05T19:44:50.163033Z","iopub.status.idle":"2022-02-05T19:44:50.331027Z","shell.execute_reply.started":"2022-02-05T19:44:50.163003Z","shell.execute_reply":"2022-02-05T19:44:50.330396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_submission_df['predictions'] = predictions\nsamp_submission_df['predictions'].head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:44:58.734417Z","iopub.execute_input":"2022-02-05T19:44:58.734736Z","iopub.status.idle":"2022-02-05T19:44:58.754856Z","shell.execute_reply.started":"2022-02-05T19:44:58.734697Z","shell.execute_reply":"2022-02-05T19:44:58.754219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:45:02.913666Z","iopub.execute_input":"2022-02-05T19:45:02.913913Z","iopub.status.idle":"2022-02-05T19:45:03.035379Z","shell.execute_reply.started":"2022-02-05T19:45:02.913885Z","shell.execute_reply":"2022-02-05T19:45:03.034677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:45:06.300063Z","iopub.execute_input":"2022-02-05T19:45:06.300765Z","iopub.status.idle":"2022-02-05T19:45:06.333101Z","shell.execute_reply.started":"2022-02-05T19:45:06.300726Z","shell.execute_reply":"2022-02-05T19:45:06.332342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:45:12.327249Z","iopub.execute_input":"2022-02-05T19:45:12.327954Z","iopub.status.idle":"2022-02-05T19:45:12.338271Z","shell.execute_reply.started":"2022-02-05T19:45:12.327918Z","shell.execute_reply":"2022-02-05T19:45:12.337587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}