{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport tensorflow as tf\nimport os\nimport multiprocessing\nfrom tqdm import tqdm\nfrom PIL import ImageDraw\ntrain_on_gpu = True\n\nPath = '/kaggle/input/histopathologic-cancer-detection/'\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-15T19:51:24.743976Z","iopub.execute_input":"2024-04-15T19:51:24.744389Z","iopub.status.idle":"2024-04-15T19:51:39.838630Z","shell.execute_reply.started":"2024-04-15T19:51:24.744352Z","shell.execute_reply":"2024-04-15T19:51:39.836964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(Path + 'train_labels.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:39.842758Z","iopub.execute_input":"2024-04-15T19:51:39.843457Z","iopub.status.idle":"2024-04-15T19:51:40.196362Z","shell.execute_reply.started":"2024-04-15T19:51:39.843420Z","shell.execute_reply":"2024-04-15T19:51:40.195390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samples = pd.read_csv(Path + 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:40.197429Z","iopub.execute_input":"2024-04-15T19:51:40.197709Z","iopub.status.idle":"2024-04-15T19:51:40.330053Z","shell.execute_reply.started":"2024-04-15T19:51:40.197685Z","shell.execute_reply":"2024-04-15T19:51:40.329111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = \"/kaggle/input/histopathologic-cancer-detection/train/\"\ntest = \"/kaggle/input/histopathologic-cancer-detection/test\"\n\nprint(\"Number of training images: {}\".format(len(os.listdir(train))))\nprint(\"Number of test images: {}\".format(len(os.listdir(test))))","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:40.333409Z","iopub.execute_input":"2024-04-15T19:51:40.334072Z","iopub.status.idle":"2024-04-15T19:51:45.688282Z","shell.execute_reply.started":"2024-04-15T19:51:40.334037Z","shell.execute_reply":"2024-04-15T19:51:45.687604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(30, 6))\ntrain_imgs = os.listdir(Path+\"train\")\nfor idx, img in enumerate(np.random.choice(train_imgs, 20)):\n    ax = fig.add_subplot(2, 20//2, idx+1, xticks=[], yticks=[])\n    im = Image.open(Path+\"train/\" + img)\n    plt.imshow(im)\n    label = train_df.loc[train_df['id'] == img.split('.')[0], 'label'].values[0]\n    ax.set_title(f'Label: {label}')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:45.689122Z","iopub.execute_input":"2024-04-15T19:51:45.690327Z","iopub.status.idle":"2024-04-15T19:51:48.228450Z","shell.execute_reply.started":"2024-04-15T19:51:45.690278Z","shell.execute_reply":"2024-04-15T19:51:48.226624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = train_df.isnull().sum()\nmissing_values","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.230367Z","iopub.execute_input":"2024-04-15T19:51:48.230724Z","iopub.status.idle":"2024-04-15T19:51:48.247590Z","shell.execute_reply.started":"2024-04-15T19:51:48.230693Z","shell.execute_reply":"2024-04-15T19:51:48.246198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.shape)\nprint(train_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.248995Z","iopub.execute_input":"2024-04-15T19:51:48.250030Z","iopub.status.idle":"2024-04-15T19:51:48.258050Z","shell.execute_reply.started":"2024-04-15T19:51:48.249993Z","shell.execute_reply":"2024-04-15T19:51:48.256337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(im).shape","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.259620Z","iopub.execute_input":"2024-04-15T19:51:48.260003Z","iopub.status.idle":"2024-04-15T19:51:48.279057Z","shell.execute_reply.started":"2024-04-15T19:51:48.259967Z","shell.execute_reply":"2024-04-15T19:51:48.277587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.280021Z","iopub.execute_input":"2024-04-15T19:51:48.280355Z","iopub.status.idle":"2024-04-15T19:51:48.297703Z","shell.execute_reply.started":"2024-04-15T19:51:48.280330Z","shell.execute_reply":"2024-04-15T19:51:48.296392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].value_counts().plot(kind='pie', legend=True, autopct='%1.0f%%')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.301238Z","iopub.execute_input":"2024-04-15T19:51:48.301615Z","iopub.status.idle":"2024-04-15T19:51:48.461636Z","shell.execute_reply.started":"2024-04-15T19:51:48.301581Z","shell.execute_reply":"2024-04-15T19:51:48.460836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"malignant = train_df.loc[train_df['label']==1]['id'].values    \nnormal = train_df.loc[train_df['label']==0]['id'].values      \ntrain_df['label'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.464222Z","iopub.execute_input":"2024-04-15T19:51:48.464552Z","iopub.status.idle":"2024-04-15T19:51:48.667013Z","shell.execute_reply.started":"2024-04-15T19:51:48.464527Z","shell.execute_reply":"2024-04-15T19:51:48.666286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"malignant = train_df.loc[train_df['label']==1]['id'].values    \nnormal = train_df.loc[train_df['label']==0]['id'].values      \ntrain_df['label'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.668199Z","iopub.execute_input":"2024-04-15T19:51:48.668898Z","iopub.status.idle":"2024-04-15T19:51:48.874508Z","shell.execute_reply.started":"2024-04-15T19:51:48.668860Z","shell.execute_reply":"2024-04-15T19:51:48.873326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_fig(ids,title,nrows=3,ncols=3):\n\n    fig,ax = plt.subplots(nrows,ncols,figsize=(7,7))\n    plt.subplots_adjust(wspace=0, hspace=0) \n    for i,j in enumerate(ids[:nrows*ncols]):\n        fname = os.path.join(train ,j +'.tif')\n        img = Image.open(fname)\n        idcol = ImageDraw.Draw(img)\n        idcol.rectangle(((0,0),(95,95)),outline='white')\n        plt.subplot(nrows, ncols, i+1) \n        plt.imshow(np.array(img))\n        plt.axis('off')\n\n    plt.suptitle(title, y=0.94)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.875711Z","iopub.execute_input":"2024-04-15T19:51:48.876277Z","iopub.status.idle":"2024-04-15T19:51:48.883481Z","shell.execute_reply.started":"2024-04-15T19:51:48.876245Z","shell.execute_reply":"2024-04-15T19:51:48.882521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_fig(malignant,'Malignant cases')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:48.884460Z","iopub.execute_input":"2024-04-15T19:51:48.885244Z","iopub.status.idle":"2024-04-15T19:51:49.435743Z","shell.execute_reply.started":"2024-04-15T19:51:48.885215Z","shell.execute_reply":"2024-04-15T19:51:49.433943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_fig(normal,'Normal cases')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:49.437365Z","iopub.execute_input":"2024-04-15T19:51:49.437763Z","iopub.status.idle":"2024-04-15T19:51:49.962780Z","shell.execute_reply.started":"2024-04-15T19:51:49.437731Z","shell.execute_reply":"2024-04-15T19:51:49.961241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_image(image_file, label=0, with_label=True, subset='train'):\n    image_name = image_file.split('.')[0]\n    \n    if with_label:\n#         label = train_df[train_df['id'] == image_name]['label'].iloc[0]\n        output_dir = f'png/{subset}/{label}'\n    else:\n        output_dir = f'png/{subset}'\n\n    os.makedirs(output_dir, exist_ok=True)\n\n    output_path = f'{output_dir}/{image_name}.png'\n    if not os.path.exists(output_path):\n        with Image.open(Path + f'{subset}/{image_file}') as tiff_img:\n            png = tiff_img.convert(\"RGB\")\n            png.save(output_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:49.964529Z","iopub.execute_input":"2024-04-15T19:51:49.965049Z","iopub.status.idle":"2024-04-15T19:51:49.972076Z","shell.execute_reply.started":"2024-04-15T19:51:49.965015Z","shell.execute_reply":"2024-04-15T19:51:49.970848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_image_wrapper(args):\n    return convert_image(*args)\n\ndef process_images(image_files, labels=[], with_label=True, subset='train'):\n#     num_processes = multiprocessing.cpu_count()\n    \n#     with multiprocessing.Pool(processes=num_processes) as pool:\n#         tasks = [(filename, labels[index] if with_label else 0, with_label, subset) for index, filename in enumerate(image_files)]\n#         for _ in tqdm(pool.imap_unordered(convert_image_wrapper, tasks), total=len(tasks)):\n#             pass\n    tasks = [(filename, labels[index] if with_label else 0, with_label, subset) for index, filename in enumerate(image_files)]\n    for task in tqdm(tasks):\n        convert_image_wrapper(task)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:49.973404Z","iopub.execute_input":"2024-04-15T19:51:49.973739Z","iopub.status.idle":"2024-04-15T19:51:49.986334Z","shell.execute_reply.started":"2024-04-15T19:51:49.973711Z","shell.execute_reply":"2024-04-15T19:51:49.984570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_images((train_df['id']+'.tif').values.tolist(), labels=train_df['label'].values.tolist())","metadata":{"execution":{"iopub.status.busy":"2024-04-15T19:51:49.989215Z","iopub.execute_input":"2024-04-15T19:51:49.989704Z","iopub.status.idle":"2024-04-15T20:33:20.968843Z","shell.execute_reply.started":"2024-04-15T19:51:49.989665Z","shell.execute_reply":"2024-04-15T20:33:20.966765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_images(os.listdir(Path+\"test\"), with_label=False, subset='test')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:33:20.971877Z","iopub.execute_input":"2024-04-15T20:33:20.973759Z","iopub.status.idle":"2024-04-15T20:46:23.374575Z","shell.execute_reply.started":"2024-04-15T20:33:20.973715Z","shell.execute_reply":"2024-04-15T20:46:23.373615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = tf.keras.utils.image_dataset_from_directory('/kaggle/working/png/train', \n                                                            label_mode='binary',\n                                                            image_size=(96,96), \n                                                            seed=42,\n                                                            validation_split=0.2,\n                                                            subset='training',\n                                                            batch_size=128)\n\nval_dataset = tf.keras.utils.image_dataset_from_directory('/kaggle/working/png/train', \n                                                            label_mode='binary',\n                                                            image_size=(96,96), \n                                                            seed=42,\n                                                            validation_split=0.2,\n                                                            subset='validation',\n                                                            batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:46:23.375963Z","iopub.execute_input":"2024-04-15T20:46:23.376452Z","iopub.status.idle":"2024-04-15T20:46:42.860733Z","shell.execute_reply.started":"2024-04-15T20:46:23.376425Z","shell.execute_reply":"2024-04-15T20:46:42.858917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading the test dataset:\ntest_dataset = tf.keras.utils.image_dataset_from_directory('/kaggle/working/png/test',\n                                                            label_mode=None,\n                                                            image_size=(96,96),\n                                                            shuffle=False,\n                                                            batch_size=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:46:42.863023Z","iopub.execute_input":"2024-04-15T20:46:42.863475Z","iopub.status.idle":"2024-04-15T20:46:44.900665Z","shell.execute_reply.started":"2024-04-15T20:46:42.863443Z","shell.execute_reply":"2024-04-15T20:46:44.899180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers, models\n\nmodel = tf.keras.Sequential()\nmodel.add(layers.Rescaling(1./255, input_shape=(96,96,3)))\nmodel.add(layers.Conv2D(32, (2, 2), strides=(2,2), activation='relu'))\nmodel.add(layers.MaxPooling2D((2, 2)))\nmodel.add(layers.Conv2D(64, (2, 2), strides=(2,2), activation='relu'))\nmodel.add(layers.MaxPooling2D((2, 2)))\nmodel.add(layers.Conv2D(128, (2, 2), strides=(2,2), activation='relu'))\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(64, activation='relu'))\nmodel.add(layers.Dense(32, activation='relu'))\nmodel.add(layers.Dense(1, activation='sigmoid'))\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:46:44.902309Z","iopub.execute_input":"2024-04-15T20:46:44.902712Z","iopub.status.idle":"2024-04-15T20:46:45.090979Z","shell.execute_reply.started":"2024-04-15T20:46:44.902678Z","shell.execute_reply":"2024-04-15T20:46:45.089345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer = 'adam',\n  #  loss = 'categorical_crossentropy',\n    loss=tf.keras.losses.BinaryCrossentropy(),\n    metrics = ['accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:46:45.092698Z","iopub.execute_input":"2024-04-15T20:46:45.093619Z","iopub.status.idle":"2024-04-15T20:46:45.112419Z","shell.execute_reply.started":"2024-04-15T20:46:45.093556Z","shell.execute_reply":"2024-04-15T20:46:45.110097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_dataset, epochs=1, validation_data=val_dataset)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:46:45.114909Z","iopub.execute_input":"2024-04-15T20:46:45.115482Z","iopub.status.idle":"2024-04-15T20:50:04.212689Z","shell.execute_reply.started":"2024-04-15T20:46:45.115442Z","shell.execute_reply":"2024-04-15T20:50:04.211420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'], label='accuracy')\nplt.plot(history.history['val_accuracy'], label = 'val_accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.ylim([0.5, 1])\nplt.title('Train vs Validation Accuracy Per Epoch')\nplt.legend(loc='lower right')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:50:04.214716Z","iopub.execute_input":"2024-04-15T20:50:04.215110Z","iopub.status.idle":"2024-04-15T20:50:04.456357Z","shell.execute_reply.started":"2024-04-15T20:50:04.215059Z","shell.execute_reply":"2024-04-15T20:50:04.455711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'], label='loss')\nplt.plot(history.history['val_loss'], label = 'val_loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Train vs Validation Loss Per Epoch')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:50:04.457232Z","iopub.execute_input":"2024-04-15T20:50:04.457582Z","iopub.status.idle":"2024-04-15T20:50:04.664065Z","shell.execute_reply.started":"2024-04-15T20:50:04.457561Z","shell.execute_reply":"2024-04-15T20:50:04.662841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs = os.listdir(\"/kaggle/working/png/test\")\nsub_pred_df = pd.DataFrame(columns=['id', 'label'])\ntest_imgs=sorted(test_imgs)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:50:04.665211Z","iopub.execute_input":"2024-04-15T20:50:04.666075Z","iopub.status.idle":"2024-04-15T20:50:04.723158Z","shell.execute_reply.started":"2024-04-15T20:50:04.666015Z","shell.execute_reply":"2024-04-15T20:50:04.721596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:50:04.728637Z","iopub.execute_input":"2024-04-15T20:50:04.729030Z","iopub.status.idle":"2024-04-15T20:53:26.660819Z","shell.execute_reply.started":"2024-04-15T20:50:04.728998Z","shell.execute_reply":"2024-04-15T20:53:26.659152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pred_df['id'] = [filename.split('.')[0] for filename in test_imgs]\nsub_pred_df['label'] = np.round(predictions.flatten()).astype('int')\nsub_pred_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:26.667948Z","iopub.execute_input":"2024-04-15T20:53:26.668821Z","iopub.status.idle":"2024-04-15T20:53:26.719728Z","shell.execute_reply.started":"2024-04-15T20:53:26.668781Z","shell.execute_reply":"2024-04-15T20:53:26.718687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Running Submission Job\")","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:26.721175Z","iopub.execute_input":"2024-04-15T20:53:26.722061Z","iopub.status.idle":"2024-04-15T20:53:26.726982Z","shell.execute_reply.started":"2024-04-15T20:53:26.722017Z","shell.execute_reply":"2024-04-15T20:53:26.725857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r \"/kaggle/working/png\"","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:26.728403Z","iopub.execute_input":"2024-04-15T20:53:26.728895Z","iopub.status.idle":"2024-04-15T20:53:35.541119Z","shell.execute_reply.started":"2024-04-15T20:53:26.728854Z","shell.execute_reply":"2024-04-15T20:53:35.540165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from glob import glob\n\nfiles = []\nfor file in glob(\"/kaggle/input/histopathologic-cancer-detection/test/*.tif\"):\n    files.append(file)\nfiles[0]","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:35.542558Z","iopub.execute_input":"2024-04-15T20:53:35.542893Z","iopub.status.idle":"2024-04-15T20:53:35.703454Z","shell.execute_reply.started":"2024-04-15T20:53:35.542863Z","shell.execute_reply":"2024-04-15T20:53:35.701656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_SUB = '/kaggle/input/histopathologic-cancer-detection/sample_submission.csv'\nsample_df = pd.read_csv(SAMPLE_SUB)\nsample_list = list(sample_df.id)\nlen(sample_list)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:35.705482Z","iopub.execute_input":"2024-04-15T20:53:35.705831Z","iopub.status.idle":"2024-04-15T20:53:35.799786Z","shell.execute_reply.started":"2024-04-15T20:53:35.705791Z","shell.execute_reply":"2024-04-15T20:53:35.798408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_dict = dict((key, value) for (key, value) in zip(sub_pred_df.id, sub_pred_df.label))\npred_corr = [pred_dict[id] for id in sample_list]","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:35.801209Z","iopub.execute_input":"2024-04-15T20:53:35.801495Z","iopub.status.idle":"2024-04-15T20:53:36.182470Z","shell.execute_reply.started":"2024-04-15T20:53:35.801472Z","shell.execute_reply":"2024-04-15T20:53:36.180450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\"id\": sample_list, \"label\": pred_corr})\nsubmission_df.to_csv(\"submission.csv\", header=True, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T20:53:36.183447Z","iopub.status.idle":"2024-04-15T20:53:36.183880Z","shell.execute_reply.started":"2024-04-15T20:53:36.183680Z","shell.execute_reply":"2024-04-15T20:53:36.183695Z"},"trusted":true},"execution_count":null,"outputs":[]}]}