{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Conv1D TF Model with Separate Pipelines for defog and tdcsfog data\nFeature Column TimeSeries grouping adapted from https://www.kaggle.com/code/mayukh18/pytorch-fog-end-to-end-baseline-lb-0-254\n\nSubject-wise GroupKFold splitting adapted from https://www.kaggle.com/code/xzj19013742/groupkfold-cross-validation-tsflex\n### In this Notebook\n- Tensorflow Model with Conv1D blocks \n- Models trained separately for defog and tdcsfog data\n- Event Stratified Subject Grouped KFold splitting","metadata":{}},{"cell_type":"markdown","source":"## Imports and Config","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport numpy as np\nfrom numpy.random import default_rng\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom os.path import basename, dirname, join, exists\nfrom time import perf_counter\nfrom collections import defaultdict as dd\nfrom functools import partial\n\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, StratifiedGroupKFold\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import StandardScaler as Scaler\nfrom scipy.special import expit\n\nimport tensorflow as tf\nprint(f\"TF version: {tf.__version__}\")\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-04T09:23:27.596476Z","iopub.execute_input":"2023-04-04T09:23:27.596840Z","iopub.status.idle":"2023-04-04T09:23:36.481299Z","shell.execute_reply.started":"2023-04-04T09:23:27.596807Z","shell.execute_reply":"2023-04-04T09:23:36.480063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Constants\n\nBASE_DIR = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction\"\nTRAIN_DIR = join(BASE_DIR, \"train\")\nTEST_DIR = join(BASE_DIR, \"test\")\n\nIS_PUBLIC = len(glob(join(TEST_DIR, \"*/*.csv\")))==2","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:23:36.483580Z","iopub.execute_input":"2023-04-04T09:23:36.485714Z","iopub.status.idle":"2023-04-04T09:23:36.502295Z","shell.execute_reply.started":"2023-04-04T09:23:36.485667Z","shell.execute_reply":"2023-04-04T09:23:36.501257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    train_sub_dirs = [\n        join(TRAIN_DIR, \"defog\"),\n        join(TRAIN_DIR, \"tdcsfog\")\n    ]\n    \n    metadata_paths = [\n        join(BASE_DIR, \"defog_metadata.csv\"),\n        join(BASE_DIR, \"tdcsfog_metadata.csv\")\n    ]\n    \n    splits = 10\n\n    batch_size = 1024\n    window_size = 64\n    window_future = 16\n    window_past = window_size - window_future # Includes current value\n    \n    wx = 8\n    \n    model_dropout = 0.2\n    model_hidden = 128\n    model_nblocks = 3\n    \n    lr = 0.00015\n    num_epochs = 5\n    \n    feature_list = ['AccV', 'AccML', 'AccAP']\n    label_list = ['StartHesitation', 'Turn', 'Walking']\n    \n    n_features = len(feature_list)\n    n_labels = len(label_list)    \n    \ncfg = Config()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:23:36.505724Z","iopub.execute_input":"2023-04-04T09:23:36.506566Z","iopub.status.idle":"2023-04-04T09:23:36.514939Z","shell.execute_reply.started":"2023-04-04T09:23:36.506518Z","shell.execute_reply":"2023-04-04T09:23:36.513520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stratified Group K Fold","metadata":{}},{"cell_type":"code","source":"# Create Mapping between Id and Subject\nid2sub_df = pd.concat([\n    pd.read_csv(f, usecols=['Id', 'Subject']).assign(Module=basename(f).split('_')[0]) for f in cfg.metadata_paths\n]).astype(\"category\").set_index(\"Id\")\nprint(f\"id2sub_df length: {len(id2sub_df)}, unique Ids: {id2sub_df.index.nunique()}, unique Subjects: {id2sub_df.Subject.nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:23:36.519460Z","iopub.execute_input":"2023-04-04T09:23:36.520312Z","iopub.status.idle":"2023-04-04T09:23:36.564413Z","shell.execute_reply.started":"2023-04-04T09:23:36.520268Z","shell.execute_reply":"2023-04-04T09:23:36.563220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read csv files and add metadata (Id, Subject, Event)\ndef reader(filepath, usecols, getid=False, getsub=False, getevent=False, dtype=None, exclude=['notype']):\n    fog_type = basename(dirname(filepath))\n    if fog_type in exclude:\n        return None\n    df = pd.read_csv(filepath, index_col=\"Time\", usecols=usecols, dtype=dtype)\n    if getid:\n        df['Id'] = basename(filepath).split('.')[0] + '_' + df.index.astype(str)\n    if getsub:\n        df['Subject'] = id2sub_df.loc[basename(filepath).split('.')[0], 'Subject']\n    if getevent:\n        df['Event'] = np.select(\n            [df[col].astype(bool) for col in cfg.label_list], \n            np.arange(1,cfg.n_labels+1), default=0\n        ).astype('int8')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:23:36.566414Z","iopub.execute_input":"2023-04-04T09:23:36.566841Z","iopub.status.idle":"2023-04-04T09:23:36.576437Z","shell.execute_reply.started":"2023-04-04T09:23:36.566800Z","shell.execute_reply":"2023-04-04T09:23:36.575119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create common train Dataframe\ntrain_paths = glob(join(TRAIN_DIR, '*/*.csv'))\ndtype = {col:'int8' for col in cfg.label_list}\ndtype['Time'] = 'int32'\nusecols = ['Time', *cfg.label_list]\n\ntrain_reader = partial(reader, usecols=usecols, dtype=dtype, getsub=True, getevent=True)\ntrain_df = pd.concat([train_reader(f) for f in tqdm(train_paths)]).reset_index(drop=True)\ntrain_df.Subject = train_df.Subject.astype('category')\ndisplay(train_df.Event.value_counts().to_frame().style.background_gradient())","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:23:36.578273Z","iopub.execute_input":"2023-04-04T09:23:36.578851Z","iopub.status.idle":"2023-04-04T09:24:17.386114Z","shell.execute_reply.started":"2023-04-04T09:23:36.578811Z","shell.execute_reply":"2023-04-04T09:24:17.384954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save paths for each Stratified Group Fold for defog and tdcsfog separately\nsgkf = StratifiedGroupKFold(n_splits=cfg.splits, random_state=42, shuffle=True)\nfold_train_fpaths, fold_valid_fpaths = {'defog': [], 'tdcsfog':[]}, {'defog': [], 'tdcsfog':[]}\ndf_paths = {'defog':glob(join(cfg.train_sub_dirs[0],'*.csv')), 'tdcsfog': glob(join(cfg.train_sub_dirs[1],'*.csv'))}\nfor module, paths in df_paths.items():\n    print(f\"{module}:\")\n    sub_train_df = train_df[train_df.Subject.isin(id2sub_df.loc[id2sub_df.Module==module, 'Subject'])].reset_index(drop=True)\n    for i, (train_index, test_index) in enumerate(sgkf.split(sub_train_df.index, sub_train_df.Event, groups=sub_train_df.Subject)):\n        print(f\"\\tFold {i}:\", end=\" \")\n        train_subs = sub_train_df.loc[train_index, 'Subject'].unique()\n        test_subs = sub_train_df.loc[test_index, 'Subject'].unique()\n        print(f\"Subjects->train:{len(train_subs)}|test:{len(test_subs)}\")\n        train_ids = set(id2sub_df[id2sub_df.Subject.isin(train_subs)].index)\n        test_ids = set(id2sub_df[id2sub_df.Subject.isin(test_subs)].index)\n        fold_train_fpaths[module].append([f for f in paths if basename(f).split('.')[0] in train_ids])\n        fold_valid_fpaths[module].append([f for f in paths if basename(f).split('.')[0] in test_ids])\n        del train_subs, test_subs, train_ids, test_ids\n        gc.collect()\n    del sub_train_df\ndel train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:24:17.387810Z","iopub.execute_input":"2023-04-04T09:24:17.388480Z","iopub.status.idle":"2023-04-04T09:25:31.376487Z","shell.execute_reply.started":"2023-04-04T09:24:17.388438Z","shell.execute_reply":"2023-04-04T09:25:31.375433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"code","source":"# Adapted from FOGDataset of https://www.kaggle.com/code/mayukh18/pytorch-fog-end-to-end-baseline-lb-0-254\nclass FOGSequence(tf.keras.utils.Sequence):\n\n    def __init__(self, df_paths, cfg=cfg, split=\"train\"):\n        _time = perf_counter()\n        \n        self.rng = default_rng(42)\n        self.cfg = cfg\n        self.split = split\n        \n        self.past_pad = self.cfg.wx*(self.cfg.window_past-1)\n        self.future_pad = self.cfg.wx*self.cfg.window_future\n        \n        if self.split == \"test\":\n            self.Ids = []\n        _values = [self._read(f) for f in df_paths]\n        \n        self.mapping = []\n        _length = 0\n        for _value in _values:\n            _shape = _value.shape[0]\n            self.mapping.extend(range(_length+self.past_pad, _length+_shape-self.future_pad))\n            _length += _shape\n            \n        self.values = np.concatenate(_values, axis=0)\n        self.mapping = np.array(self.mapping)\n        if self.split != \"test\":\n            # Keep only vaild and task rows\n            _valid_pos = self.values[self.mapping,self.valid_position] > 0\n            _task_pos = self.values[self.mapping,self.task_position] > 0\n            self.mapping = self.mapping[_valid_pos&_task_pos]\n        self.length = self.mapping.shape[0]\n        \n        print(f\"Valid Dataset of size {self.length:,} initialized in {perf_counter() - _time:.3f} secs!\")\n        gc.collect()\n    \n    def _read(self, path):\n        _is_tdcs = basename(dirname(path)).startswith('tdcs')\n        df = pd.read_csv(path)\n        \n        if self.split == \"test\":\n            _ids = basename(path).split('.')[0] + '_' + df.Time.astype(str)\n            self.Ids.extend(_ids.tolist())\n            return self._df_to_array(df, self.cfg.feature_list)\n        \n        _cols = [*self.cfg.feature_list, *self.cfg.label_list, 'Valid', 'Task']\n        self.valid_position = self.cfg.n_features + self.cfg.n_labels\n        self.task_position = self.valid_position + 1\n        \n        if _is_tdcs:\n            # Fill Valid and Task columns for tdcsfog\n            df['Valid'] = 1\n            df['Task'] = 1\n            \n        return self._df_to_array(df, _cols)\n    \n    def _df_to_array(self, df, cols):\n        # Pads past and future rows to dataframe values for indexing \n        _values = df[cols].values.astype(np.float16)\n        return np.pad(_values, ((self.past_pad, self.future_pad),(0,0)), 'edge')\n    \n    def __len__(self):\n        return int(np.ceil(self.length / self.cfg.batch_size))\n    \n    def __getitem__(self, idx):\n        \n        if self.split == \"train\":\n            # Onlt train set has randomly selected batches\n            _idxs = self.rng.choice(self.mapping, size=self.cfg.batch_size, replace=False)\n        else:\n            _idxs = self._get_indices(idx)\n            \n        # For test return only features\n        if self.split == \"test\":\n            return self._get_X(_idxs)\n        # For train and val splits return y also\n        return self._get_X_y(_idxs)\n    \n    def _get_indices(self, idx):\n        _low = idx * self.cfg.batch_size\n        # Cap high at self.length so overflow does not occur\n        _high = min(_low + self.cfg.batch_size, self.length)\n        return self.mapping[_low:_high]\n    \n    def _get_X(self, indices):\n        _X = np.empty((len(indices), self.cfg.window_size, self.cfg.n_features), dtype=np.float16)\n        for i, idx in enumerate(indices):\n            _X[i] = self.values[idx-self.past_pad:idx+self.future_pad+1:self.cfg.wx, :self.cfg.n_features]\n        return _X\n    \n    def _get_X_y(self, indices):\n        _X = np.empty((len(indices), self.cfg.window_size, self.cfg.n_features), dtype=np.float16)\n        for i, idx in enumerate(indices):\n            _X[i] = self.values[idx-self.past_pad: idx+self.future_pad+1:self.cfg.wx, :self.cfg.n_features]\n        return _X, self.values[indices, self.cfg.n_features:self.cfg.n_features+self.cfg.n_labels]","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:25:31.378311Z","iopub.execute_input":"2023-04-04T09:25:31.378808Z","iopub.status.idle":"2023-04-04T09:25:31.401404Z","shell.execute_reply.started":"2023-04-04T09:25:31.378763Z","shell.execute_reply":"2023-04-04T09:25:31.400178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"# average_precision_score with positive sample added if no true positive cases are present\ndef calculate_precision(y_true, y_pred):\n    pad_width = ((0,0),(0,0)) if y_true.any(axis=0).all() else ((1,0),(0,0))\n    y_true, y_pred = np.pad(y_true, pad_width, constant_values=1), np.pad(y_pred, pad_width, constant_values=1)\n    return average_precision_score(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:25:31.403074Z","iopub.execute_input":"2023-04-04T09:25:31.403470Z","iopub.status.idle":"2023-04-04T09:25:31.419007Z","shell.execute_reply.started":"2023-04-04T09:25:31.403431Z","shell.execute_reply":"2023-04-04T09:25:31.417772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Note: Not same result as average_precision_score\nclass AveragePrecision(tf.keras.metrics.Metric):\n\n    def __init__(self, num_classes, thresholds=None, name='avg_precision', **kwargs):\n        super(AveragePrecision, self).__init__(name=name, **kwargs)\n        self.class_precision = [tf.keras.metrics.Precision(thresholds) for _ in range(num_classes)]\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        for i, precision in enumerate(self.class_precision):\n            precision.update_state(y_true[...,i], y_pred[..., i])\n\n    def result(self):\n        return tf.math.reduce_mean([precision.result() for precision in self.class_precision])\n    \n    def reset_state(self):\n        for precision in self.class_precision:\n            precision.reset_state()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:25:31.422959Z","iopub.execute_input":"2023-04-04T09:25:31.423419Z","iopub.status.idle":"2023-04-04T09:25:31.434144Z","shell.execute_reply.started":"2023-04-04T09:25:31.423376Z","shell.execute_reply":"2023-04-04T09:25:31.432935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model adapted from https://keras.io/examples/timeseries/timeseries_classification_from_scratch/\ndef get_model(checkpoint_path = None):\n    model = tf.keras.models.Sequential()\n    model.add(tf.keras.Input(shape=(cfg.window_size, cfg.n_features), dtype='float16'))\n    for _ in range(cfg.model_nblocks):\n        model.add(tf.keras.layers.Conv1D(filters=cfg.model_hidden, kernel_size=15, padding=\"same\"))\n        model.add(tf.keras.layers.BatchNormalization())\n        model.add(tf.keras.layers.ReLU())\n        model.add(tf.keras.layers.Dropout(cfg.model_dropout))\n    model.add(tf.keras.layers.GlobalAveragePooling1D())\n    model.add(tf.keras.layers.Dense(cfg.n_labels, activation=None))\n\n    if checkpoint_path is not None:\n        model.load_weights(checkpoint_path)\n    model.compile(\n        tf.keras.optimizers.Adam(learning_rate=cfg.lr), \n        loss = tf.keras.losses.BinaryCrossentropy(from_logits=True),\n        metrics=[AveragePrecision(cfg.n_labels, thresholds=0.0)]\n    )\n    return model\n\nget_model().summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:25:31.435863Z","iopub.execute_input":"2023-04-04T09:25:31.436273Z","iopub.status.idle":"2023-04-04T09:25:33.903424Z","shell.execute_reply.started":"2023-04-04T09:25:31.436235Z","shell.execute_reply":"2023-04-04T09:25:33.902554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train","metadata":{}},{"cell_type":"code","source":"def predict_select_model(fold, ds, model_save_dir=''):\n    best_path, best_score = None, -1\n    print(f\"Validation for fold{fold}:\")\n    for model_path in sorted(glob(join(model_save_dir, f\"fold{fold}_*.h5\"))):\n        pred_time = perf_counter()\n        gc.collect()\n        score = calculate_precision(\n            ds.values[ds.mapping, cfg.n_features:cfg.n_features + cfg.n_labels], \n            expit(get_model(model_path).predict(ds, verbose=0)) # expit converts to sigmoid output\n        )\n        if best_score < score:\n            best_score = score\n            best_path = model_path\n        gc.collect()\n        print(\"\\t\", basename(model_path), f\": score-{score:.4f} in {(perf_counter()-pred_time)/60:.2f} mins\")\n    print(basename(best_path), \"selected with score\", best_score)\n    return best_path","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:25:33.904664Z","iopub.execute_input":"2023-04-04T09:25:33.905087Z","iopub.status.idle":"2023-04-04T09:25:33.917924Z","shell.execute_reply.started":"2023-04-04T09:25:33.905047Z","shell.execute_reply":"2023-04-04T09:25:33.917057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_loop(train_paths, valid_paths, fold, model_save_dir=''):\n    gc.collect()\n    \n    # train_paths, test_paths = train_test_split(train_paths, test_size=0.4)\n    train_ds = FOGSequence(train_paths)\n    val_ds = FOGSequence(valid_paths, split=\"val\")\n    # test_ds = FOGSequence(test_paths, split=\"val\")\n    \n    model = get_model()\n    ckpt = tf.keras.callbacks.ModelCheckpoint(join(model_save_dir, f\"fold{fold}_model_\"+\"{epoch:02d}.h5\"), save_weights_only=True)\n    history = model.fit(train_ds, epochs=cfg.num_epochs, verbose=2, workers=5, validation_data=val_ds, use_multiprocessing=True, callbacks=[ckpt])\n    \n    best_model_path = predict_select_model(fold, val_ds, model_save_dir)\n    # score = calculate_precision(test_ds.values[test_ds.mapping, cfg.n_features:cfg.n_features + cfg.n_labels], expit(get_model(best_model_path).predict(test_ds, verbose=0)))\n    \n    del train_ds, val_ds, model, ckpt, history\n    gc.collect()\n    return best_model_path","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:25:33.919290Z","iopub.execute_input":"2023-04-04T09:25:33.919841Z","iopub.status.idle":"2023-04-04T09:25:33.933125Z","shell.execute_reply.started":"2023-04-04T09:25:33.919799Z","shell.execute_reply":"2023-04-04T09:25:33.932263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Main training loop\nmodel_paths = {'defog': [], 'tdcsfog':[]}\nfor module in model_paths:\n    module_start = perf_counter()\n    print(f\"***Training {module}{'*'*75}\")\n    if not exists(module): \n        os.mkdir(module)\n    for fold, (train_fpaths, valid_fpaths) in enumerate(zip(fold_train_fpaths[module], fold_valid_fpaths[module])):\n        fold_start = perf_counter()\n        print(f\"Fold {fold}{'-'*25}\")\n        model_paths[module].append(train_loop(train_fpaths, valid_fpaths, fold, model_save_dir=module))\n        print(f\"Fold {fold} done in {(perf_counter()-fold_start)/60:.2f} min\")\n    print(f\"***{module} done in {(perf_counter()-module_start)/3600:.2f} hrs{'*'*50}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-04-04T08:47:09.041219Z","iopub.execute_input":"2023-04-04T08:47:09.042066Z","iopub.status.idle":"2023-04-04T09:06:01.256473Z","shell.execute_reply.started":"2023-04-04T08:47:09.042030Z","shell.execute_reply":"2023-04-04T09:06:01.255137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"test_defog_paths = glob(join(TEST_DIR, \"defog/*.csv\"))\ntest_tdcsfog_paths = glob(join(TEST_DIR, \"tdcsfog/*.csv\"))\n\ntest_ds_dict = {\n    'defog':FOGSequence(test_defog_paths, split=\"test\"), \n    'tdcsfog':FOGSequence(test_tdcsfog_paths, split=\"test\")\n}\n\n# Get test predictions\ndf_list = []\nfor module, test_ds in test_ds_dict.items():\n    y_pred_list = []\n    for model_path in model_paths[module]:\n        model = get_model(model_path)\n        y_pred_list.append(expit(model.predict(test_ds, verbose=0)))  # expit converts to sigmoid output\n    y_pred = np.mean(y_pred_list, axis=0)\n    df_list.append(pd.DataFrame(\n        {'Id': test_ds.Ids, 'StartHesitation': y_pred[:,0], 'Turn': y_pred[:,1], 'Walking': y_pred[:,2]}))","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:06:01.270977Z","iopub.execute_input":"2023-04-04T09:06:01.271624Z","iopub.status.idle":"2023-04-04T09:06:11.801757Z","shell.execute_reply.started":"2023-04-04T09:06:01.271586Z","shell.execute_reply":"2023-04-04T09:06:11.800667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate Prediction to DataFrames\nsubmission = pd.concat(df_list)\n\n# Only keep Ids in sample_submission\nsample_submission = pd.read_csv(join(BASE_DIR, \"sample_submission.csv\"))\nsubmission = pd.merge(sample_submission[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index=False, float_format='%.5f') # round to 5 decimal places while keeping point notation","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:06:11.804310Z","iopub.execute_input":"2023-04-04T09:06:11.805087Z","iopub.status.idle":"2023-04-04T09:06:13.952700Z","shell.execute_reply.started":"2023-04-04T09:06:11.805043Z","shell.execute_reply":"2023-04-04T09:06:13.951598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head -5 submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-04-04T09:06:13.954138Z","iopub.execute_input":"2023-04-04T09:06:13.955169Z","iopub.status.idle":"2023-04-04T09:06:15.016791Z","shell.execute_reply.started":"2023-04-04T09:06:13.955128Z","shell.execute_reply":"2023-04-04T09:06:15.015494Z"},"trusted":true},"execution_count":null,"outputs":[]}]}