{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport torch\nfrom torch import nn\nfrom torch import optim\nfrom torch.nn import functional as F\nfrom torch.utils.data import DataLoader, Dataset\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T20:11:25.413883Z","iopub.execute_input":"2024-11-21T20:11:25.414289Z","iopub.status.idle":"2024-11-21T20:11:30.620362Z","shell.execute_reply.started":"2024-11-21T20:11:25.414252Z","shell.execute_reply":"2024-11-21T20:11:30.61928Z"}},"outputs":[],"execution_count":1},{"cell_type":"code","source":"data_path = '/kaggle/input/amex-default-prediction/train_data.csv'\nlabels_path = '/kaggle/input/amex-default-prediction/train_labels.csv'\ndata_rounded = pd.read_csv(data_path)\nlabels = pd.read_csv(labels_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T20:52:00.320831Z","iopub.execute_input":"2024-11-21T20:52:00.321266Z","iopub.status.idle":"2024-11-21T20:52:01.346696Z","shell.execute_reply.started":"2024-11-21T20:52:00.321227Z","shell.execute_reply":"2024-11-21T20:52:01.345212Z"}},"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mParserError\u001b[0m                               Traceback (most recent call last)","Cell \u001b[0;32mIn[8], line 3\u001b[0m\n\u001b[1;32m      1\u001b[0m data_path \u001b[38;5;241m=\u001b[39m \u001b[38;5;124m'\u001b[39m\u001b[38;5;124m/kaggle/input/amex-default-prediction/train_data.csv\u001b[39m\u001b[38;5;124m'\u001b[39m\n\u001b[1;32m      2\u001b[0m labels_path \u001b[38;5;241m=\u001b[39m \u001b[38;5;124m'\u001b[39m\u001b[38;5;124m/kaggle/input/amex-default-prediction/train_labels.csv\u001b[39m\u001b[38;5;124m'\u001b[39m\n\u001b[0;32m----> 3\u001b[0m data_rounded \u001b[38;5;241m=\u001b[39m \u001b[43mpd\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mread_csv\u001b[49m\u001b[43m(\u001b[49m\u001b[43mdata_path\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m      4\u001b[0m labels \u001b[38;5;241m=\u001b[39m pd\u001b[38;5;241m.\u001b[39mread_csv(labels_path)\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/pandas/io/parsers/readers.py:1026\u001b[0m, in \u001b[0;36mread_csv\u001b[0;34m(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, delim_whitespace, low_memory, memory_map, float_precision, storage_options, dtype_backend)\u001b[0m\n\u001b[1;32m   1013\u001b[0m kwds_defaults \u001b[38;5;241m=\u001b[39m _refine_defaults_read(\n\u001b[1;32m   1014\u001b[0m     dialect,\n\u001b[1;32m   1015\u001b[0m     delimiter,\n\u001b[0;32m   (...)\u001b[0m\n\u001b[1;32m   1022\u001b[0m     dtype_backend\u001b[38;5;241m=\u001b[39mdtype_backend,\n\u001b[1;32m   1023\u001b[0m )\n\u001b[1;32m   1024\u001b[0m kwds\u001b[38;5;241m.\u001b[39mupdate(kwds_defaults)\n\u001b[0;32m-> 1026\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[43m_read\u001b[49m\u001b[43m(\u001b[49m\u001b[43mfilepath_or_buffer\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43mkwds\u001b[49m\u001b[43m)\u001b[49m\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/pandas/io/parsers/readers.py:626\u001b[0m, in \u001b[0;36m_read\u001b[0;34m(filepath_or_buffer, kwds)\u001b[0m\n\u001b[1;32m    623\u001b[0m     \u001b[38;5;28;01mreturn\u001b[39;00m parser\n\u001b[1;32m    625\u001b[0m \u001b[38;5;28;01mwith\u001b[39;00m parser:\n\u001b[0;32m--> 626\u001b[0m     \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[43mparser\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mread\u001b[49m\u001b[43m(\u001b[49m\u001b[43mnrows\u001b[49m\u001b[43m)\u001b[49m\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/pandas/io/parsers/readers.py:1923\u001b[0m, in \u001b[0;36mTextFileReader.read\u001b[0;34m(self, nrows)\u001b[0m\n\u001b[1;32m   1916\u001b[0m nrows \u001b[38;5;241m=\u001b[39m validate_integer(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mnrows\u001b[39m\u001b[38;5;124m\"\u001b[39m, nrows)\n\u001b[1;32m   1917\u001b[0m \u001b[38;5;28;01mtry\u001b[39;00m:\n\u001b[1;32m   1918\u001b[0m     \u001b[38;5;66;03m# error: \"ParserBase\" has no attribute \"read\"\u001b[39;00m\n\u001b[1;32m   1919\u001b[0m     (\n\u001b[1;32m   1920\u001b[0m         index,\n\u001b[1;32m   1921\u001b[0m         columns,\n\u001b[1;32m   1922\u001b[0m         col_dict,\n\u001b[0;32m-> 1923\u001b[0m     ) \u001b[38;5;241m=\u001b[39m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_engine\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mread\u001b[49m\u001b[43m(\u001b[49m\u001b[43m  \u001b[49m\u001b[38;5;66;43;03m# type: ignore[attr-defined]\u001b[39;49;00m\n\u001b[1;32m   1924\u001b[0m \u001b[43m        \u001b[49m\u001b[43mnrows\u001b[49m\n\u001b[1;32m   1925\u001b[0m \u001b[43m    \u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m   1926\u001b[0m \u001b[38;5;28;01mexcept\u001b[39;00m \u001b[38;5;167;01mException\u001b[39;00m:\n\u001b[1;32m   1927\u001b[0m     \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mclose()\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/pandas/io/parsers/c_parser_wrapper.py:234\u001b[0m, in \u001b[0;36mCParserWrapper.read\u001b[0;34m(self, nrows)\u001b[0m\n\u001b[1;32m    232\u001b[0m \u001b[38;5;28;01mtry\u001b[39;00m:\n\u001b[1;32m    233\u001b[0m     \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mlow_memory:\n\u001b[0;32m--> 234\u001b[0m         chunks \u001b[38;5;241m=\u001b[39m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_reader\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mread_low_memory\u001b[49m\u001b[43m(\u001b[49m\u001b[43mnrows\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m    235\u001b[0m         \u001b[38;5;66;03m# destructive to chunks\u001b[39;00m\n\u001b[1;32m    236\u001b[0m         data \u001b[38;5;241m=\u001b[39m _concatenate_chunks(chunks)\n","File \u001b[0;32mparsers.pyx:838\u001b[0m, in \u001b[0;36mpandas._libs.parsers.TextReader.read_low_memory\u001b[0;34m()\u001b[0m\n","File \u001b[0;32mparsers.pyx:905\u001b[0m, in \u001b[0;36mpandas._libs.parsers.TextReader._read_rows\u001b[0;34m()\u001b[0m\n","File \u001b[0;32mparsers.pyx:874\u001b[0m, in \u001b[0;36mpandas._libs.parsers.TextReader._tokenize_rows\u001b[0;34m()\u001b[0m\n","File \u001b[0;32mparsers.pyx:891\u001b[0m, in \u001b[0;36mpandas._libs.parsers.TextReader._check_tokenize_status\u001b[0;34m()\u001b[0m\n","File \u001b[0;32mparsers.pyx:2061\u001b[0m, in \u001b[0;36mpandas._libs.parsers.raise_parser_error\u001b[0;34m()\u001b[0m\n","\u001b[0;31mParserError\u001b[0m: Error tokenizing data. C error: Calling read(nbytes) on source failed. Try engine='python'."],"ename":"ParserError","evalue":"Error tokenizing data. C error: Calling read(nbytes) on source failed. Try engine='python'.","output_type":"error"}],"execution_count":8},{"cell_type":"code","source":"num_selected = 50_000\nbatch_size = 256","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T20:15:28.466682Z","iopub.execute_input":"2024-11-21T20:15:28.467079Z","iopub.status.idle":"2024-11-21T20:15:28.472123Z","shell.execute_reply.started":"2024-11-21T20:15:28.467044Z","shell.execute_reply":"2024-11-21T20:15:28.470945Z"}},"outputs":[],"execution_count":3},{"cell_type":"code","source":"class Preprocesser:\n    def __init__(self):\n        self.scaler = StandardScaler()\n        self.con_col_means = None\n        \n        \n    def preprocess_data(self,data_rounded, num_selected):\n        self.scaler = StandardScaler()\n        \n        selected_cus_id = list(data_rounded['customer_ID'].unique())[:num_selected]\n        data_rounded = data_rounded[data_rounded['customer_ID'].isin(selected_cus_id)]\n        \n        cus_id = data_rounded.groupby('customer_ID').size()\n        # discarded_cus_id = list(cus_id[cus_id != 13].index)\n        # for length not equals 13, add padding of zero vector\n        \n        # discard rows where customer id in discarded_cus_id\n        # data_rounded = data_rounded[~data_rounded['customer_ID'].isin(discarded_cus_id)]\n        \n        # replace -32768 with NaN values\n        data_rounded = data_rounded.replace(-32768, np.nan)\n        \n        # drop cols with > 10% NaN values\n        # cols_to_drop = data_rounded.columns[data_rounded.isnull().mean() > 0.1]\n        # data_rounded = data_rounded.drop(cols_to_drop, axis=1)\n        \n        cat_columns = ['D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68', 'B_30', 'B_38']\n        cat_columns = [col for col in cat_columns if col in data_rounded.columns]\n        con_columnns = [col for col in data_rounded.columns[2:] if col not in cat_columns]\n        \n        data_rounded[con_columnns] = data_rounded[con_columnns].apply(lambda col: col.fillna(col.mean()))\n        self.con_col_means = data_rounded[con_columnns].mean()\n        \n        # one hot encode the following data frame\n        df = data_rounded[cat_columns]\n        cat_df = pd.get_dummies(df, columns=cat_columns)\n\n        # make it integer instead of boolean\n        cat_df = cat_df.astype(int)\n        data_rounded = data_rounded.drop(cat_columns, axis=1)\n        # concat columns cat_df\n        data_rounded = pd.concat([data_rounded, cat_df], axis=1)\n        \n        train_cus_id, test_cus_id = train_test_split(cus_id, test_size=0.1, random_state=42)\n        \n        train_data = data_rounded[data_rounded['customer_ID'].isin(train_cus_id.index)]\n        test_data = data_rounded[data_rounded['customer_ID'].isin(test_cus_id.index)]\n        \n        # normalize the con_columns using library\n        train_data[con_columnns] = self.scaler.fit_transform(train_data[con_columnns])\n        test_data[con_columnns] = self.scaler.transform(test_data[con_columnns])\n        \n        train_extra_pad = 13 - train_data.groupby('customer_ID').size()\n        test_extra_pad = 13 - test_data.groupby('customer_ID').size()\n        \n        return train_data, test_data, train_extra_pad, test_extra_pad\n    \n    def preprocess_eval_data(self, data_rounded):\n        \n        cus_id = data_rounded.groupby('customer_ID').size()\n        # discarded_cus_id = list(cus_id[cus_id != 13].index)\n        # for length not equals 13, add padding of zero vector\n        \n        # discard rows where customer id in discarded_cus_id\n        # data_rounded = data_rounded[~data_rounded['customer_ID'].isin(discarded_cus_id)]\n        \n        # replace -32768 with NaN values\n        data_rounded = data_rounded.replace(-32768, np.nan)\n        \n        # drop cols with > 10% NaN values\n        # cols_to_drop = data_rounded.columns[data_rounded.isnull().mean() > 0.1]\n        # data_rounded = data_rounded.drop(cols_to_drop, axis=1)\n        \n        cat_columns = ['D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68', 'B_30', 'B_38']\n        cat_columns = [col for col in cat_columns if col in data_rounded.columns]\n        con_columnns = [col for col in data_rounded.columns[2:] if col not in cat_columns]\n        \n        data_rounded[con_columnns] = data_rounded[con_columnns].apply(\n            lambda col: col.fillna(self.con_col_means[col.name])\n        )\n        \n        # one hot encode the following data frame\n        df = data_rounded[cat_columns]\n        cat_df = pd.get_dummies(df, columns=cat_columns)\n\n        # make it integer instead of boolean\n        cat_df = cat_df.astype(int)\n        data_rounded = data_rounded.drop(cat_columns, axis=1)\n        # concat columns cat_df\n        eval_data = pd.concat([data_rounded, cat_df], axis=1)\n        \n        \n        # normalize the con_columns using library\n        eval_data[con_columnns] = self.scaler.transform(eval_data[con_columnns])\n        \n        eval_extra_pad = 13 - eval_data.groupby('customer_ID').size()\n        \n        return eval_data, eval_extra_pad","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = Preprocesser()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, test_data, train_extra_pad, test_extra_pad = preprocessor.preprocess_data(data_rounded, num_selected)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    def __init__(self, data, labels, extra_pad, eval=False):\n        self.eval = eval\n        self.data = data.copy()\n        if not self.eval:\n            self.labels = labels.set_index(\"customer_ID\")[\"target\"]\n        self.groups = data.groupby(\"customer_ID\")\n        self.customer_ids = list(self.groups.groups.keys())\n\n        extra_pad = extra_pad.reset_index()\n        extra_pad.columns = ['customer_ID', 'extra_pad']\n        extra_pad['customer_ID'] = extra_pad['customer_ID'].astype(str)\n        self.extra_pad = extra_pad.set_index('customer_ID')\n        \n        input_vector = self.data[self.data.columns[2:]].apply(lambda row: list(row), axis=1)\n        self.data['input_vector'] = input_vector\n        \n        self.data = self.data.groupby('customer_ID').apply(\n            lambda x: x['input_vector'].tolist()\n        ).reset_index(name='input_vector')\n        \n        self.data = self.data.join(self.extra_pad, on='customer_ID', how='inner')\n        \n        if not self.eval:\n            self.data = self.data.join(self.labels, on='customer_ID', how='inner')\n            self.data.columns = ['customer_ID', 'input_vector', 'extra_pad', 'target']\n        else:\n            self.data.columns = ['customer_ID', 'input_vector', 'extra_pad']\n        \n        print(\"loaded data\")\n        \n        if not self.eval:\n            self.y = self.data['target'].tolist()\n        \n    def __len__(self):\n        return len(self.customer_ids)\n\n    def __getitem__(self, idx):\n        customer_id = self.customer_ids[idx]\n        customer_data = self.data.iloc[idx]\n        \n        x = customer_data['input_vector'] + [list([0]*222) for _ in range(customer_data['extra_pad'])]\n        \n        if self.eval:\n            target = None\n        else:\n            target = int(self.labels[customer_id])\n            \n        return customer_id, torch.tensor(x, dtype=torch.float32), target\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = CustomDataset(data=train_data, labels=labels, extra_pad=train_extra_pad)\ntest_dataset = CustomDataset(data=test_data, labels=labels, extra_pad=test_extra_pad)\n\ntrain_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\ntest_dataloader = torch.utils.data.DataLoader(test_dataset, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, data in enumerate(train_dataloader):\n    customer_id, customer_data, target = data\n    print(customer_id)\n    print(customer_data)\n    print(target)\n    break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class LSTMClassifier(nn.Module):\n    def __init__(self, input_size, hidden_size, num_layers):\n        super(LSTMClassifier, self).__init__()\n        self.hidden_size = hidden_size\n        self.num_layers = num_layers\n        \n        self.lstm = nn.LSTM(input_size, hidden_size, num_layers, batch_first=True)\n        self.fc_1 = nn.Linear(hidden_size, hidden_size)\n        self.fc_2 = nn.Linear(hidden_size, 1)\n        self.dropout = nn.Dropout(0.2)\n        self.sigmoid = nn.Sigmoid()\n\n    def forward(self, x):\n        # Set initial hidden and cell states\n        h0 = torch.zeros(self.num_layers, x.size(0), self.hidden_size).to(x.device)\n        c0 = torch.zeros(self.num_layers, x.size(0), self.hidden_size).to(x.device)\n        \n        # Forward propagate LSTM\n        out, _ = self.lstm(x, (h0, c0))\n        \n        out = self.dropout(out)\n        out = self.fc_1(out[:, -1, :])\n        out = self.fc_2(out)\n        out = self.sigmoid(out)\n        return out\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train(model, dataloader, criterion, optimizer, device):\n    model.train()\n    running_loss = 0\n    total_correct = 0\n    total_samples = 0\n    \n    for i, (customer_id, data, target) in enumerate(dataloader):\n        data = data.float().to(device)\n        target = target.float().to(device)\n        \n        output = model(data)\n        total_correct += (output.squeeze().round() == target).sum().item()\n        \n        optimizer.zero_grad()\n        loss = criterion(output.squeeze(), target)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n        total_samples += data.size(0)\n        \n    running_loss = running_loss / len(dataloader)\n    accuracy = total_correct / total_samples\n    \n    return running_loss, accuracy\n\ndef validation(model, dataloader, criterion, device):\n    model.eval()\n    running_loss = 0\n    total_correct = 0\n    total_samples = 0\n    \n    with torch.no_grad():\n        for i, (customer_id, data, target) in enumerate(dataloader):\n            data = data.float().to(device)\n            target = target.float().to(device)\n            \n            output = model(data)\n            total_correct += (output.squeeze().round() == target).sum()\n            \n            loss = criterion(output.squeeze(), target)\n            running_loss += loss.item()\n            total_samples += data.size(0)\n            \n    running_loss = running_loss / len(dataloader)\n    accuracy = total_correct / total_samples\n    \n    return running_loss, accuracy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage\ninput_size = customer_data.shape[-1]\nhidden_size = 128\nnum_layers = 1\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)\n\nmodel = LSTMClassifier(input_size, hidden_size, num_layers).to(device)\nprint(model)\n\ncriterion = nn.BCELoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# num_epochs = 30\n\n# for epoch in range(num_epochs):\n#     test_loss, test_accuracy = validation(model, test_dataloader, criterion, device)\n#     train_loss, train_accuracy = train(model, train_dataloader, criterion, optimizer, device)\n    \n#     print(f\"Epoch {epoch+1:2d}/{num_epochs} | Train Loss: {train_loss:.4f} | Train Accuracy: {train_accuracy:.4f} | Test Loss: {test_loss:.4f} | Test Accuracy: {test_accuracy:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_data_path = 'dataset/test_data_rounded.parquet'\nsubmission_data = pd.read_parquet(submission_data_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_data, test_data, data_rounded, labels, train_dataset, test_dataset, train_dataloader, test_dataloader\neval_data, eval_extra_pad = preprocessor.preprocess_eval_data(submission_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eval_dataset = CustomDataset(data=eval_data, labels=None, extra_pad=eval_extra_pad, eval=True)\neval_dataloader = torch.utils.data.DataLoader(eval_dataset, batch_size=batch_size, shuffle=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def eval(model, dataloader, device):\n    model.eval()\n    customer_ids = []\n    predictions = []\n    \n    with torch.no_grad():\n        for i, (customer_id, data, _) in enumerate(dataloader):\n            data = data.float().to(device)\n            output = model(data)\n            \n            customer_ids.extend(customer_id)\n            predictions.extend(output.squeeze().round().int().tolist())\n            \n    predictions_df = pd.DataFrame({'customer_ID': customer_ids, 'prediction': predictions})\n    \n    return predictions_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}