{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"このコードはデータの読み込みと前処理、学習、推論、提出を一通りできます。が、突っ込みどころ満載です。  \nThis code can go through from reading data to submission, but it is full of problems.","metadata":{}},{"cell_type":"markdown","source":"致命的な弱点はコード上で * をつけて説明をつけました。  \nFatal weaknesses are marked with an asterisk ( * ) in the code.","metadata":{}},{"cell_type":"code","source":"import copy\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport random\nfrom IPython.display import clear_output\n# from livelossplot import PlotLosses\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\n# import timm\n# from timm.scheduler import CosineLRScheduler\nimport torch\nfrom torch import nn\nimport torchaudio\nimport torchaudio.transforms as T\n# from torchinfo import summary\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import datasets\nfrom torchvision.transforms import ToTensor\nfrom tqdm import tqdm\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:28.101288Z","iopub.execute_input":"2022-08-05T08:13:28.102451Z","iopub.status.idle":"2022-08-05T08:13:29.176499Z","shell.execute_reply.started":"2022-08-05T08:13:28.102355Z","shell.execute_reply":"2022-08-05T08:13:29.175288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device=\"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Using {device} device\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.184275Z","iopub.execute_input":"2022-08-05T08:13:29.185064Z","iopub.status.idle":"2022-08-05T08:13:29.229149Z","shell.execute_reply.started":"2022-08-05T08:13:29.185023Z","shell.execute_reply":"2022-08-05T08:13:29.227576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## データの確認 small EDA","metadata":{}},{"cell_type":"code","source":"data_dir = '../input/dfl-bundesliga-data-shootout/'\ntrain_csv = pd.read_table(data_dir+'train.csv',sep=',')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.231037Z","iopub.execute_input":"2022-08-05T08:13:29.234782Z","iopub.status.idle":"2022-08-05T08:13:29.253399Z","shell.execute_reply.started":"2022-08-05T08:13:29.234749Z","shell.execute_reply":"2022-08-05T08:13:29.252482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_csv.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.257905Z","iopub.execute_input":"2022-08-05T08:13:29.258234Z","iopub.status.idle":"2022-08-05T08:13:29.278567Z","shell.execute_reply.started":"2022-08-05T08:13:29.258204Z","shell.execute_reply":"2022-08-05T08:13:29.277344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv[\"event\"].value_counts(sort=False).plot.bar();","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.281767Z","iopub.execute_input":"2022-08-05T08:13:29.282041Z","iopub.status.idle":"2022-08-05T08:13:29.621669Z","shell.execute_reply.started":"2022-08-05T08:13:29.282015Z","shell.execute_reply":"2022-08-05T08:13:29.620534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv[\"event_attributes\"].value_counts(sort=False).plot.bar();","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.623848Z","iopub.execute_input":"2022-08-05T08:13:29.624578Z","iopub.status.idle":"2022-08-05T08:13:29.868743Z","shell.execute_reply.started":"2022-08-05T08:13:29.624537Z","shell.execute_reply":"2022-08-05T08:13:29.867557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"event_list = train_csv[\"event\"].unique()\nprint(event_list)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.870510Z","iopub.execute_input":"2022-08-05T08:13:29.870910Z","iopub.status.idle":"2022-08-05T08:13:29.877906Z","shell.execute_reply.started":"2022-08-05T08:13:29.870874Z","shell.execute_reply":"2022-08-05T08:13:29.876759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"event_list = train_csv[\"event_attributes\"].unique()\nprint(event_list)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.879393Z","iopub.execute_input":"2022-08-05T08:13:29.880029Z","iopub.status.idle":"2022-08-05T08:13:29.889839Z","shell.execute_reply.started":"2022-08-05T08:13:29.879992Z","shell.execute_reply":"2022-08-05T08:13:29.888668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cap = cv2.VideoCapture(data_dir+'train/3c993bd2_0.mp4')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.891627Z","iopub.execute_input":"2022-08-05T08:13:29.891991Z","iopub.status.idle":"2022-08-05T08:13:29.925850Z","shell.execute_reply.started":"2022-08-05T08:13:29.891956Z","shell.execute_reply":"2022-08-05T08:13:29.924836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir+'train/3c993bd2_0.mp4'","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.927401Z","iopub.execute_input":"2022-08-05T08:13:29.927788Z","iopub.status.idle":"2022-08-05T08:13:29.934548Z","shell.execute_reply.started":"2022-08-05T08:13:29.927752Z","shell.execute_reply":"2022-08-05T08:13:29.933488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\n\nprint(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n\nprint(cap.get(cv2.CAP_PROP_FPS))\n\nprint(cap.get(cv2.CAP_PROP_FRAME_COUNT))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.936250Z","iopub.execute_input":"2022-08-05T08:13:29.936942Z","iopub.status.idle":"2022-08-05T08:13:29.944643Z","shell.execute_reply.started":"2022-08-05T08:13:29.936904Z","shell.execute_reply":"2022-08-05T08:13:29.943651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## データの前処置 Preprocess data","metadata":{}},{"cell_type":"code","source":"train_y_event = []\ntrain_y_event_attributes = []\ntrain_X = []\nvideo_id_tmp = 'tmp'\nmode = 'end'\nlast_frame = 0\ncount_FPS = 1\nfor video_id, time, event, event_attributes in zip(tqdm(train_csv['video_id']), train_csv['time'], train_csv['event'], train_csv['event_attributes']):\n    current_frame = int(time//(1/cap.get(cv2.CAP_PROP_FPS)))\n    if video_id != video_id_tmp: #ビデオが変わったら新しいビデオを読み込む　Load new video when video changes\n#         print(video_id)\n        video_id_tmp = video_id\n        current_frame = 0 #フレーム数も初期化 Initialize the number of frames\n        cap = cv2.VideoCapture(data_dir+'train/'+video_id+'.mp4')\n\n    for i in range(current_frame - last_frame): #現在のコマまで送る Move frame to current position\n        if len(np.array(train_y_event)) > 100: #メモリ爆発をさけるため　*1\n            break\n        ret, frame = cap.read() #frameに1フレーム分の画像が格納されている load image\n\n        if count_FPS == 5: #5枚に1枚を保存するようにしているのでFPSは5になってる FPS become 5 because one is saved every 5\n            count_FPS = 0\n            if mode == 'start':\n                train_X.append(np.array(frame)/256)\n                train_y_event.append(0.0)\n                train_y_event_attributes.append(0.0)\n\n        count_FPS += 1\n\n    if event == 'start' or event == 'end':\n        mode = event\n    elif event == 'challenge' or event == 'throwin' or event == 'play':\n        train_y_event.pop(-1)\n        train_y_event.append(1.0) #本当はタイプごとに数字変える *2\n        train_y_event_attributes.pop(-1)\n        train_y_event_attributes.append(1.0) #同上\n    last_frame = current_frame\nprint(len(train_y_event))\nprint(len(train_y_event_attributes))\nprint(len(train_X))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:29.946351Z","iopub.execute_input":"2022-08-05T08:13:29.947119Z","iopub.status.idle":"2022-08-05T08:13:35.402536Z","shell.execute_reply.started":"2022-08-05T08:13:29.947082Z","shell.execute_reply":"2022-08-05T08:13:35.401458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*1 メモリが爆発するので全てのデータが読み込めません。もっと画像を粗くして保存、もっとFPSを下げる、白黒にするなどの手が考えられますがにしても全部はむりな気がするので、npyか何かで保存して学習中にバッチ毎に読み出す形にしたほうが良い気がしてます。  \nAll data cannot be read because the memory explodes. I can think of ways to make the image rougher, lower the FPS more, make it black and white. However it seems impossible to read all the data in advance, so save it with npy and reading every batch during training might be better.","metadata":{}},{"cell_type":"markdown","source":"*2 本来eventは3種類、event_attributesは14種類あるのでそのクラス分類にしたらいいと思うんですが、とりあえずeventがあるかどうかの二値分類にして最後全てplayにしちゃいました。  \nOriginally, there are 3 types of events and 14 types of event_attributes. Therefore it should be classify them, but this ccode is a binary classification for simplicity.","metadata":{}},{"cell_type":"code","source":"#numpu to torch\ntrain_y_event = torch.from_numpy(np.array(train_y_event))\nprint(train_y_event.shape)\n\ntrain_y_event_attributes = torch.from_numpy(np.array(train_y_event_attributes))\nprint(train_y_event_attributes.shape)\n\ntrain_X = torch.from_numpy(np.array(train_X).transpose(0, 3, 1, 2))\nprint(train_X.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:35.407281Z","iopub.execute_input":"2022-08-05T08:13:35.408368Z","iopub.status.idle":"2022-08-05T08:13:37.234063Z","shell.execute_reply.started":"2022-08-05T08:13:35.408325Z","shell.execute_reply":"2022-08-05T08:13:37.232876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_X = train_X[-20:,:,:] #*3\nval_y_event = train_y_event[-20:]\nval_y_event_attributes = train_y_event_attributes[-20:]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:37.236880Z","iopub.execute_input":"2022-08-05T08:13:37.237278Z","iopub.status.idle":"2022-08-05T08:13:37.242613Z","shell.execute_reply.started":"2022-08-05T08:13:37.237247Z","shell.execute_reply":"2022-08-05T08:13:37.241440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*3 バリデーションデータが学習データの部分集合になってしまっています。  \nValidation data is a subset of training data.","metadata":{}},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"batch_size = 6\n# Create data loaders.\ntrain_dataset = torch.utils.data.TensorDataset(train_X, train_y_event, train_y_event_attributes)\nval_dataset = torch.utils.data.TensorDataset(val_X, val_y_event, val_y_event_attributes)\ntrain_dataloader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=0) # *4\nval_dataloader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:37.244372Z","iopub.execute_input":"2022-08-05T08:13:37.244759Z","iopub.status.idle":"2022-08-05T08:13:37.254672Z","shell.execute_reply.started":"2022-08-05T08:13:37.244722Z","shell.execute_reply":"2022-08-05T08:13:37.253782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*4　時系列データをなんの工夫もなしにシャッフルしたら良くない  \nIt is not good to shuffle time series data without any ingenuity","metadata":{}},{"cell_type":"code","source":"# del train_X, train_y_event, train_y_event_attributes, val_X, val_y_event, val_y_event_attributes","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:37.256325Z","iopub.execute_input":"2022-08-05T08:13:37.257014Z","iopub.status.idle":"2022-08-05T08:13:37.264670Z","shell.execute_reply.started":"2022-08-05T08:13:37.256978Z","shell.execute_reply":"2022-08-05T08:13:37.263630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_output = 1\nn_hidden = 100\n\n\nclass Model(nn.Module): # *5\n    def __init__(self, n_output, n_hidden):\n        super().__init__()\n\n        self.conv1 = nn.Conv2d(3, 16, 5, stride=1, padding=2)\n        self.conv6 = nn.Conv2d(16, 16, 3, stride=1, padding=1)\n\n        self.bn = nn.BatchNorm2d(16, eps=1e-03, momentum=0.01)\n        self.relu = nn.ReLU()\n        self.sigmoid = nn.Sigmoid()\n        self.dropout = nn.Dropout(p=0.3)\n        self.maxpool1 = nn.MaxPool2d((3,2))\n        self.maxpool2 = nn.MaxPool2d((6,4))\n        self.flatten = nn.Flatten()\n        self.l1 = nn.Linear(1382400, n_hidden) \n        self.l2 = nn.Linear(n_hidden, n_output)\n\n        \n    def forward(self, x):\n        x = self.relu(self.bn(self.conv1(x)))\n        x = self.relu(self.bn(self.conv6(x)))\n \n        x = self.dropout(self.maxpool2(x))\n        x = self.flatten(x)\n        # print(x.shape)\n        x = self.relu(self.l1(x))\n        x = self.dropout(x)\n        x = self.sigmoid(self.l2(x))\n        return x\n\nmodel = Model(n_output, n_hidden).to(device)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:37.266259Z","iopub.execute_input":"2022-08-05T08:13:37.266932Z","iopub.status.idle":"2022-08-05T08:13:40.326697Z","shell.execute_reply.started":"2022-08-05T08:13:37.266893Z","shell.execute_reply":"2022-08-05T08:13:40.325648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*5 だめモデルです。LSTMとかがよさそうかも? https://proceedings-of-deim.github.io/DEIM2022/papers/C23-2.pdf  \nIt is a useless model. Maybe LSTM or GRU would be good?","metadata":{}},{"cell_type":"code","source":"# loss_fn = nn.CrossEntropyLoss(reduction=\"sum\")\nloss_fn = nn.BCELoss(reduction=\"sum\")\noptimizer = torch.optim.RAdam(model.parameters(), lr=1e-3)\n# scheduler = CosineLRScheduler(optimizer, t_initial=100, lr_min=1e-6, warmup_t=5, warmup_lr_init=5e-5, warmup_prefix=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:40.328269Z","iopub.execute_input":"2022-08-05T08:13:40.328668Z","iopub.status.idle":"2022-08-05T08:13:40.336149Z","shell.execute_reply.started":"2022-08-05T08:13:40.328612Z","shell.execute_reply":"2022-08-05T08:13:40.333899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(dataloader, model, loss_fn, optimizer, t):\n    size = len(dataloader.dataset)\n    train_loss = 0\n    n_train = 0\n    correct = 0\n    model.train()\n\n    for batch, (X, y, y_attributes) in enumerate(dataloader):\n \n\n        X, y, y_attributes = X.to(device), y.to(device), y_attributes.to(device)\n\n        # Compute prediction error\n        pred = model(X.float())\n        pred = torch.reshape(pred, (-1,))\n#         print(pred,y)\n        loss = loss_fn(pred, y.float())\n\n        # Backpropagation\n        optimizer.zero_grad()\n        loss.backward() \n        optimizer.step()\n#         scheduler.step(t+1)\n        train_loss += loss.item()\n        \n        # _, predicted = torch.max(pred.detach(), 1)\n        # _, y_predicted = torch.max(y.detach(), 1)\n        predicted = torch.round(pred)\n        y_predicted = y\n        correct += (predicted == y_predicted).sum().item()\n        \n        n_train += len(X)\n        if batch % 500 == 0:\n            loss_current, acc_current, current = train_loss/n_train, correct/n_train, batch * len(X)\n            print(f\"Train Epoch: {t+1} loss: {loss_current:>7f}  accuracy: {acc_current:>7f} [{current:>5d}/{size:>5d}]\")\n    \n    loss_current, acc_current = train_loss/n_train, correct/n_train   \n    return loss_current, acc_current","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:40.337571Z","iopub.execute_input":"2022-08-05T08:13:40.338404Z","iopub.status.idle":"2022-08-05T08:13:40.349627Z","shell.execute_reply.started":"2022-08-05T08:13:40.338368Z","shell.execute_reply":"2022-08-05T08:13:40.348474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def val(dataloader, model, loss_fn):\n    size = len(dataloader.dataset)\n    val_loss = 0\n    n_val = 0\n    correct = 0\n    model.eval()\n    with torch.no_grad():        \n        for batch, (X, y, y_attributes) in enumerate(dataloader):\n\n            X, y, y_attributes = X.to(device), y.to(device), y_attributes.to(device)\n            \n            pred = model(X.float())\n            pred = torch.reshape(pred, (-1,))\n            loss = loss_fn(pred, y.float())\n\n            val_loss += loss.item()\n            \n            # _, predicted = torch.max(pred.detach(), 1)\n            # _, y_predicted = torch.max(y.detach(), 1)\n            predicted = torch.round(pred)\n            y_predicted = y\n            correct += (predicted == y_predicted).sum().item()\n            \n            n_val += len(X)\n            if batch % 500 == 0:\n                loss_current, acc_current, current = val_loss/n_val, correct/n_val, batch*len(X)\n                print(f\"Val Epoch: {t+1} loss: {loss_current:>7f}  accuracy: {acc_current:>7f} [{current:>5d}/{size:>5d}]\")\n                \n    loss_current, acc_current = val_loss/n_val, correct/n_val\n    return loss_current, acc_current","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:40.353477Z","iopub.execute_input":"2022-08-05T08:13:40.354346Z","iopub.status.idle":"2022-08-05T08:13:40.363925Z","shell.execute_reply.started":"2022-08-05T08:13:40.354319Z","shell.execute_reply":"2022-08-05T08:13:40.362923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# liveloss = PlotLosses()\n# min_loss = 5\nepoch_num = 3\nfor t in range(epoch_num): \n    logs = {}\n    train_loss, train_acc = train(train_dataloader, model, loss_fn, optimizer, t)\n    val_loss, val_acc = val(val_dataloader, model, loss_fn)\n    logs[\"log loss\"] = train_loss\n    logs[\"val_log loss\"] = val_loss\n    logs[\"acc\"] = train_acc\n    logs[\"val_acc\"] = val_acc\n#     liveloss.update(logs)\n#     liveloss.send()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:13:40.365426Z","iopub.execute_input":"2022-08-05T08:13:40.365929Z","iopub.status.idle":"2022-08-05T08:14:20.715557Z","shell.execute_reply.started":"2022-08-05T08:13:40.365832Z","shell.execute_reply":"2022-08-05T08:14:20.714468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## inference","metadata":{}},{"cell_type":"code","source":"import glob\nfiles = glob.glob(data_dir+'test/*')\nfiles","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:14:20.717107Z","iopub.execute_input":"2022-08-05T08:14:20.717474Z","iopub.status.idle":"2022-08-05T08:14:20.732957Z","shell.execute_reply.started":"2022-08-05T08:14:20.717437Z","shell.execute_reply":"2022-08-05T08:14:20.731878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nmodel.eval()\ntest_submission = []\ntest_X = torch.zeros((1, 3,1080, 1920))\nprint(test_X.size())\nfor file in files:\n    print(file)\n    cap = cv2.VideoCapture(file)\n    count_FPS = 1\n    for i in range((45*2+20)*60*60*25): #試合45分*2と余裕見て20分みてます。 45*2 minutes of match plus 20 minutes of frames with plenty of time\n        ret, frame = cap.read() #frameに1フレーム分の画像が格納されている\n        if not ret: #画像がなくなるとretがfalseになるのでそしたら次の動画に行く　When the image runs out, ret becomes false.\n            break\n        if count_FPS == 25: # every 1s\n            count_FPS = 0\n            test_X[0,:,:,:] = torch.from_numpy((np.array(frame)/256).transpose(2, 0, 1))\n            pred = model(test_X.to(device).float())\n#             if pred.to('cpu').detach().numpy().copy()[0] > 1e-35:\n            df = pd.DataFrame({ # *6\n            \"video_id\": [os.path.basename(file).replace(\".mp4\", \"\")],\n            \"time\": i*0.04,\n            \"event\": \"play\",\n            \"score\": pred.to('cpu').detach().numpy().copy()[0]\n            })\n            test_submission.append(df)\n        count_FPS += 1","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:14:20.734760Z","iopub.execute_input":"2022-08-05T08:14:20.735123Z","iopub.status.idle":"2022-08-05T08:19:22.674200Z","shell.execute_reply.started":"2022-08-05T08:14:20.735087Z","shell.execute_reply":"2022-08-05T08:19:22.672985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*6 ここは壊滅的な問題が恐らく沢山ある。とりあえずモデルの出力をそのまま使用している。torelenceとかここで使うのだろうか。  \nThere are probably a lot of crippling issues here. For the time being, the output of the model is used as it is and the label is set to play. I wonder if torelence is used here.","metadata":{}},{"cell_type":"code","source":"test_submission = pd.concat(test_submission)\ntest_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:19:22.679767Z","iopub.execute_input":"2022-08-05T08:19:22.682279Z","iopub.status.idle":"2022-08-05T08:19:22.952714Z","shell.execute_reply.started":"2022-08-05T08:19:22.682236Z","shell.execute_reply":"2022-08-05T08:19:22.951574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:19:22.954709Z","iopub.execute_input":"2022-08-05T08:19:22.955425Z","iopub.status.idle":"2022-08-05T08:19:22.965150Z","shell.execute_reply.started":"2022-08-05T08:19:22.955386Z","shell.execute_reply":"2022-08-05T08:19:22.964185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thank you for watching until the end.  \nThere may still be serious problems lurking that I am unaware of.","metadata":{}}]}