{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport os \nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-10T20:20:21.330614Z","iopub.execute_input":"2021-07-10T20:20:21.331264Z","iopub.status.idle":"2021-07-10T20:20:22.765053Z","shell.execute_reply.started":"2021-07-10T20:20:21.331159Z","shell.execute_reply":"2021-07-10T20:20:22.763809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import the data","metadata":{}},{"cell_type":"code","source":"PATH = '../input/predict-volcanic-eruptions-ingv-oe/'\n\ntrain_list = os.listdir('../input/predict-volcanic-eruptions-ingv-oe/train')\ntest_list = os.listdir(\"../input/predict-volcanic-eruptions-ingv-oe/test\")\ntrain_time = pd.read_csv(PATH + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:20:56.532660Z","iopub.execute_input":"2021-07-10T20:20:56.533102Z","iopub.status.idle":"2021-07-10T20:20:57.507190Z","shell.execute_reply.started":"2021-07-10T20:20:56.533064Z","shell.execute_reply":"2021-07-10T20:20:57.506068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train and Test size","metadata":{}},{"cell_type":"code","source":"print('Number of train files: {}'.format(len(train_list)))\nprint('Number of test files: {}'.format(len(test_list )))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:00.604103Z","iopub.execute_input":"2021-07-10T20:21:00.604537Z","iopub.status.idle":"2021-07-10T20:21:00.612426Z","shell.execute_reply.started":"2021-07-10T20:21:00.604503Z","shell.execute_reply":"2021-07-10T20:21:00.610988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = pd.read_csv(PATH + 'train/' + train_list[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:02.478157Z","iopub.execute_input":"2021-07-10T20:21:02.478530Z","iopub.status.idle":"2021-07-10T20:21:02.626108Z","shell.execute_reply.started":"2021-07-10T20:21:02.478499Z","shell.execute_reply":"2021-07-10T20:21:02.624899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can transform our signals in 1 row","metadata":{}},{"cell_type":"code","source":"example[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:03.058287Z","iopub.execute_input":"2021-07-10T20:21:03.058726Z","iopub.status.idle":"2021-07-10T20:21:03.099829Z","shell.execute_reply.started":"2021-07-10T20:21:03.058690Z","shell.execute_reply":"2021-07-10T20:21:03.098480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_test = pd.read_csv(PATH + 'test/' + test_list[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:03.498163Z","iopub.execute_input":"2021-07-10T20:21:03.498789Z","iopub.status.idle":"2021-07-10T20:21:03.649145Z","shell.execute_reply.started":"2021-07-10T20:21:03.498748Z","shell.execute_reply":"2021-07-10T20:21:03.647912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_test[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:04.392411Z","iopub.execute_input":"2021-07-10T20:21:04.392873Z","iopub.status.idle":"2021-07-10T20:21:04.417404Z","shell.execute_reply.started":"2021-07-10T20:21:04.392833Z","shell.execute_reply":"2021-07-10T20:21:04.416091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:04.853459Z","iopub.execute_input":"2021-07-10T20:21:04.853842Z","iopub.status.idle":"2021-07-10T20:21:04.861567Z","shell.execute_reply.started":"2021-07-10T20:21:04.853811Z","shell.execute_reply":"2021-07-10T20:21:04.860508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_time","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:06.616821Z","iopub.execute_input":"2021-07-10T20:21:06.617170Z","iopub.status.idle":"2021-07-10T20:21:06.631350Z","shell.execute_reply.started":"2021-07-10T20:21:06.617143Z","shell.execute_reply":"2021-07-10T20:21:06.630139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at one of the train signal","metadata":{"_kg_hide-output":true}},{"cell_type":"code","source":"example.plot(figsize=(15,15), subplots=True);","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:07.964488Z","iopub.execute_input":"2021-07-10T20:21:07.964903Z","iopub.status.idle":"2021-07-10T20:21:11.581099Z","shell.execute_reply.started":"2021-07-10T20:21:07.964867Z","shell.execute_reply":"2021-07-10T20:21:11.580173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_time[train_time.segment_id == int(train_list[0].split('.')[0])]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:11.582404Z","iopub.execute_input":"2021-07-10T20:21:11.582788Z","iopub.status.idle":"2021-07-10T20:21:11.613403Z","shell.execute_reply.started":"2021-07-10T20:21:11.582714Z","shell.execute_reply":"2021-07-10T20:21:11.612102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(example.fillna(0).describe().iloc[1:, :].unstack()).reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:11.615420Z","iopub.execute_input":"2021-07-10T20:21:11.615789Z","iopub.status.idle":"2021-07-10T20:21:11.691840Z","shell.execute_reply.started":"2021-07-10T20:21:11.615751Z","shell.execute_reply":"2021-07-10T20:21:11.690422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process = pd.DataFrame(example.fillna(0).describe().iloc[1:, :].unstack()).reset_index()\nprocess = process.rename(columns={0: 'value'})\nprocess['feature'] = process['level_0'] + '_' + process['level_1']","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:11.693367Z","iopub.execute_input":"2021-07-10T20:21:11.693676Z","iopub.status.idle":"2021-07-10T20:21:11.751102Z","shell.execute_reply.started":"2021-07-10T20:21:11.693647Z","shell.execute_reply":"2021-07-10T20:21:11.749899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:12.646306Z","iopub.execute_input":"2021-07-10T20:21:12.646657Z","iopub.status.idle":"2021-07-10T20:21:12.667207Z","shell.execute_reply.started":"2021-07-10T20:21:12.646627Z","shell.execute_reply":"2021-07-10T20:21:12.665902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process = process.drop(['level_0', 'level_1'], axis=1).set_index('feature').T","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:13.635607Z","iopub.execute_input":"2021-07-10T20:21:13.636132Z","iopub.status.idle":"2021-07-10T20:21:13.644879Z","shell.execute_reply.started":"2021-07-10T20:21:13.636089Z","shell.execute_reply":"2021-07-10T20:21:13.643636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:14.903508Z","iopub.execute_input":"2021-07-10T20:21:14.903873Z","iopub.status.idle":"2021-07-10T20:21:14.930765Z","shell.execute_reply.started":"2021-07-10T20:21:14.903840Z","shell.execute_reply":"2021-07-10T20:21:14.929702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process['time'] = train_time[train_time.segment_id == int(train_list[0].split('.')[0])].time_to_eruption.values[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:16.100085Z","iopub.execute_input":"2021-07-10T20:21:16.100547Z","iopub.status.idle":"2021-07-10T20:21:16.108369Z","shell.execute_reply.started":"2021-07-10T20:21:16.100504Z","shell.execute_reply":"2021-07-10T20:21:16.107330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:18.156454Z","iopub.execute_input":"2021-07-10T20:21:18.156887Z","iopub.status.idle":"2021-07-10T20:21:18.184032Z","shell.execute_reply.started":"2021-07-10T20:21:18.156851Z","shell.execute_reply":"2021-07-10T20:21:18.183118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(example.fillna(0).skew()).T","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:21:18.482971Z","iopub.execute_input":"2021-07-10T20:21:18.483570Z","iopub.status.idle":"2021-07-10T20:21:18.526288Z","shell.execute_reply.started":"2021-07-10T20:21:18.483532Z","shell.execute_reply":"2021-07-10T20:21:18.525161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing Train and Test","metadata":{}},{"cell_type":"markdown","source":"Create a function for data preparation","metadata":{}},{"cell_type":"code","source":"def create_frame(data, data_time=None, type_data='train'):\n    data = data.fillna(0)\n    \n    # основные статистика\n    data_transform = data.describe().iloc[1:, :]\n    \n    # Дополнительные параметры\n    # Коэффициент асимметрии\n    data_transform.loc['skew'] = data.skew().tolist()\n    \n    #Среднее абсолютное отклонение\n    data_transform.loc['mad'] = data.mad().tolist()\n    \n    # Коэффициент эксцесса — мера остроты пика распределения случайной величины.\n    data_transform.loc['kurtosis'] = data.kurtosis().tolist()\n    \n    # добавление квантилей\n    for i in range(0, 100, 5):\n        if ((i!=25) & (i!=50)):\n                str_col = f\"{i}%\"\n                int_col = float(i)/100\n                data_transform.loc[str_col] = data_transform.quantile(int_col).tolist()\n        else:\n            continue\n            \n    data_transform = pd.DataFrame(data_transform.unstack()).reset_index()\n    data_transform = data_transform.rename(columns={0: 'value'})\n    data_transform['feature'] = data_transform['level_0'] + '_' + data_transform['level_1']\n    data_transform = data_transform.drop(['level_0', 'level_1'], axis=1).set_index('feature').T\n    \n    if type_data=='train':\n        data_transform['time'] = data_time\n    return data_transform","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:22:12.870596Z","iopub.execute_input":"2021-07-10T20:22:12.871052Z","iopub.status.idle":"2021-07-10T20:22:12.881904Z","shell.execute_reply.started":"2021-07-10T20:22:12.871015Z","shell.execute_reply":"2021-07-10T20:22:12.880856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train = pd.DataFrame()\n\nfor file in tqdm(train_list):\n    df = pd.read_csv(PATH + 'train/' + file)\n    data_time = train_time[train_time.segment_id == int(file.split('.')[0])].time_to_eruption.values[0]\n    df = create_frame(df, data_time, type_data='train')\n    all_train = all_train.append(df)\n\nall_train = all_train.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:22:16.018987Z","iopub.execute_input":"2021-07-10T20:22:16.019388Z","iopub.status.idle":"2021-07-10T20:44:40.278349Z","shell.execute_reply.started":"2021-07-10T20:22:16.019352Z","shell.execute_reply":"2021-07-10T20:44:40.276704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_test = pd.DataFrame()\n\nfor file in tqdm(test_list):\n    df = pd.read_csv(PATH + 'test/' + file)\n    df = create_frame(df, data_time=None, type_data='test')\n    all_test = all_test.append(df)\n\nall_test = all_test.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T20:44:40.281268Z","iopub.execute_input":"2021-07-10T20:44:40.281624Z","iopub.status.idle":"2021-07-10T21:06:15.905479Z","shell.execute_reply.started":"2021-07-10T20:44:40.281582Z","shell.execute_reply":"2021-07-10T21:06:15.903774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:06:15.910201Z","iopub.execute_input":"2021-07-10T21:06:15.910603Z","iopub.status.idle":"2021-07-10T21:06:15.964791Z","shell.execute_reply.started":"2021-07-10T21:06:15.910564Z","shell.execute_reply":"2021-07-10T21:06:15.963410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_test[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:06:15.966763Z","iopub.execute_input":"2021-07-10T21:06:15.967105Z","iopub.status.idle":"2021-07-10T21:06:16.007977Z","shell.execute_reply.started":"2021-07-10T21:06:15.967070Z","shell.execute_reply":"2021-07-10T21:06:16.006846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"X = all_train.drop('time',axis=1)\ny = all_train['time']\n\ntest = all_test.copy()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:07:02.079541Z","iopub.execute_input":"2021-07-10T21:07:02.080145Z","iopub.status.idle":"2021-07-10T21:07:02.120876Z","shell.execute_reply.started":"2021-07-10T21:07:02.080092Z","shell.execute_reply":"2021-07-10T21:07:02.119714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Baseline","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.25, shuffle=True, random_state=10)\n\nX_train, X_val, y_train, y_val = train_test_split(\n    X_train, y_train, test_size=0.25, shuffle=True, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:07:08.545272Z","iopub.execute_input":"2021-07-10T21:07:08.545714Z","iopub.status.idle":"2021-07-10T21:07:08.574464Z","shell.execute_reply.started":"2021-07-10T21:07:08.545670Z","shell.execute_reply":"2021-07-10T21:07:08.573150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mape(y_true, y_pred):\n    return np.mean(np.abs((y_pred-y_true)/y_true))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:07:09.833467Z","iopub.execute_input":"2021-07-10T21:07:09.833931Z","iopub.status.idle":"2021-07-10T21:07:09.839787Z","shell.execute_reply.started":"2021-07-10T21:07:09.833891Z","shell.execute_reply":"2021-07-10T21:07:09.838102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = CatBoostRegressor(loss_function='MAPE')  \ntrain_dataset = Pool(data=X_train,\n                     label=y_train,\n                     )\n    \neval_dataset = Pool(data=X_val,\n                    label=y_val,\n                    )\n    \nclf.fit(train_dataset,\n          use_best_model=True,\n          verbose = 0,\n          eval_set=eval_dataset)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:07:15.554106Z","iopub.execute_input":"2021-07-10T21:07:15.554458Z","iopub.status.idle":"2021-07-10T21:08:05.497505Z","shell.execute_reply.started":"2021-07-10T21:07:15.554428Z","shell.execute_reply":"2021-07-10T21:08:05.496358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict(Pool(data=X_test))\n    \nprint(f\"MAPE: {mape(y_test, y_pred)}\")\nprint(f\"MAE: {mean_absolute_error(y_test, y_pred)}\")\nprint(f\"RMSE: {np.sqrt(mean_squared_error(y_test, y_pred))}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:08:05.499212Z","iopub.execute_input":"2021-07-10T21:08:05.499512Z","iopub.status.idle":"2021-07-10T21:08:05.534609Z","shell.execute_reply.started":"2021-07-10T21:08:05.499483Z","shell.execute_reply":"2021-07-10T21:08:05.533230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Using KFold with some parametrs","metadata":{}},{"cell_type":"markdown","source":"We are going to use KFold with CatBoostRegressor. We didn't use GridSearch because save time","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.25, shuffle=True, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T21:08:49.672216Z","iopub.execute_input":"2021-07-10T21:08:49.672590Z","iopub.status.idle":"2021-07-10T21:08:49.692359Z","shell.execute_reply.started":"2021-07-10T21:08:49.672559Z","shell.execute_reply":"2021-07-10T21:08:49.690956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_fold = 5\ncv = KFold(n_splits=n_fold, shuffle=True, random_state=10)\nprediction = np.zeros(len(test))\nmape_, mae, rmse = [], [], []\n\nparams = {\n            'iterations':1000,\n            'learning_rate':0.1,\n            'depth':6,\n            'eval_metric':'RMSE'\n}\n\nfor fold, (train_index, val_index) in enumerate(cv.split(X)):\n    X_train = X.iloc[train_index,:]\n    X_val = X.iloc[val_index,:]\n\n    y_train = y.iloc[train_index]\n    y_val = y.iloc[val_index]\n          \n    clf = CatBoostRegressor(**params)  \n    \n    train_dataset = Pool(data=X_train,\n                     label=y_train,\n                     )\n    \n    eval_dataset = Pool(data=X_val,\n                    label=y_val,\n                    )\n    \n    clf.fit(train_dataset,\n              use_best_model=True,\n              verbose = 0,\n              eval_set=eval_dataset)\n   \n    y_pred = clf.predict(Pool(data=X_test))\n    \n    mape_.append(mape(y_test, y_pred))\n    mae.append(mean_absolute_error(y_test, y_pred))\n    rmse.append(np.sqrt(mean_squared_error(y_test, y_pred)))\n\n    print(f\"fold: {fold}, MAPE: {mape(y_test, y_pred)}\")\n    print(f\"fold: {fold}, MAE: {mean_absolute_error(y_test, y_pred)}\")\n    print(f\"fold: {fold}, RMSE: {np.sqrt(mean_squared_error(y_test, y_pred))}\")\n\n    # test array predictions\n    prediction += clf.predict(Pool(data=test))\n        \nprediction /= n_fold\n\nprint('CV mean MAPE:  {0:.4f}, std: {1:.4f}.'.format(np.mean(mape_), np.std(mape_)))\nprint('CV mean MAE: {0:.4f}, std: {1:.4f}.'.format(np.mean(mae), np.std(mae)))\nprint('CV mean RMSE: {0:.4f}, std: {1:.4f}.'.format(np.mean(rmse), np.std(rmse)))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T22:18:56.830139Z","iopub.execute_input":"2021-07-10T22:18:56.830550Z","iopub.status.idle":"2021-07-10T22:22:39.595224Z","shell.execute_reply.started":"2021-07-10T22:18:56.830514Z","shell.execute_reply":"2021-07-10T22:22:39.593528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://catboost.ai/docs/concepts/python-usages-examples.html","metadata":{}},{"cell_type":"code","source":"sub_example = pd.read_csv(PATH + 'sample_submission.csv')\nsub_example[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T22:22:47.945682Z","iopub.execute_input":"2021-07-10T22:22:47.946172Z","iopub.status.idle":"2021-07-10T22:22:47.969893Z","shell.execute_reply.started":"2021-07-10T22:22:47.946133Z","shell.execute_reply":"2021-07-10T22:22:47.968420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_index = [int(i.split('.')[0]) for i in test_list]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T22:22:51.211470Z","iopub.execute_input":"2021-07-10T22:22:51.211872Z","iopub.status.idle":"2021-07-10T22:22:51.220358Z","shell.execute_reply.started":"2021-07-10T22:22:51.211834Z","shell.execute_reply":"2021-07-10T22:22:51.218923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_index[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T22:22:51.671635Z","iopub.execute_input":"2021-07-10T22:22:51.672102Z","iopub.status.idle":"2021-07-10T22:22:51.679527Z","shell.execute_reply.started":"2021-07-10T22:22:51.672064Z","shell.execute_reply":"2021-07-10T22:22:51.678420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['segment_id'] = test_index\nsubmission['time_to_eruption'] = prediction\nsubmission.to_csv('submission.csv', header=True, index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T22:22:53.884310Z","iopub.execute_input":"2021-07-10T22:22:53.884783Z","iopub.status.idle":"2021-07-10T22:22:53.927073Z","shell.execute_reply.started":"2021-07-10T22:22:53.884721Z","shell.execute_reply":"2021-07-10T22:22:53.925802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T22:22:55.610078Z","iopub.execute_input":"2021-07-10T22:22:55.610472Z","iopub.status.idle":"2021-07-10T22:22:55.622932Z","shell.execute_reply.started":"2021-07-10T22:22:55.610439Z","shell.execute_reply":"2021-07-10T22:22:55.621712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}