{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-05T00:10:39.025037Z","iopub.execute_input":"2022-10-05T00:10:39.026904Z","iopub.status.idle":"2022-10-05T00:10:39.064379Z","shell.execute_reply.started":"2022-10-05T00:10:39.026792Z","shell.execute_reply":"2022-10-05T00:10:39.063331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Import needed libraries**","metadata":{}},{"cell_type":"code","source":"# we will be working with compressed feather files, thanks to \"ŞAFAK TÜRKELI\"\n# !pip install dask_ml\n\nimport gc\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom xgboost import XGBClassifier\n# from lgbm import LGBMClassifier\nfrom sklearn.metrics import roc_auc_score, f1_score\nfrom sklearn.model_selection import StratifiedKFold, cross_validate\n\n# import dask.dataframe as dd\n# import dask_ml","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:22:01.526200Z","iopub.execute_input":"2022-10-05T01:22:01.526609Z","iopub.status.idle":"2022-10-05T01:22:02.788516Z","shell.execute_reply.started":"2022-10-05T01:22:01.526525Z","shell.execute_reply":"2022-10-05T01:22:02.787291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load the data**","metadata":{}},{"cell_type":"code","source":"# helper function to help optimize our memory by converting numerical col dtypes to min dtype needed for work\ndef optimize_memory_usage(df, show_result = False):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            \n            if str(col_type)[:3] == \"int\": \n                # if col type is int\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else :\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype(\"category\")\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    \n    if(show_result):\n        print(\"Memory usage was optimized from {:5.2f} Mb to {:5.2f}\".format(start_mem, end_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:22:06.676287Z","iopub.execute_input":"2022-10-05T01:22:06.676847Z","iopub.status.idle":"2022-10-05T01:22:06.711287Z","shell.execute_reply.started":"2022-10-05T01:22:06.676806Z","shell.execute_reply":"2022-10-05T01:22:06.710070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = optimize_memory_usage(pd.read_feather(\"../input/tpsoct22-feather-files/train_0.feather\").sample(frac = 0.21), True)\nfor i in range(1, 10):\n    train = pd.concat([train, optimize_memory_usage(pd.read_feather(f\"../input/tpsoct22-feather-files/train_{i}.feather\").sample(frac = 0.21), True)])\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:22:09.165401Z","iopub.execute_input":"2022-10-05T01:22:09.165780Z","iopub.status.idle":"2022-10-05T01:24:16.539256Z","shell.execute_reply.started":"2022-10-05T01:22:09.165733Z","shell.execute_reply":"2022-10-05T01:24:16.538148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntrain.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:16.541594Z","iopub.execute_input":"2022-10-05T01:24:16.541996Z","iopub.status.idle":"2022-10-05T01:24:16.686710Z","shell.execute_reply.started":"2022-10-05T01:24:16.541955Z","shell.execute_reply":"2022-10-05T01:24:16.685566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Done Loading Data now time for Quick EDA(could later load just a small portion of data and do proper EDA)**","metadata":{}},{"cell_type":"markdown","source":"from data we saw that `game_num`, `event_id`, `event_time`, `player_scoring_next`, `team_scoring_next` where just in train data and not in test.\nalso this are not part of our label, so drop them and help save us space\n","metadata":{}},{"cell_type":"code","source":"train = train.drop(columns = [\"game_num\", 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next'])\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:16.688244Z","iopub.execute_input":"2022-10-05T01:24:16.690067Z","iopub.status.idle":"2022-10-05T01:24:17.963342Z","shell.execute_reply.started":"2022-10-05T01:24:16.690018Z","shell.execute_reply":"2022-10-05T01:24:17.962252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:17.966284Z","iopub.execute_input":"2022-10-05T01:24:17.967202Z","iopub.status.idle":"2022-10-05T01:24:17.997962Z","shell.execute_reply.started":"2022-10-05T01:24:17.967158Z","shell.execute_reply":"2022-10-05T01:24:17.996774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find is any nan rows(will be replaced later)\n# all p columns are nan when player is demolished or will respawn soon\n# train_df.isnull().sum().compute()\n\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:17.999593Z","iopub.execute_input":"2022-10-05T01:24:18.000283Z","iopub.status.idle":"2022-10-05T01:24:19.153600Z","shell.execute_reply.started":"2022-10-05T01:24:18.000238Z","shell.execute_reply":"2022-10-05T01:24:19.152547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to fillNan values with 0 since player is dead or will soon respawn\ndef fill_null(df):\n#     cols = [\n#     \"p0_pos_x\",\"p1_pos_x\",\"p2_pos_x\",\"p3_pos_x\",\"p4_pos_x\",\"p5_pos_x\",\n#     \"p0_pos_y\",\"p1_pos_y\",\"p2_pos_y\",\"p3_pos_y\",\"p4_pos_y\",\"p5_pos_y\",\n#     \"p0_pos_z\",\"p1_pos_z\",\"p2_pos_z\",\"p3_pos_z\",\"p4_pos_z\",\"p5_pos_z\",\n    \n#     \"p0_vel_x\",\"p1_vel_x\",\"p2_vel_x\",\"p3_vel_x\",\"p4_vel_x\",\"p5_vel_x\",\n#     \"p0_vel_y\",\"p1_vel_y\",\"p2_vel_y\",\"p3_vel_y\",\"p4_vel_y\",\"p5_vel_y\",\n#     \"p0_vel_z\",\"p1_vel_z\",\"p2_vel_z\",\"p3_vel_z\",\"p4_vel_z\",\"p5_vel_z\",\n    \n#     'p0_boost','p1_boost','p2_boost','p3_boost','p4_boost','p5_boost'\n#     ]\n    \n#     for col in cols:\n#         df[col] = df[col].fillna(0)\n    \n    # rather than filling nan rows with constant value of 0\n    # lets drop it as it helps to also save us memory\n    df = df.dropna()\n    \n#     del cols\n    gc.collect()\n   \n    return df\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-04T23:23:59.707494Z","iopub.execute_input":"2022-10-04T23:23:59.708476Z","iopub.status.idle":"2022-10-04T23:23:59.857885Z","shell.execute_reply.started":"2022-10-04T23:23:59.708421Z","shell.execute_reply":"2022-10-04T23:23:59.856637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:19.158600Z","iopub.execute_input":"2022-10-05T01:24:19.161280Z","iopub.status.idle":"2022-10-05T01:24:22.383820Z","shell.execute_reply.started":"2022-10-05T01:24:19.161233Z","shell.execute_reply":"2022-10-05T01:24:22.382748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:22.385517Z","iopub.execute_input":"2022-10-05T01:24:22.385982Z","iopub.status.idle":"2022-10-05T01:24:23.194561Z","shell.execute_reply.started":"2022-10-05T01:24:22.385939Z","shell.execute_reply":"2022-10-05T01:24:23.193548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:23.196301Z","iopub.execute_input":"2022-10-05T01:24:23.197022Z","iopub.status.idle":"2022-10-05T01:24:23.326549Z","shell.execute_reply.started":"2022-10-05T01:24:23.196971Z","shell.execute_reply":"2022-10-05T01:24:23.325089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# seems to have a normal distribution, same thing for vel_y and vel_z\ntrain[\"ball_vel_x\"].sample(frac = 0.0005).plot(kind=\"kde\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T23:12:52.100106Z","iopub.execute_input":"2022-10-04T23:12:52.100524Z","iopub.status.idle":"2022-10-04T23:12:52.811316Z","shell.execute_reply.started":"2022-10-04T23:12:52.100489Z","shell.execute_reply":"2022-10-04T23:12:52.810391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# similar distribution for all boost_timer cols\ntrain[\"boost1_timer\"].sample(frac = 0.0005).plot(kind=\"kde\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T23:12:52.813721Z","iopub.execute_input":"2022-10-04T23:12:52.814528Z","iopub.status.idle":"2022-10-04T23:12:53.459317Z","shell.execute_reply.started":"2022-10-04T23:12:52.814486Z","shell.execute_reply":"2022-10-04T23:12:53.458294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# seen that chance of scoring is less than chance of not scoring. which is quite normal\nsns.countplot(train[\"team_A_scoring_within_10sec\"].sample(frac = 0.0005))","metadata":{"execution":{"iopub.status.busy":"2022-10-04T22:30:36.472463Z","iopub.execute_input":"2022-10-04T22:30:36.472863Z","iopub.status.idle":"2022-10-04T22:30:37.158410Z","shell.execute_reply.started":"2022-10-04T22:30:36.472832Z","shell.execute_reply":"2022-10-04T22:30:37.157121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# seen that chance of scoring is less than chance of not scoring. which is quite normal\nsns.countplot(train[\"team_B_scoring_within_10sec\"].sample(frac = 0.0005))","metadata":{"execution":{"iopub.status.busy":"2022-10-04T22:32:41.920796Z","iopub.execute_input":"2022-10-04T22:32:41.921238Z","iopub.status.idle":"2022-10-04T22:32:42.607844Z","shell.execute_reply.started":"2022-10-04T22:32:41.921203Z","shell.execute_reply":"2022-10-04T22:32:42.606452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10, 10))\nsns.heatmap(train.sample(frac = 0.0005).corr(), cmap =\"YlGnBu\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T22:34:44.843671Z","iopub.execute_input":"2022-10-04T22:34:44.844123Z","iopub.status.idle":"2022-10-04T22:34:46.209281Z","shell.execute_reply.started":"2022-10-04T22:34:44.844087Z","shell.execute_reply":"2022-10-04T22:34:46.208116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T00:12:37.836282Z","iopub.execute_input":"2022-10-05T00:12:37.836750Z","iopub.status.idle":"2022-10-05T00:12:37.874246Z","shell.execute_reply.started":"2022-10-05T00:12:37.836710Z","shell.execute_reply":"2022-10-05T00:12:37.873321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"# function to get ball distance from players location    \ndef get_new_features(df):\n    for i in range(6):\n        # done this way to get absolute values\n        df[f\"p{i}_ball_distance\"] = ((df[\"ball_pos_x\"] - df[f\"p{i}_pos_x\"]) **2 + \\\n        (df[\"ball_pos_y\"] - df[f\"p{i}_pos_y\"]) **2 + \\\n        (df[\"ball_pos_z\"] - df[f\"p{i}_pos_z\"]) **2) ** 0.5\n        \n        # if player in this x and y cordinates which is pos of BigBoostOrb\n        # and boost timer is 0 then set okayer boost to 100\n        x_cord_condition = [-61.4, 61.4, -71.7, 71.7, -61.4, 61.4]\n        y_cord_condition = [-81.9, -81.9, 0, 0, 81.9, 81.9]\n        boost_timer_condition = 0\n        \n        # pandas way of doing dask code below\n        df.loc[(df[f\"p{i}_pos_x\"].isin(x_cord_condition)) & (df[f\"p{i}_pos_y\"].isin(y_cord_condition)) \\\n              & (df[f\"boost{i}_timer\"] == boost_timer_condition), f\"p{i}_boost\"] = 100\n\n        # dask way of doing code above\n#         df[f\"p{i}_boost\"] = df[f\"p{i}_boost\"].mask((df[f\"p{i}_pos_x\"].isin(x_cord_condition)) & (df[f\"p{i}_pos_y\"].isin(y_cord_condition)) \\\n#               & (df[f\"boost{i}_timer\"] == boost_timer_condition), 100)\n        \n        \n        del x_cord_condition\n        del y_cord_condition\n        del boost_timer_condition\n        gc.collect()\n    \n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:23.328855Z","iopub.execute_input":"2022-10-05T01:24:23.329629Z","iopub.status.idle":"2022-10-05T01:24:23.341819Z","shell.execute_reply.started":"2022-10-05T01:24:23.329587Z","shell.execute_reply":"2022-10-05T01:24:23.340251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = get_new_features(train)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:23.346915Z","iopub.execute_input":"2022-10-05T01:24:23.347602Z","iopub.status.idle":"2022-10-05T01:24:29.800427Z","shell.execute_reply.started":"2022-10-05T01:24:23.347553Z","shell.execute_reply":"2022-10-05T01:24:29.799099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:29.802359Z","iopub.execute_input":"2022-10-05T01:24:29.802801Z","iopub.status.idle":"2022-10-05T01:24:30.015158Z","shell.execute_reply.started":"2022-10-05T01:24:29.802742Z","shell.execute_reply":"2022-10-05T01:24:30.013953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Training and Making Predictions Simultaneously, due to low ram**\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T12:58:42.719456Z","iopub.execute_input":"2022-10-03T12:58:42.720338Z","iopub.status.idle":"2022-10-03T12:58:43.127439Z","shell.execute_reply.started":"2022-10-03T12:58:42.720262Z","shell.execute_reply":"2022-10-03T12:58:43.125481Z"}}},{"cell_type":"code","source":"gc.collect()\nyA = train['team_A_scoring_within_10sec']\nyB = train['team_B_scoring_within_10sec']\n\nx = train.drop(columns = [\"team_A_scoring_within_10sec\", \"team_B_scoring_within_10sec\"])\n\nx_test =  get_new_features(optimize_memory_usage(pd.read_csv(\"../input/tabular-playground-series-oct-2022/test.csv\")))\nx_test.fillna(0)\ntest_id = x_test[\"id\"]\nx_test.drop(columns = [\"id\"], inplace =True)\ndel train\ngc.collect()\n\nlen(x.columns), len(x_test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:24:30.016964Z","iopub.execute_input":"2022-10-05T01:24:30.017397Z","iopub.status.idle":"2022-10-05T01:24:45.534073Z","shell.execute_reply.started":"2022-10-05T01:24:30.017356Z","shell.execute_reply":"2022-10-05T01:24:45.533003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(x_test.index)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:20:03.715219Z","iopub.execute_input":"2022-10-05T01:20:03.715580Z","iopub.status.idle":"2022-10-05T01:20:03.721995Z","shell.execute_reply.started":"2022-10-05T01:20:03.715550Z","shell.execute_reply":"2022-10-05T01:20:03.720969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'tree_method': 'gpu_hist',\n          'n_estimators': 240,\n          'max_depth': 8,\n          'learning_rate': 0.1,\n          'objective': 'binary:logistic',\n         }\n\n# used to train model for modelA and modelB\ndef train_model(x, y, x_test):\n    gc.collect()\n    roc_scores = []\n    test_predictions_proba = []\n    cv = StratifiedKFold(n_splits = 5, shuffle = True)\n    \n    for fold, (train_idx, val_idx) in enumerate(cv.split(x, y)):\n        x_train, x_val = x.iloc[train_idx], x.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        model = XGBClassifier(**params)\n        model.fit(x_train, y_train)\n        predictions_prob = model.predict_proba(x_val)[:, 1]\n        score = roc_auc_score(y_val, predictions_prob)\n        roc_scores.append(score)\n        test_predictions_proba.append(model.predict_proba(x_test)[:, 1])\n        print(f\"Fold {fold + 1} \\t\\t AUC: {score}\")\n        \n        del model\n        del x_train\n        del x_val\n        del y_train\n        del y_val\n        gc.collect()\n    \n    print(\"Overall roc_auc_score: \", np.mean(roc_scores))\n    return test_predictions_proba\n        ","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:37:57.090239Z","iopub.execute_input":"2022-10-05T01:37:57.090615Z","iopub.status.idle":"2022-10-05T01:37:57.100936Z","shell.execute_reply.started":"2022-10-05T01:37:57.090584Z","shell.execute_reply":"2022-10-05T01:37:57.099817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_A = train_model(x, yA, x_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:38:00.220179Z","iopub.execute_input":"2022-10-05T01:38:00.220580Z","iopub.status.idle":"2022-10-05T01:41:13.737713Z","shell.execute_reply.started":"2022-10-05T01:38:00.220545Z","shell.execute_reply":"2022-10-05T01:41:13.736524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\npredictions_B = train_model(x, yB, x_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:41:20.789833Z","iopub.execute_input":"2022-10-05T01:41:20.790225Z","iopub.status.idle":"2022-10-05T01:44:30.821776Z","shell.execute_reply.started":"2022-10-05T01:41:20.790193Z","shell.execute_reply":"2022-10-05T01:44:30.819392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# viz one prediction result\n# each element in predictions A match to model predictions when trained on fold 1\npredictions_A[0]","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:44:37.179387Z","iopub.execute_input":"2022-10-05T01:44:37.179753Z","iopub.status.idle":"2022-10-05T01:44:37.189915Z","shell.execute_reply.started":"2022-10-05T01:44:37.179723Z","shell.execute_reply":"2022-10-05T01:44:37.187522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_mean_A = np.mean(predictions_A, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:44:39.929259Z","iopub.execute_input":"2022-10-05T01:44:39.929628Z","iopub.status.idle":"2022-10-05T01:44:39.941340Z","shell.execute_reply.started":"2022-10-05T01:44:39.929597Z","shell.execute_reply":"2022-10-05T01:44:39.940139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_mean_B = np.mean(predictions_B, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:44:41.889055Z","iopub.execute_input":"2022-10-05T01:44:41.889404Z","iopub.status.idle":"2022-10-05T01:44:41.901134Z","shell.execute_reply.started":"2022-10-05T01:44:41.889376Z","shell.execute_reply":"2022-10-05T01:44:41.899881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"id\": test_id,\n    \"team_A_scoring_within_10sec\": predictions_mean_A,\n    \"team_B_scoring_within_10sec\": predictions_mean_B})","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:44:43.739215Z","iopub.execute_input":"2022-10-05T01:44:43.739579Z","iopub.status.idle":"2022-10-05T01:44:43.747435Z","shell.execute_reply.started":"2022-10-05T01:44:43.739550Z","shell.execute_reply":"2022-10-05T01:44:43.746154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission_on_20%_sample_of_whole_data_tweak.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:46:49.631904Z","iopub.execute_input":"2022-10-05T01:46:49.632307Z","iopub.status.idle":"2022-10-05T01:46:51.911181Z","shell.execute_reply.started":"2022-10-05T01:46:49.632257Z","shell.execute_reply":"2022-10-05T01:46:51.910001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del predictions_mean_A\ndel predictions_mean_B\ndel submission\ndel predictions_A\ndel predictions_B","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:50:59.119573Z","iopub.execute_input":"2022-10-05T01:50:59.119958Z","iopub.status.idle":"2022-10-05T01:50:59.125126Z","shell.execute_reply.started":"2022-10-05T01:50:59.119928Z","shell.execute_reply":"2022-10-05T01:50:59.124038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Come back later to find out if there's any other feature combination**","metadata":{"execution":{"iopub.status.busy":"2022-10-05T01:51:14.613981Z","iopub.execute_input":"2022-10-05T01:51:14.614377Z","iopub.status.idle":"2022-10-05T01:51:14.619355Z","shell.execute_reply.started":"2022-10-05T01:51:14.614343Z","shell.execute_reply":"2022-10-05T01:51:14.618179Z"}}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-10-04T16:35:09.301017Z","iopub.execute_input":"2022-10-04T16:35:09.301418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}