{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Today I tried to make a very simple baseline model for this competition. This competitions is the hardest one yet. With way more data than usual. I will try to improve this model in the coming days.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-06T12:21:42.970571Z","iopub.execute_input":"2022-10-06T12:21:42.971131Z","iopub.status.idle":"2022-10-06T12:21:42.989349Z","shell.execute_reply.started":"2022-10-06T12:21:42.971006Z","shell.execute_reply":"2022-10-06T12:21:42.987629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 10 train files.","metadata":{}},{"cell_type":"code","source":"train0_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv')\ntest_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:21:42.991367Z","iopub.execute_input":"2022-10-06T12:21:42.992127Z","iopub.status.idle":"2022-10-06T12:22:18.177097Z","shell.execute_reply.started":"2022-10-06T12:21:42.992085Z","shell.execute_reply":"2022-10-06T12:22:18.175974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:18.179374Z","iopub.execute_input":"2022-10-06T12:22:18.179843Z","iopub.status.idle":"2022-10-06T12:22:18.704752Z","shell.execute_reply.started":"2022-10-06T12:22:18.179810Z","shell.execute_reply":"2022-10-06T12:22:18.703533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"over 2 million rows with 61 columns. This is huge.","metadata":{}},{"cell_type":"code","source":"train0_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:18.706013Z","iopub.execute_input":"2022-10-06T12:22:18.706349Z","iopub.status.idle":"2022-10-06T12:22:19.039441Z","shell.execute_reply.started":"2022-10-06T12:22:18.706318Z","shell.execute_reply":"2022-10-06T12:22:19.038216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's cut the columns that's not in the test file.","metadata":{}},{"cell_type":"code","source":"use_cols = [i for i in test_df.columns]\nuse_cols.pop(0)\nuse_cols.append('team_A_scoring_within_10sec')\nuse_cols.append('team_B_scoring_within_10sec')\nuse_cols","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:19.041969Z","iopub.execute_input":"2022-10-06T12:22:19.042418Z","iopub.status.idle":"2022-10-06T12:22:19.051169Z","shell.execute_reply.started":"2022-10-06T12:22:19.042385Z","shell.execute_reply":"2022-10-06T12:22:19.049979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df = train0_df[use_cols]","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:19.052611Z","iopub.execute_input":"2022-10-06T12:22:19.053045Z","iopub.status.idle":"2022-10-06T12:22:19.444478Z","shell.execute_reply.started":"2022-10-06T12:22:19.053016Z","shell.execute_reply":"2022-10-06T12:22:19.442817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fill in NAs with median values","metadata":{}},{"cell_type":"markdown","source":"train0_df.fillna(lambda x: x.median()) does not work. It does not fill NAs with float for string value, they fill them with functions. So when you want to process the values that was NAs, which is now functions, it cannot be done and returns an error.","metadata":{}},{"cell_type":"code","source":"for i in train0_df.columns:     #df.columns[w:] if you have w column of line description \n    train0_df[i] = train0_df[i].fillna(train0_df[i].median() )","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:19.445991Z","iopub.execute_input":"2022-10-06T12:22:19.446323Z","iopub.status.idle":"2022-10-06T12:22:22.138816Z","shell.execute_reply.started":"2022-10-06T12:22:19.446294Z","shell.execute_reply":"2022-10-06T12:22:22.137636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:22.140308Z","iopub.execute_input":"2022-10-06T12:22:22.140772Z","iopub.status.idle":"2022-10-06T12:22:22.275862Z","shell.execute_reply.started":"2022-10-06T12:22:22.140728Z","shell.execute_reply":"2022-10-06T12:22:22.274561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')\nsample_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:22.277738Z","iopub.execute_input":"2022-10-06T12:22:22.278917Z","iopub.status.idle":"2022-10-06T12:22:22.405950Z","shell.execute_reply.started":"2022-10-06T12:22:22.278875Z","shell.execute_reply":"2022-10-06T12:22:22.404797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For submission, we need to find two values. I will train two models to predict each.","metadata":{}},{"cell_type":"code","source":"train_A_df = train0_df.drop(['team_B_scoring_within_10sec'], axis=1)\ntrain_A_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:22.407323Z","iopub.execute_input":"2022-10-06T12:22:22.407683Z","iopub.status.idle":"2022-10-06T12:22:23.295714Z","shell.execute_reply.started":"2022-10-06T12:22:22.407651Z","shell.execute_reply":"2022-10-06T12:22:23.294345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_A_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:23.300940Z","iopub.execute_input":"2022-10-06T12:22:23.301745Z","iopub.status.idle":"2022-10-06T12:22:27.921234Z","shell.execute_reply.started":"2022-10-06T12:22:23.301704Z","shell.execute_reply":"2022-10-06T12:22:27.920381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fill in the NAs of test_df","metadata":{}},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:27.922582Z","iopub.execute_input":"2022-10-06T12:22:27.923453Z","iopub.status.idle":"2022-10-06T12:22:28.009661Z","shell.execute_reply.started":"2022-10-06T12:22:27.923416Z","shell.execute_reply":"2022-10-06T12:22:28.008522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test_df.columns:     #df.columns[w:] if you have w column of line description \n    test_df[i] = test_df[i].fillna(test_df[i].median() )","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:28.010942Z","iopub.execute_input":"2022-10-06T12:22:28.011269Z","iopub.status.idle":"2022-10-06T12:22:29.154241Z","shell.execute_reply.started":"2022-10-06T12:22:28.011240Z","shell.execute_reply":"2022-10-06T12:22:29.152802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:29.155505Z","iopub.execute_input":"2022-10-06T12:22:29.155912Z","iopub.status.idle":"2022-10-06T12:22:29.240772Z","shell.execute_reply.started":"2022-10-06T12:22:29.155876Z","shell.execute_reply":"2022-10-06T12:22:29.239394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I could not fit all of the 2 million rows into the sklearn's LogisticRegression algorithm so at first I tried to split the data and just train with a tenth of the training data.","metadata":{}},{"cell_type":"code","source":"# import random\n# from sklearn.model_selection import train_test_split\n\n# # AX, y = train_A_df.drop(['team_A_scoring_within_10sec'], axis=1), train_A_df['team_A_scoring_within_10sec']\n# # x_train, AX_train, y, y_train = train_test_split(AX, y, test_size=0.1, random_state=random.randint(1,100))","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:29.242696Z","iopub.execute_input":"2022-10-06T12:22:29.243064Z","iopub.status.idle":"2022-10-06T12:22:29.247998Z","shell.execute_reply.started":"2022-10-06T12:22:29.243032Z","shell.execute_reply":"2022-10-06T12:22:29.246806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But then I realized that I could just train 10% of the data once at a time.","metadata":{}},{"cell_type":"markdown","source":"The score was not very good. Then I realized that I should predict the probability rather than the binary outcome.","metadata":{}},{"cell_type":"code","source":"import random\nfrom sklearn.linear_model import LogisticRegression\n\nAX, Ay = train_A_df.drop(['team_A_scoring_within_10sec'], axis=1), train_A_df['team_A_scoring_within_10sec']\nA_clf = LogisticRegression(random_state=random.randint(1,100))\nfor i in range(1,11):\n    X, y = AX.iloc[200000*(i-1):200000*i,:], Ay[200000*(i-1):200000*i]\n    print(len(X),len(y))\n    A_clf.fit(X,y)\nX, y = AX.iloc[200000*(10):,:], Ay[200000*(10):]\nA_clf.fit(X,y)\nA_preds = A_clf.predict_proba(test_df.drop(['id'], axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:22:29.249343Z","iopub.execute_input":"2022-10-06T12:22:29.249760Z","iopub.status.idle":"2022-10-06T12:23:20.348167Z","shell.execute_reply.started":"2022-10-06T12:22:29.249726Z","shell.execute_reply":"2022-10-06T12:23:20.346236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.linear_model import LogisticRegression\n\n# clf = LogisticRegression(random_state=random.randint(1,100)).fit(AX_train, y_train)\n# A_preds = clf.predict(test_df.drop(['id'], axis=1))\n# A_preds","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:23:20.350439Z","iopub.execute_input":"2022-10-06T12:23:20.351514Z","iopub.status.idle":"2022-10-06T12:23:20.358198Z","shell.execute_reply.started":"2022-10-06T12:23:20.351427Z","shell.execute_reply":"2022-10-06T12:23:20.355930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do the same for B","metadata":{}},{"cell_type":"code","source":"train_B_df = train0_df.drop(['team_A_scoring_within_10sec'], axis=1)\ntrain_B_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:23:20.361033Z","iopub.execute_input":"2022-10-06T12:23:20.362349Z","iopub.status.idle":"2022-10-06T12:23:21.271461Z","shell.execute_reply.started":"2022-10-06T12:23:20.362277Z","shell.execute_reply":"2022-10-06T12:23:21.270539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import random\n# from sklearn.model_selection import train_test_split\n\n# BX, y = train_B_df.drop(['team_B_scoring_within_10sec'], axis=1), train_B_df['team_B_scoring_within_10sec']\n# x_train, BX_train, y, y_train = train_test_split(BX, y, test_size=0.1, random_state=random.randint(1,100))","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:23:21.272812Z","iopub.execute_input":"2022-10-06T12:23:21.274055Z","iopub.status.idle":"2022-10-06T12:23:21.279123Z","shell.execute_reply.started":"2022-10-06T12:23:21.273952Z","shell.execute_reply":"2022-10-06T12:23:21.277722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clf = LogisticRegression(random_state=random.randint(1,100)).fit(BX_train, y_train)\n# B_preds = clf.predict(test_df.drop(['id'], axis=1))\n# B_preds","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:23:21.280837Z","iopub.execute_input":"2022-10-06T12:23:21.281196Z","iopub.status.idle":"2022-10-06T12:23:21.295605Z","shell.execute_reply.started":"2022-10-06T12:23:21.281164Z","shell.execute_reply":"2022-10-06T12:23:21.294139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nBX, By = train_B_df.drop(['team_B_scoring_within_10sec'], axis=1), train_B_df['team_B_scoring_within_10sec']\nB_clf = LogisticRegression(random_state=random.randint(1,100))\nfor i in range(1,11):\n    X, y = BX.iloc[200000*(i-1):200000*i,:], By[200000*(i-1):200000*i]\n    print(len(X),len(y))\n    B_clf.fit(X,y)\nX, y = BX.iloc[200000*(10):,:], By[200000*(10):]\nB_clf.fit(X,y)\nB_preds = B_clf.predict_proba(test_df.drop(['id'], axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:23:21.297246Z","iopub.execute_input":"2022-10-06T12:23:21.297642Z","iopub.status.idle":"2022-10-06T12:24:14.196064Z","shell.execute_reply.started":"2022-10-06T12:23:21.297589Z","shell.execute_reply":"2022-10-06T12:24:14.194413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_preds = B_clf.predict_proba(test_df.drop(['id'], axis=1))\nB_preds[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:24:14.198099Z","iopub.execute_input":"2022-10-06T12:24:14.202729Z","iopub.status.idle":"2022-10-06T12:24:14.406101Z","shell.execute_reply.started":"2022-10-06T12:24:14.202623Z","shell.execute_reply":"2022-10-06T12:24:14.404496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from xgboost import XGBClassifier\n\n# AX, y = train_A_df.drop(['team_A_scoring_within_10sec'], axis=1), train_A_df['team_A_scoring_within_10sec']\n\n# xgbc = XGBClassifier()\n# xgbc.fit(AX, y)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:24:14.408748Z","iopub.execute_input":"2022-10-06T12:24:14.409759Z","iopub.status.idle":"2022-10-06T12:24:14.417172Z","shell.execute_reply.started":"2022-10-06T12:24:14.409684Z","shell.execute_reply":"2022-10-06T12:24:14.415511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# A_pred = xgbc.predict(test_df)\n# A_pred","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:24:14.418857Z","iopub.execute_input":"2022-10-06T12:24:14.419409Z","iopub.status.idle":"2022-10-06T12:24:14.429617Z","shell.execute_reply.started":"2022-10-06T12:24:14.419358Z","shell.execute_reply":"2022-10-06T12:24:14.428160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And submit!","metadata":{}},{"cell_type":"code","source":"sample_df['team_A_scoring_within_10sec'] = A_preds[:,1]\nsample_df['team_B_scoring_within_10sec'] = B_preds[:,1]\nsample_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:24:35.132747Z","iopub.execute_input":"2022-10-06T12:24:35.133717Z","iopub.status.idle":"2022-10-06T12:24:35.155907Z","shell.execute_reply.started":"2022-10-06T12:24:35.133665Z","shell.execute_reply":"2022-10-06T12:24:35.154714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:24:14.521285Z","iopub.execute_input":"2022-10-06T12:24:14.521814Z","iopub.status.idle":"2022-10-06T12:24:17.105143Z","shell.execute_reply.started":"2022-10-06T12:24:14.521779Z","shell.execute_reply":"2022-10-06T12:24:17.103702Z"},"trusted":true},"execution_count":null,"outputs":[]}]}