{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"There are 3 goals for today. Create a pipeline so that I can automate everthing I did from the last notebook. Find out a way to minimize the size of the dataset. And finally tune the parameters of the lgbm model.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport lightgbm as lgbm\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-23T06:03:33.727904Z","iopub.execute_input":"2022-10-23T06:03:33.728378Z","iopub.status.idle":"2022-10-23T06:03:33.739552Z","shell.execute_reply.started":"2022-10-23T06:03:33.728338Z","shell.execute_reply":"2022-10-23T06:03:33.738150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:03:33.741986Z","iopub.execute_input":"2022-10-23T06:03:33.742428Z","iopub.status.idle":"2022-10-23T06:04:09.794149Z","shell.execute_reply.started":"2022-10-23T06:03:33.742377Z","shell.execute_reply":"2022-10-23T06:04:09.792949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:09.796442Z","iopub.execute_input":"2022-10-23T06:04:09.796822Z","iopub.status.idle":"2022-10-23T06:04:18.422878Z","shell.execute_reply.started":"2022-10-23T06:04:09.796788Z","shell.execute_reply":"2022-10-23T06:04:18.421475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_cols = [col for col in test_df]\nuse_cols.remove('id')\nuse_cols","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.424744Z","iopub.execute_input":"2022-10-23T06:04:18.425910Z","iopub.status.idle":"2022-10-23T06:04:18.436017Z","shell.execute_reply.started":"2022-10-23T06:04:18.425854Z","shell.execute_reply":"2022-10-23T06:04:18.434996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df[use_cols]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.439349Z","iopub.execute_input":"2022-10-23T06:04:18.440228Z","iopub.status.idle":"2022-10-23T06:04:18.549651Z","shell.execute_reply.started":"2022-10-23T06:04:18.440113Z","shell.execute_reply":"2022-10-23T06:04:18.548391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"na_cols = [col for col in test_df.columns if test_df[col].isnull().sum() > 0]\nna_cols","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.551082Z","iopub.execute_input":"2022-10-23T06:04:18.551505Z","iopub.status.idle":"2022-10-23T06:04:18.659403Z","shell.execute_reply.started":"2022-10-23T06:04:18.551467Z","shell.execute_reply":"2022-10-23T06:04:18.658371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in na_cols: # impute missing values\n#     test_df[col] = test_df[col].fillna(test_df[col].median())","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.660826Z","iopub.execute_input":"2022-10-23T06:04:18.661949Z","iopub.status.idle":"2022-10-23T06:04:18.667298Z","shell.execute_reply.started":"2022-10-23T06:04:18.661906Z","shell.execute_reply":"2022-10-23T06:04:18.665699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_cols = [col for col in test_df]\nuse_cols.append('team_A_scoring_within_10sec')\nuse_cols.append('team_B_scoring_within_10sec')\nuse_cols","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.669029Z","iopub.execute_input":"2022-10-23T06:04:18.669565Z","iopub.status.idle":"2022-10-23T06:04:18.684257Z","shell.execute_reply.started":"2022-10-23T06:04:18.669510Z","shell.execute_reply":"2022-10-23T06:04:18.682890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data(i):\n    df = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/train_{i}.csv') # read csv file\n    df = df[use_cols] # trim columns\n    na_cols = [col for col in df.columns if df[col].isnull().sum() > 0]\n    for col in na_cols: # impute missing values\n        df[col] = df[col].fillna(df[col].median())\n    # prepare training set\n    AX = df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'],axis=1)\n    Ay = df['team_A_scoring_within_10sec']\n    BX = df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'],axis=1)\n    By = df['team_B_scoring_within_10sec']\n    \n    #train the data\n    print('Training for A')\n    A_clf.fit(AX, Ay)\n    print('Training for B')\n    B_clf.fit(BX, By)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.685722Z","iopub.execute_input":"2022-10-23T06:04:18.686830Z","iopub.status.idle":"2022-10-23T06:04:18.696164Z","shell.execute_reply.started":"2022-10-23T06:04:18.686786Z","shell.execute_reply":"2022-10-23T06:04:18.694863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A_clf = lgbm.LGBMClassifier(verbose=1)\nB_clf = lgbm.LGBMClassifier(verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.697506Z","iopub.execute_input":"2022-10-23T06:04:18.697914Z","iopub.status.idle":"2022-10-23T06:04:18.712478Z","shell.execute_reply.started":"2022-10-23T06:04:18.697880Z","shell.execute_reply":"2022-10-23T06:04:18.710931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I first tried training the model with the whole dataset but the score was actually lower than what I traied on a single dataset.","metadata":{}},{"cell_type":"code","source":"# %%time\n# for i in range(10):\n#     print(f'Training on dataset {i}')\n#     process_data(i)\n#     print(f'Finished Training on dataset {i}')\n#     print()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.717925Z","iopub.execute_input":"2022-10-23T06:04:18.718459Z","iopub.status.idle":"2022-10-23T06:04:18.724488Z","shell.execute_reply.started":"2022-10-23T06:04:18.718420Z","shell.execute_reply":"2022-10-23T06:04:18.723123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Takes a little over 20 minutes to train the entire dataset","metadata":{}},{"cell_type":"markdown","source":"So let's figure out the feature importance and trim out some columns.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/train_0.csv') # read csv file\ndf = df[use_cols] # trim columns\nna_cols = [col for col in df.columns if df[col].isnull().sum() > 0]\n# for col in na_cols: # impute missing values\n#     df[col] = df[col].fillna(df[col].median())\n# prepare training set\nAX = df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'],axis=1)\nAy = df['team_A_scoring_within_10sec']\nBX = df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'],axis=1)\nBy = df['team_B_scoring_within_10sec']\n\n#train the data\nprint('Training for A')\nA_clf.fit(AX, Ay)\nprint('Training for B')\nB_clf.fit(BX, By)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:04:18.726660Z","iopub.execute_input":"2022-10-23T06:04:18.727527Z","iopub.status.idle":"2022-10-23T06:06:28.040114Z","shell.execute_reply.started":"2022-10-23T06:04:18.727474Z","shell.execute_reply":"2022-10-23T06:06:28.039087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt \n\ndef plot_feature_importance(importance,names,model_type):\n\n    #Create arrays from feature importance and feature names\n    feature_importance = np.array(importance)\n    feature_names = np.array(names)\n\n    #Create a DataFrame using a Dictionary\n    data={'feature_names':feature_names,'feature_importance':feature_importance}\n    fi_df = pd.DataFrame(data)\n\n    #Sort the DataFrame in order decreasing feature importance\n    fi_df.sort_values(by=['feature_importance'], ascending=False,inplace=True)\n\n    #Define size of bar plot\n    plt.figure(figsize=(20,16))\n    #Plot Searborn bar chart\n    sns.barplot(x=fi_df['feature_importance'], y=fi_df['feature_names'])\n    #Add chart labels\n    plt.title(model_type + 'FEATURE IMPORTANCE')\n    plt.xlabel('FEATURE IMPORTANCE')\n    plt.ylabel('FEATURE NAMES')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:28.045406Z","iopub.execute_input":"2022-10-23T06:06:28.045866Z","iopub.status.idle":"2022-10-23T06:06:28.152538Z","shell.execute_reply.started":"2022-10-23T06:06:28.045827Z","shell.execute_reply":"2022-10-23T06:06:28.151070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_importance(A_clf.feature_importances_, AX.columns, 'LGBM')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:28.154255Z","iopub.execute_input":"2022-10-23T06:06:28.155638Z","iopub.status.idle":"2022-10-23T06:06:29.317135Z","shell.execute_reply.started":"2022-10-23T06:06:28.155578Z","shell.execute_reply":"2022-10-23T06:06:29.316207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_importance(B_clf.feature_importances_, BX.columns, 'LGBM')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:29.318921Z","iopub.execute_input":"2022-10-23T06:06:29.319657Z","iopub.status.idle":"2022-10-23T06:06:30.652400Z","shell.execute_reply.started":"2022-10-23T06:06:29.319616Z","shell.execute_reply":"2022-10-23T06:06:30.650660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AX.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:30.666151Z","iopub.execute_input":"2022-10-23T06:06:30.666598Z","iopub.status.idle":"2022-10-23T06:06:30.675952Z","shell.execute_reply.started":"2022-10-23T06:06:30.666560Z","shell.execute_reply":"2022-10-23T06:06:30.674772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AX.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:30.666151Z","iopub.execute_input":"2022-10-23T06:06:30.666598Z","iopub.status.idle":"2022-10-23T06:06:30.675952Z","shell.execute_reply.started":"2022-10-23T06:06:30.666560Z","shell.execute_reply":"2022-10-23T06:06:30.674772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = {'col':AX.columns, 'fi':A_clf.feature_importances_}\nfi_df = pd.DataFrame(data)\nfi_cols = fi_df.sort_values(by='fi').iloc[:int(len(AX.columns))]['col']\nfi_cols = list(fi_cols)\nfi_cols.append('team_A_scoring_within_10sec')\nfi_cols.append('team_B_scoring_within_10sec')\nfi_cols","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:30.677983Z","iopub.execute_input":"2022-10-23T06:06:30.678435Z","iopub.status.idle":"2022-10-23T06:06:30.693566Z","shell.execute_reply.started":"2022-10-23T06:06:30.678398Z","shell.execute_reply":"2022-10-23T06:06:30.691972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A_fi_clf = lgbm.LGBMClassifier(verbose=1)\nB_fi_clf = lgbm.LGBMClassifier(verbose=1)\n\ndf = df[fi_cols]\n\n# prepare training set\nAX = df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'],axis=1)\nAy = df['team_A_scoring_within_10sec']\nBX = df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'],axis=1)\nBy = df['team_B_scoring_within_10sec']\n\n#train the data\nprint('Training for A')\nA_fi_clf.fit(AX, Ay)\nprint('Training for B')\nB_fi_clf.fit(BX, By)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:06:30.695249Z","iopub.execute_input":"2022-10-23T06:06:30.695733Z","iopub.status.idle":"2022-10-23T06:08:11.334166Z","shell.execute_reply.started":"2022-10-23T06:06:30.695695Z","shell.execute_reply":"2022-10-23T06:08:11.333040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_cols.remove('team_A_scoring_within_10sec')\nfi_cols.remove('team_B_scoring_within_10sec')\ntest_df = test_df[fi_cols]\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:08:11.336033Z","iopub.execute_input":"2022-10-23T06:08:11.337130Z","iopub.status.idle":"2022-10-23T06:08:11.554458Z","shell.execute_reply.started":"2022-10-23T06:08:11.337037Z","shell.execute_reply":"2022-10-23T06:08:11.553049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A_preds = A_fi_clf.predict_proba(test_df)[:,1]\nA_preds","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:08:11.556449Z","iopub.execute_input":"2022-10-23T06:08:11.557633Z","iopub.status.idle":"2022-10-23T06:08:13.532956Z","shell.execute_reply.started":"2022-10-23T06:08:11.557581Z","shell.execute_reply":"2022-10-23T06:08:13.531935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_preds = B_fi_clf.predict_proba(test_df)[:,1]\nB_preds","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:08:13.535105Z","iopub.execute_input":"2022-10-23T06:08:13.536408Z","iopub.status.idle":"2022-10-23T06:08:15.419956Z","shell.execute_reply.started":"2022-10-23T06:08:13.536349Z","shell.execute_reply":"2022-10-23T06:08:15.418766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')\nsample_df","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:08:15.422341Z","iopub.execute_input":"2022-10-23T06:08:15.422871Z","iopub.status.idle":"2022-10-23T06:08:15.580687Z","shell.execute_reply.started":"2022-10-23T06:08:15.422817Z","shell.execute_reply":"2022-10-23T06:08:15.579403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df['team_A_scoring_within_10sec'] = A_preds\nsample_df['team_B_scoring_within_10sec'] = B_preds\nsample_df","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:08:15.582220Z","iopub.execute_input":"2022-10-23T06:08:15.582628Z","iopub.status.idle":"2022-10-23T06:08:15.606730Z","shell.execute_reply.started":"2022-10-23T06:08:15.582595Z","shell.execute_reply":"2022-10-23T06:08:15.605410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.to_csv('submisison.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T06:08:15.608180Z","iopub.execute_input":"2022-10-23T06:08:15.608549Z","iopub.status.idle":"2022-10-23T06:08:18.349082Z","shell.execute_reply.started":"2022-10-23T06:08:15.608517Z","shell.execute_reply":"2022-10-23T06:08:18.347755Z"},"trusted":true},"execution_count":null,"outputs":[]}]}