{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing the necessary libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings( 'ignore' )\npd.options.display.max_colwidth = 250\npd.options.display.max_columns = 50","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:42:56.957746Z","iopub.execute_input":"2022-07-27T15:42:56.958283Z","iopub.status.idle":"2022-07-27T15:42:57.597086Z","shell.execute_reply.started":"2022-07-27T15:42:56.958170Z","shell.execute_reply":"2022-07-27T15:42:57.595979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Much thanks to https://www.kaggle.com/code/flaviocavalcante/tps-may-22-just-eda/notebook\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:05.734961Z","iopub.execute_input":"2022-07-27T15:43:05.735383Z","iopub.status.idle":"2022-07-27T15:43:05.747697Z","shell.execute_reply.started":"2022-07-27T15:43:05.735347Z","shell.execute_reply":"2022-07-27T15:43:05.746559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data into 3 dataframes\n\ntrain_df = pd.read_csv('../input/tabular-playground-series-may-2022/train.csv')\ntest_df = pd.read_csv('../input/tabular-playground-series-may-2022/test.csv')\nsub = pd.read_csv('../input/tabular-playground-series-may-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:09.620759Z","iopub.execute_input":"2022-07-27T15:43:09.621122Z","iopub.status.idle":"2022-07-27T15:43:23.849099Z","shell.execute_reply.started":"2022-07-27T15:43:09.621084Z","shell.execute_reply":"2022-07-27T15:43:23.847990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce the memory usage\n\nreduce_mem_usage(train_df)\nreduce_mem_usage(test_df)\nreduce_mem_usage(sub)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:26.768595Z","iopub.execute_input":"2022-07-27T15:43:26.769006Z","iopub.status.idle":"2022-07-27T15:43:28.067860Z","shell.execute_reply.started":"2022-07-27T15:43:26.768971Z","shell.execute_reply":"2022-07-27T15:43:28.066541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# review samples of the data\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:37.318124Z","iopub.execute_input":"2022-07-27T15:43:37.318618Z","iopub.status.idle":"2022-07-27T15:43:37.352441Z","shell.execute_reply.started":"2022-07-27T15:43:37.318570Z","shell.execute_reply":"2022-07-27T15:43:37.351235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:40.905953Z","iopub.execute_input":"2022-07-27T15:43:40.906833Z","iopub.status.idle":"2022-07-27T15:43:40.938590Z","shell.execute_reply.started":"2022-07-27T15:43:40.906776Z","shell.execute_reply":"2022-07-27T15:43:40.937730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:41.875194Z","iopub.execute_input":"2022-07-27T15:43:41.875839Z","iopub.status.idle":"2022-07-27T15:43:41.890523Z","shell.execute_reply.started":"2022-07-27T15:43:41.875806Z","shell.execute_reply":"2022-07-27T15:43:41.889576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the shape of the dataframes\n\ntrain_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:43.488102Z","iopub.execute_input":"2022-07-27T15:43:43.489687Z","iopub.status.idle":"2022-07-27T15:43:43.499688Z","shell.execute_reply.started":"2022-07-27T15:43:43.489616Z","shell.execute_reply":"2022-07-27T15:43:43.498718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the stats of the data\n\ntrain_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:45.004995Z","iopub.execute_input":"2022-07-27T15:43:45.005846Z","iopub.status.idle":"2022-07-27T15:43:47.570087Z","shell.execute_reply.started":"2022-07-27T15:43:45.005796Z","shell.execute_reply":"2022-07-27T15:43:47.568873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:48.663524Z","iopub.execute_input":"2022-07-27T15:43:48.663921Z","iopub.status.idle":"2022-07-27T15:43:48.830933Z","shell.execute_reply.started":"2022-07-27T15:43:48.663888Z","shell.execute_reply":"2022-07-27T15:43:48.829542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe(include=\"O\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:49.886741Z","iopub.execute_input":"2022-07-27T15:43:49.887519Z","iopub.status.idle":"2022-07-27T15:43:50.708008Z","shell.execute_reply.started":"2022-07-27T15:43:49.887469Z","shell.execute_reply":"2022-07-27T15:43:50.706746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:51.331239Z","iopub.execute_input":"2022-07-27T15:43:51.331607Z","iopub.status.idle":"2022-07-27T15:43:53.316833Z","shell.execute_reply.started":"2022-07-27T15:43:51.331577Z","shell.execute_reply":"2022-07-27T15:43:53.315669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:53.318888Z","iopub.execute_input":"2022-07-27T15:43:53.319320Z","iopub.status.idle":"2022-07-27T15:43:53.440002Z","shell.execute_reply.started":"2022-07-27T15:43:53.319276Z","shell.execute_reply":"2022-07-27T15:43:53.438618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.describe(include=\"O\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:55.523530Z","iopub.execute_input":"2022-07-27T15:43:55.523954Z","iopub.status.idle":"2022-07-27T15:43:56.184878Z","shell.execute_reply.started":"2022-07-27T15:43:55.523924Z","shell.execute_reply":"2022-07-27T15:43:56.183986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the nan values\n\ntrain_df.isnull().sum().sum(),test_df.isnull().sum().sum() ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:56.967867Z","iopub.execute_input":"2022-07-27T15:43:56.968241Z","iopub.status.idle":"2022-07-27T15:43:57.194704Z","shell.execute_reply.started":"2022-07-27T15:43:56.968211Z","shell.execute_reply":"2022-07-27T15:43:57.193660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking duplicated values\n\ntrain_df.duplicated().sum(), test_df.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:43:59.339336Z","iopub.execute_input":"2022-07-27T15:43:59.339877Z","iopub.status.idle":"2022-07-27T15:44:02.297079Z","shell.execute_reply.started":"2022-07-27T15:43:59.339838Z","shell.execute_reply":"2022-07-27T15:44:02.295733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting the categorical column into 10 other columns and one more for the number of unique values.\nfor df in [train_df, test_df]:\n    for i in range(10):\n        df[f'ch{i}'] = df.f_27.str.get(i).apply(ord) - ord('A')\n    df[\"unique_characters\"] = df.f_27.apply(lambda s: len(set(s)))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:44:04.006737Z","iopub.execute_input":"2022-07-27T15:44:04.007140Z","iopub.status.idle":"2022-07-27T15:44:15.213631Z","shell.execute_reply.started":"2022-07-27T15:44:04.007105Z","shell.execute_reply":"2022-07-27T15:44:15.212554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reduce_mem_usage(train_df)\nreduce_mem_usage(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:53:40.003480Z","iopub.execute_input":"2022-07-27T15:53:40.004031Z","iopub.status.idle":"2022-07-27T15:53:41.624369Z","shell.execute_reply.started":"2022-07-27T15:53:40.003990Z","shell.execute_reply":"2022-07-27T15:53:41.623374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:44:16.423988Z","iopub.execute_input":"2022-07-27T15:44:16.424765Z","iopub.status.idle":"2022-07-27T15:44:16.456859Z","shell.execute_reply.started":"2022-07-27T15:44:16.424718Z","shell.execute_reply":"2022-07-27T15:44:16.455569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:44:30.104427Z","iopub.execute_input":"2022-07-27T15:44:30.105733Z","iopub.status.idle":"2022-07-27T15:44:30.137507Z","shell.execute_reply.started":"2022-07-27T15:44:30.105682Z","shell.execute_reply":"2022-07-27T15:44:30.136394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(['id', 'target', 'f_27'], axis = 1)\ny = train_df['target']\ntest = test_df.drop(['id','f_27'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:54:08.621671Z","iopub.execute_input":"2022-07-27T15:54:08.622094Z","iopub.status.idle":"2022-07-27T15:54:08.738553Z","shell.execute_reply.started":"2022-07-27T15:54:08.622054Z","shell.execute_reply":"2022-07-27T15:54:08.737403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cont = X.select_dtypes(include=['float16'])\nquant = X.select_dtypes(include=['int8'])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:54:10.713455Z","iopub.execute_input":"2022-07-27T15:54:10.713864Z","iopub.status.idle":"2022-07-27T15:54:10.756285Z","shell.execute_reply.started":"2022-07-27T15:54:10.713825Z","shell.execute_reply":"2022-07-27T15:54:10.755445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(4,4,figsize=(25,18))\nk=0\nj=0\nfor col in cont:\n    sns.kdeplot(X[col], ax=ax[k,j],\n                shade=True,\n                color='#2f5586', edgecolor='black',\n                linewidth=1.5, alpha=0.9,\n                zorder=3\n               )\n\n    ax[k,j].set_xlabel(col, fontsize=17, color=\"k\")\n    ax[k,j].set_ylabel(\"Density\", fontsize=17, color=\"k\")\n    #ax[k,j].set_xticklabels(fontsize=11, color=\"k\")\n    if j>=3:\n        k+=1\n        j=-1\n    j+=1\nfig.suptitle('Float Variables Density (Train Dataset)', fontsize=25, color=\"k\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:41.348297Z","iopub.execute_input":"2022-07-27T15:46:41.348774Z","iopub.status.idle":"2022-07-27T15:47:41.795569Z","shell.execute_reply.started":"2022-07-27T15:46:41.348737Z","shell.execute_reply":"2022-07-27T15:47:41.794458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(4,4,figsize=(25,18))\nk=0\nj=0\nfor col in cont:\n    sns.kdeplot(test[col], ax=ax[k,j],\n                shade=True,\n                color='#2f5586', edgecolor='black',\n                linewidth=1.5, alpha=0.9,\n                zorder=3\n               )\n\n    ax[k,j].set_xlabel(col, fontsize=17, color=\"k\")\n    ax[k,j].set_ylabel(\"Density\", fontsize=17, color=\"k\")\n    #ax[k,j].set_xticklabels(fontsize=11, color=\"k\")\n    if j>=3:\n        k+=1\n        j=-1\n    j+=1\nfig.suptitle('Float Variables Density (Test Dataset)', fontsize=25, color=\"k\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:48:18.651461Z","iopub.execute_input":"2022-07-27T15:48:18.652289Z","iopub.status.idle":"2022-07-27T15:49:06.113014Z","shell.execute_reply.started":"2022-07-27T15:48:18.652244Z","shell.execute_reply":"2022-07-27T15:49:06.111896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(7,4,figsize=(25,18))\nk=0\nj=0\nfor col in quant:\n    ax[k,j].hist(X[col], label=\"Train Dataset\", alpha=0.8, color=\"orange\", bins=15)\n    ax[k,j].hist(test_df[col], label=\"Test Dataset\",alpha=0.4, color=\"b\", bins=15)\n    \n    ax[k,j].set_xlabel(col, fontsize=17, color=\"k\")\n    ax[k,j].set_ylabel(\"Frequency\", fontsize=17, color=\"k\")\n    #ax[k,j].set_xticklabels(fontsize=11, color=\"k\")\n    if j>=3:\n        k+=1\n        j=-1\n    j+=1\nfig.suptitle('Quantitative Variables Distribution (Train & Test)', fontsize=25, color=\"k\")\nfig.legend([\"Train Dataset\",\"Test Dataset\"], fontsize=20);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:56:13.765015Z","iopub.execute_input":"2022-07-27T15:56:13.765395Z","iopub.status.idle":"2022-07-27T15:56:18.426175Z","shell.execute_reply.started":"2022-07-27T15:56:13.765365Z","shell.execute_reply":"2022-07-27T15:56:18.425005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Building the model \n\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom hyperopt import hp, tpe, fmin, STATUS_OK, Trials\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import roc_auc_score, roc_curve\nfrom sklearn.metrics import roc_auc_score, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:03:04.613656Z","iopub.execute_input":"2022-07-27T17:03:04.614170Z","iopub.status.idle":"2022-07-27T17:03:04.621044Z","shell.execute_reply.started":"2022-07-27T17:03:04.614132Z","shell.execute_reply":"2022-07-27T17:03:04.619698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X, test_X, train_y, test_y = train_test_split(X, y, random_state = 1, test_size=0.25)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:03:07.712427Z","iopub.execute_input":"2022-07-27T17:03:07.713330Z","iopub.status.idle":"2022-07-27T17:03:08.035232Z","shell.execute_reply.started":"2022-07-27T17:03:07.713286Z","shell.execute_reply":"2022-07-27T17:03:08.034187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Bulid first model\n\nclf = lgb.LGBMClassifier()\nclf.fit(train_X, train_y)\npreds = clf.predict(test_X)\nscore = roc_auc_score(test_y, preds)\naccuracy = accuracy_score(preds, test_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:06:16.892369Z","iopub.execute_input":"2022-07-27T16:06:16.892764Z","iopub.status.idle":"2022-07-27T16:06:27.866901Z","shell.execute_reply.started":"2022-07-27T16:06:16.892732Z","shell.execute_reply":"2022-07-27T16:06:27.865966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(score)\nprint(accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:06:44.180171Z","iopub.execute_input":"2022-07-27T16:06:44.180515Z","iopub.status.idle":"2022-07-27T16:06:44.185487Z","shell.execute_reply.started":"2022-07-27T16:06:44.180487Z","shell.execute_reply":"2022-07-27T16:06:44.184741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tuning Hyperprameters with HyperOpt**","metadata":{}},{"cell_type":"code","source":"%%time\nspace = {'n_estimators': hp.quniform('n_estimators', 10, 1000,10),\n        'learning_rate': hp.uniform('learning_rate', 0.001, 0.9),\n         'max_depth': hp.quniform('max_depth', 3,18,1),\n         'num_leaves': hp.quniform('num_leaves', 30,50,1),\n         'reg_alpha': hp.quniform('reg_alpha', 1.1,1.5,0.1),\n         'colsample_bytree': hp.uniform('colsample_bytree', 0.5, 1),\n         'reg_lambda': hp.uniform('reg_lambda', 0, 1),\n         'min_child_weight': hp.quniform('min_child_weight',0,10,1)\n        }\n\ndef objective(space):\n\n    lgbm= lgb.LGBMClassifier(n_estimators= int(space['n_estimators']),\n             learning_rate= space['learning_rate'],\n             max_depth= int(space['max_depth']),\n             num_leaves = int(space['num_leaves']),\n             reg_alpha = space['reg_alpha'],\n             colsample_bytree = space['colsample_bytree'],\n             reg_lambda= space['reg_lambda'],\n             min_child_weight = space['min_child_weight'])\n    \n    score = -cross_val_score(lgbm, train_X, train_y, cv=4, scoring='roc_auc').mean()\n    return score","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:39:50.610695Z","iopub.execute_input":"2022-07-27T17:39:50.611131Z","iopub.status.idle":"2022-07-27T17:39:50.620303Z","shell.execute_reply.started":"2022-07-27T17:39:50.611096Z","shell.execute_reply":"2022-07-27T17:39:50.619323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrials = Trials()\nbest = fmin(fn=objective,\n            space=space,\n            algo=tpe.suggest,\n            max_evals=100,\n            trials=trials)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:39:53.068733Z","iopub.execute_input":"2022-07-27T17:39:53.069130Z","iopub.status.idle":"2022-07-27T21:08:39.347521Z","shell.execute_reply.started":"2022-07-27T17:39:53.069096Z","shell.execute_reply":"2022-07-27T21:08:39.345589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Hyperopt estimated optimum {}\".format(best))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:09:30.441020Z","iopub.execute_input":"2022-07-27T21:09:30.441555Z","iopub.status.idle":"2022-07-27T21:09:30.452332Z","shell.execute_reply.started":"2022-07-27T21:09:30.441515Z","shell.execute_reply":"2022-07-27T21:09:30.451099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using the best parameters obtained from the tuning**","metadata":{}},{"cell_type":"code","source":"%%time\nmodel = lgb.LGBMClassifier(random_state=0,\n                        n_estimators= int(best['n_estimators']),\n                        learning_rate= best['learning_rate'],\n                        max_depth= int(best['max_depth']),\n                        num_leaves = int(best['num_leaves']),\n                        reg_alpha = best['reg_alpha'],\n                        colsample_bytree = best['colsample_bytree'],\n                        reg_lambda= best['reg_lambda'],\n                        min_child_weight = best['min_child_weight'])\nmodel.fit(X, y)\npreds_final = model.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:11:28.738728Z","iopub.execute_input":"2022-07-27T21:11:28.739144Z","iopub.status.idle":"2022-07-27T21:12:55.380007Z","shell.execute_reply.started":"2022-07-27T21:11:28.739095Z","shell.execute_reply":"2022-07-27T21:12:55.379182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final submission\nsub['target'] = preds_final\nsub.to_csv('submission.csv', index = False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:23:37.880418Z","iopub.execute_input":"2022-07-27T21:23:37.880892Z","iopub.status.idle":"2022-07-27T21:23:38.563509Z","shell.execute_reply.started":"2022-07-27T21:23:37.880856Z","shell.execute_reply":"2022-07-27T21:23:38.562654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}