{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !kaggle competitions download -c santander-customer-transaction-prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:35:55.425924Z","iopub.execute_input":"2022-08-01T13:35:55.427229Z","iopub.status.idle":"2022-08-01T13:35:55.452978Z","shell.execute_reply.started":"2022-08-01T13:35:55.427134Z","shell.execute_reply":"2022-08-01T13:35:55.451954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from zipfile import ZipFile","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:35:55.454884Z","iopub.execute_input":"2022-08-01T13:35:55.455487Z","iopub.status.idle":"2022-08-01T13:35:55.459543Z","shell.execute_reply.started":"2022-08-01T13:35:55.455426Z","shell.execute_reply":"2022-08-01T13:35:55.458514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with ZipFile('santander-customer-transaction-prediction.zip', 'r') as zipObj:\n#     zipObj.extractall()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:35:55.460844Z","iopub.execute_input":"2022-08-01T13:35:55.461666Z","iopub.status.idle":"2022-08-01T13:35:55.471855Z","shell.execute_reply.started":"2022-08-01T13:35:55.461632Z","shell.execute_reply":"2022-08-01T13:35:55.470516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:35:55.473960Z","iopub.execute_input":"2022-08-01T13:35:55.474562Z","iopub.status.idle":"2022-08-01T13:35:56.733362Z","shell.execute_reply.started":"2022-08-01T13:35:55.474530Z","shell.execute_reply":"2022-08-01T13:35:56.732250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:35:56.734825Z","iopub.execute_input":"2022-08-01T13:35:56.735264Z","iopub.status.idle":"2022-08-01T13:35:56.742913Z","shell.execute_reply.started":"2022-08-01T13:35:56.735223Z","shell.execute_reply":"2022-08-01T13:35:56.741615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/santander-customer-transaction-prediction/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:12.626768Z","iopub.execute_input":"2022-08-01T13:37:12.627176Z","iopub.status.idle":"2022-08-01T13:37:20.199079Z","shell.execute_reply.started":"2022-08-01T13:37:12.627133Z","shell.execute_reply":"2022-08-01T13:37:20.197868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(columns=['ID_code'], inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:20.200902Z","iopub.execute_input":"2022-08-01T13:37:20.201239Z","iopub.status.idle":"2022-08-01T13:37:20.336516Z","shell.execute_reply.started":"2022-08-01T13:37:20.201209Z","shell.execute_reply":"2022-08-01T13:37:20.335021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:20.338122Z","iopub.execute_input":"2022-08-01T13:37:20.338593Z","iopub.status.idle":"2022-08-01T13:37:20.372037Z","shell.execute_reply.started":"2022-08-01T13:37:20.338540Z","shell.execute_reply":"2022-08-01T13:37:20.370903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:20.375010Z","iopub.execute_input":"2022-08-01T13:37:20.375989Z","iopub.status.idle":"2022-08-01T13:37:22.703939Z","shell.execute_reply.started":"2022-08-01T13:37:20.375949Z","shell.execute_reply":"2022-08-01T13:37:22.702762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.corr()['target'].sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:22.705578Z","iopub.execute_input":"2022-08-01T13:37:22.705926Z","iopub.status.idle":"2022-08-01T13:37:44.651094Z","shell.execute_reply.started":"2022-08-01T13:37:22.705896Z","shell.execute_reply":"2022-08-01T13:37:44.650066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:44.652511Z","iopub.execute_input":"2022-08-01T13:37:44.652830Z","iopub.status.idle":"2022-08-01T13:37:44.746077Z","shell.execute_reply.started":"2022-08-01T13:37:44.652802Z","shell.execute_reply":"2022-08-01T13:37:44.744924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nsns.heatmap(train_df.corr())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:37:44.747530Z","iopub.execute_input":"2022-08-01T13:37:44.747851Z","iopub.status.idle":"2022-08-01T13:38:09.546049Z","shell.execute_reply.started":"2022-08-01T13:37:44.747822Z","shell.execute_reply":"2022-08-01T13:38:09.545241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='target')\nplt.title('Number of target values')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:38:09.547240Z","iopub.execute_input":"2022-08-01T13:38:09.547737Z","iopub.status.idle":"2022-08-01T13:38:09.701021Z","shell.execute_reply.started":"2022-08-01T13:38:09.547707Z","shell.execute_reply":"2022-08-01T13:38:09.699767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_feature_distribution(df):\n    plt.title(\"Distribution of all features\")\n    fig, axes = plt.subplots(20, 10, figsize=(20, 50))\n    fig.subplots_adjust(hspace=1.01, wspace=0.1)\n\n    for i, col in enumerate(df.columns[1:]):\n        plt.subplot(20,10,i+1)\n        sns.histplot(data=df, x=col, kde=True, hue='target')\n        plt.tick_params(axis='both', left=False, bottom=False, labelleft=False)\n        plt.ylabel('')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:38:09.702993Z","iopub.execute_input":"2022-08-01T13:38:09.703758Z","iopub.status.idle":"2022-08-01T13:38:09.715270Z","shell.execute_reply.started":"2022-08-01T13:38:09.703710Z","shell.execute_reply":"2022-08-01T13:38:09.713938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_distribution(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:38:09.723040Z","iopub.execute_input":"2022-08-01T13:38:09.724115Z","iopub.status.idle":"2022-08-01T13:44:08.614726Z","shell.execute_reply.started":"2022-08-01T13:38:09.724061Z","shell.execute_reply":"2022-08-01T13:44:08.613525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA results\n- There are not correlated features\n- All feature distributions are very similar to the normal distribution\n- Number of 0 is much greater than 1","metadata":{}},{"cell_type":"markdown","source":"# Feature engineering and data preprocessing","metadata":{}},{"cell_type":"code","source":"pip install feature_engine","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:44:53.904154Z","iopub.execute_input":"2022-08-01T13:44:53.904674Z","iopub.status.idle":"2022-08-01T13:45:05.104928Z","shell.execute_reply.started":"2022-08-01T13:44:53.904635Z","shell.execute_reply":"2022-08-01T13:45:05.103866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from feature_engine.selection import DropDuplicateFeatures, DropConstantFeatures, DropCorrelatedFeatures","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:45:05.107206Z","iopub.execute_input":"2022-08-01T13:45:05.107596Z","iopub.status.idle":"2022-08-01T13:45:05.456209Z","shell.execute_reply.started":"2022-08-01T13:45:05.107562Z","shell.execute_reply":"2022-08-01T13:45:05.454924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_dupl = DropDuplicateFeatures(variables=None)\ncols_const = DropConstantFeatures(tol=0.99, variables=None)\ncols_corr = DropCorrelatedFeatures(threshold=0.85, method='pearson')\ncols_to_del = set()\nunique_map = {}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:45:05.457893Z","iopub.execute_input":"2022-08-01T13:45:05.458377Z","iopub.status.idle":"2022-08-01T13:45:05.465740Z","shell.execute_reply.started":"2022-08-01T13:45:05.458331Z","shell.execute_reply":"2022-08-01T13:45:05.464536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in tqdm(train_df.columns[1:]):\n    unique_map[col] = (train_df[col].value_counts() == 1).to_dict()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:45:05.470025Z","iopub.execute_input":"2022-08-01T13:45:05.470654Z","iopub.status.idle":"2022-08-01T13:45:34.208426Z","shell.execute_reply.started":"2022-08-01T13:45:05.470605Z","shell.execute_reply":"2022-08-01T13:45:34.207049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_data(df, test=False):\n    global cols_dupl, cols_const, cols_corr, cols_to_del\n    if not test:\n        cols_const.fit(df)\n        cols_to_del = cols_to_del.union(cols_const.features_to_drop_)\n        cols_dupl.fit(df)\n        cols_to_del = cols_to_del.union(cols_dupl.features_to_drop_)\n        cols_corr.fit(df)\n        cols_to_del = cols_to_del.union(cols_corr.features_to_drop_)\n    \n    if not test:\n        for col in tqdm(df.columns[1:]):\n            unique_col = df[col].rename(f\"{col}_unique\").copy()\n            unique_col = unique_col.map(unique_map[col]).astype(bool)\n            df = pd.concat([df, pd.DataFrame(unique_col)], axis=1)\n    else:\n        for col in tqdm(df.columns):\n            unique_col = df[col].rename(f\"{col}_unique\").copy()\n            unique_col = unique_col.map(unique_map[col]).astype(bool)\n            df = pd.concat([df, pd.DataFrame(unique_col)], axis=1)\n    df.drop(columns=cols_to_del, inplace=True, axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:45:34.209727Z","iopub.execute_input":"2022-08-01T13:45:34.210042Z","iopub.status.idle":"2022-08-01T13:45:34.223674Z","shell.execute_reply.started":"2022-08-01T13:45:34.210014Z","shell.execute_reply":"2022-08-01T13:45:34.221421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = preprocess_data(train_df)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:45:34.225404Z","iopub.execute_input":"2022-08-01T13:45:34.225920Z","iopub.status.idle":"2022-08-01T13:46:48.844999Z","shell.execute_reply.started":"2022-08-01T13:45:34.225874Z","shell.execute_reply":"2022-08-01T13:46:48.843252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_del","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:46:48.847610Z","iopub.execute_input":"2022-08-01T13:46:48.847991Z","iopub.status.idle":"2022-08-01T13:46:48.856390Z","shell.execute_reply.started":"2022-08-01T13:46:48.847960Z","shell.execute_reply":"2022-08-01T13:46:48.855295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/santander-customer-transaction-prediction/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:46:48.858052Z","iopub.execute_input":"2022-08-01T13:46:48.858453Z","iopub.status.idle":"2022-08-01T13:47:01.560351Z","shell.execute_reply.started":"2022-08-01T13:46:48.858402Z","shell.execute_reply":"2022-08-01T13:47:01.559234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:01.561605Z","iopub.execute_input":"2022-08-01T13:47:01.562418Z","iopub.status.idle":"2022-08-01T13:47:01.670856Z","shell.execute_reply.started":"2022-08-01T13:47:01.562386Z","shell.execute_reply":"2022-08-01T13:47:01.669816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_code = test_df['ID_code']\ntest_df.drop(columns=['ID_code'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:01.674365Z","iopub.execute_input":"2022-08-01T13:47:01.674692Z","iopub.status.idle":"2022-08-01T13:47:01.787523Z","shell.execute_reply.started":"2022-08-01T13:47:01.674664Z","shell.execute_reply":"2022-08-01T13:47:01.786216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = preprocess_data(test_df, test=True)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:01.789397Z","iopub.execute_input":"2022-08-01T13:47:01.790249Z","iopub.status.idle":"2022-08-01T13:47:38.887278Z","shell.execute_reply.started":"2022-08-01T13:47:01.790172Z","shell.execute_reply":"2022-08-01T13:47:38.886166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model defining","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import confusion_matrix, roc_auc_score, classification_report\nfrom typing import Union","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:38.888619Z","iopub.execute_input":"2022-08-01T13:47:38.889032Z","iopub.status.idle":"2022-08-01T13:47:38.894744Z","shell.execute_reply.started":"2022-08-01T13:47:38.889000Z","shell.execute_reply":"2022-08-01T13:47:38.893544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(columns=['target']).astype(float)\nY = train_df['target']\nX.shape, Y.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:38.896429Z","iopub.execute_input":"2022-08-01T13:47:38.897163Z","iopub.status.idle":"2022-08-01T13:47:39.271875Z","shell.execute_reply.started":"2022-08-01T13:47:38.897130Z","shell.execute_reply":"2022-08-01T13:47:39.270585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.astype(float)\nX_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:39.273325Z","iopub.execute_input":"2022-08-01T13:47:39.273728Z","iopub.status.idle":"2022-08-01T13:47:39.515958Z","shell.execute_reply.started":"2022-08-01T13:47:39.273698Z","shell.execute_reply":"2022-08-01T13:47:39.514929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# algorithms = {\n#     'LogisticRegression': LogisticRegression(),\n#     'SVC': SVC(),\n#     'AdaBoostClassifier': AdaBoostClassifier(),\n#     'GradientBoostingClassifier': GradientBoostingClassifier(),\n#     'RandomForestClassifier': RandomForestClassifier(),\n#     'XGBClassifier': XGBClassifier(),\n#     'XGBRFClassifier': XGBRFClassifier()\n# }\n\n# models = {}\n\n# param_grid = {\n#     'LogisticRegression': {\n#         'C': [1.0, 5.0]\n#     },\n#     'SVC': {\n#         'C': [1.0, 5.0],\n#         'kernel': ['rbf', 'sigmoid']\n#     },\n#     'AdaBoostClassifier': {\n#         'n_estimators': [100, 300],\n#     },\n#     'GradientBoostingClassifier': {\n#         'n_estimators': [100, 300],\n#         'max_depth': [5, 7]\n#     },\n#     'RandomForestClassifier': {\n#         'n_estimators': [100, 300],\n#         'max_depth': [5, 7]\n#     },\n#     'XGBClassifier': {\n#         'n_estimators': [100, 300],\n#         'max_depth': [5, 7]\n#     },\n#     'XGBRFClassifier': {\n#         'n_estimators': [100, 300],\n#         'max_depth': [5, 7]\n#     }\n# }\n\n# scores = pd.DataFrame(data={'AUC': [], 'Accuracy': [], 'Precision': [], 'Recall': [], 'F1': []})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:39.517165Z","iopub.execute_input":"2022-08-01T13:47:39.518939Z","iopub.status.idle":"2022-08-01T13:47:39.526200Z","shell.execute_reply.started":"2022-08-01T13:47:39.518907Z","shell.execute_reply":"2022-08-01T13:47:39.524903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def add_score(model, index, y_pred):\n#     global scores\n#     scores.loc[index, \"AUC\"] = roc_auc_score(Y, y_pred)\n#     scores.loc[index, \"Accuracy\"] = accuracy_score(Y, y_pred)\n#     scores.loc[index, \"Precision\"] = precision_score(Y, y_pred)\n#     scores.loc[index, \"Recall\"] = recall_score(Y, y_pred)\n#     scores.loc[index, \"F1\"] = f1_score(Y, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:39.528022Z","iopub.execute_input":"2022-08-01T13:47:39.529420Z","iopub.status.idle":"2022-08-01T13:47:39.540980Z","shell.execute_reply.started":"2022-08-01T13:47:39.529382Z","shell.execute_reply":"2022-08-01T13:47:39.539808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def train(scaler: Union['none', 'Standard', 'MinMax']='none'):\n#     global models\n#     if scaler == 'none':\n#         for index, model in algorithms.items():\n#             models[index] = GridSearchCV(estimator=model, param_grid=param_grid[index], cv=10, refit='roc_auc',\n#                                          scoring=['roc_auc', 'accuracy', 'precision', 'recall', 'f1'])\n#             models[index].fit(X, Y)\n#             add_score(index)\n#             print(index)\n            \n#     elif scaler == 'Standard':\n#         for index, model in algorithms.items():\n#             models[index] = make_pipeline(StandardScaler(), GridSearchCV(estimator=model, param_grid=param_grid[index], cv=10, refit='roc_auc',\n#                             scoring=['roc_auc', 'accuracy', 'precision', 'recall', 'f1']))\n#             models[index].fit(X, Y)\n#             add_score(index)\n#             print(index)\n#     elif scaler == 'MinMax':\n#         for index, model in algorithms.items():\n#             models[index] = make_pipeline(MinMaxScaler(), GridSearchCV(estimator=model, param_grid=param_grid[index], cv=10, refit='roc_auc',\n#                             scoring=['roc_auc', 'accuracy', 'precision', 'recall', 'f1']))\n#             models[index].fit(X, Y)\n#             add_score(index)\n#             print(index)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:39.542296Z","iopub.execute_input":"2022-08-01T13:47:39.542702Z","iopub.status.idle":"2022-08-01T13:47:39.553680Z","shell.execute_reply.started":"2022-08-01T13:47:39.542669Z","shell.execute_reply":"2022-08-01T13:47:39.552689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train('Standard')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:39.555331Z","iopub.execute_input":"2022-08-01T13:47:39.555678Z","iopub.status.idle":"2022-08-01T13:47:39.568106Z","shell.execute_reply.started":"2022-08-01T13:47:39.555638Z","shell.execute_reply":"2022-08-01T13:47:39.567036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pipe = make_pipeline(StandardScaler(), LogisticRegression(max_iter=10000))\n# model_1 = GridSearchCV(estimator=pipe, param_grid={'logisticregression__C': [1.0, 3.0]}, cv=10, scoring='roc_auc')\n# model_1.fit(X, Y)\n# modek_1_Y = model_1.best_estimator_.predict(X)\n# add_score(model_1, \"LogisticRegression\", modek_1_Y)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:47:39.569255Z","iopub.execute_input":"2022-08-01T13:47:39.569745Z","iopub.status.idle":"2022-08-01T13:47:39.578903Z","shell.execute_reply.started":"2022-08-01T13:47:39.569691Z","shell.execute_reply":"2022-08-01T13:47:39.577765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lr = LogisticRegression(random_state=0, class_weight=\"balanced\", max_iter=10000, verbose=1)\nmodel_lr.fit(X, Y)\npred_lr = model_lr.predict_proba(X)[:,1]\n\nprint(classification_report(Y, model_lr.predict(X)))\nprint('AUC: ', roc_auc_score(Y, pred_lr))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-08-01T13:47:39.582399Z","iopub.execute_input":"2022-08-01T13:47:39.582835Z","iopub.status.idle":"2022-08-01T13:53:56.426016Z","shell.execute_reply.started":"2022-08-01T13:47:39.582806Z","shell.execute_reply":"2022-08-01T13:53:56.424606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_test = model_lr.predict_proba(X_test)[:,1]\nY_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:53:56.428017Z","iopub.execute_input":"2022-08-01T13:53:56.428838Z","iopub.status.idle":"2022-08-01T13:53:57.064531Z","shell.execute_reply.started":"2022-08-01T13:53:56.428784Z","shell.execute_reply":"2022-08-01T13:53:57.063264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution = pd.DataFrame({'ID_code': id_code, 'target': Y_test})\nsolution.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:53:57.066758Z","iopub.execute_input":"2022-08-01T13:53:57.067697Z","iopub.status.idle":"2022-08-01T13:53:57.090019Z","shell.execute_reply.started":"2022-08-01T13:53:57.067644Z","shell.execute_reply":"2022-08-01T13:53:57.088789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution.to_csv('submission.csv', sep=',', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T13:53:57.092245Z","iopub.execute_input":"2022-08-01T13:53:57.093419Z","iopub.status.idle":"2022-08-01T13:53:57.841976Z","shell.execute_reply.started":"2022-08-01T13:53:57.093363Z","shell.execute_reply":"2022-08-01T13:53:57.840432Z"},"trusted":true},"execution_count":null,"outputs":[]}]}