{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Note, the results I submitted were not from this notebook. This was my highest scoring notebook although the scores are very similar (0.79813 vs 0.79853) The difference between the two is that the submitted notebook does not use Imputation or Scaling while this one does.\n\nThe notebook:\n1. Creates two simple Light GBM and CatBoost models.\n2. Gets their prediction probabilities for defaulting (target = 1) and averages them for the final submission.\n\nI used the denoised [dataset](https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format) by ***@raddar*** with two transactions per customer as suggested in this notebook, [notebook](https://www.kaggle.com/code/junjitakeshima/amex-try-to-improve-lgbm-starter-eng) as I found that adding two transactions gave the highest score with the lowest input rows (tested with Light GBM and CatBoost).","metadata":{}},{"cell_type":"code","source":"import gc\nimport joblib\nimport pandas as pd\nimport sys\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-20T12:40:52.886946Z","iopub.execute_input":"2022-08-20T12:40:52.887303Z","iopub.status.idle":"2022-08-20T12:40:58.444869Z","shell.execute_reply.started":"2022-08-20T12:40:52.887274Z","shell.execute_reply":"2022-08-20T12:40:58.443868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Training**","metadata":{}},{"cell_type":"code","source":"num_transactions = 2\n\ntrain_data = pd.read_parquet('/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet').groupby('customer_ID').tail(num_transactions).set_index('customer_ID', drop=True).sort_index()\ntrain_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv').set_index('customer_ID', drop=True).sort_index()\ntrain_data = pd.merge(train_data, train_labels, left_index=True, right_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T12:07:11.671649Z","iopub.execute_input":"2022-08-20T12:07:11.672289Z","iopub.status.idle":"2022-08-20T12:08:01.862245Z","shell.execute_reply.started":"2022-08-20T12:07:11.67225Z","shell.execute_reply":"2022-08-20T12:08:01.861167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = ['S_2']  # list of columns to drop\ntrain_data.drop(drop_cols, inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T12:08:01.864387Z","iopub.execute_input":"2022-08-20T12:08:01.864795Z","iopub.status.idle":"2022-08-20T12:08:02.263931Z","shell.execute_reply.started":"2022-08-20T12:08:01.864757Z","shell.execute_reply":"2022-08-20T12:08:02.262885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = train_data['target']\ntrain_X = train_data.drop('target', axis=1)\n\ncol_names = train_X.columns\n\nimputer = SimpleImputer()\ntrain_X = pd.DataFrame(imputer.fit_transform(train_X))\ntrain_X.columns = col_names\n\nscaler = StandardScaler()\ntrain_X = pd.DataFrame(scaler.fit_transform(train_X), index=train_X.index, columns=train_X.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T12:08:02.265502Z","iopub.execute_input":"2022-08-20T12:08:02.265921Z","iopub.status.idle":"2022-08-20T12:08:05.43733Z","shell.execute_reply.started":"2022-08-20T12:08:02.265884Z","shell.execute_reply":"2022-08-20T12:08:05.436267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_data, train_labels\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T12:08:49.396268Z","iopub.execute_input":"2022-08-20T12:08:49.396657Z","iopub.status.idle":"2022-08-20T12:08:51.743919Z","shell.execute_reply.started":"2022-08-20T12:08:49.396626Z","shell.execute_reply":"2022-08-20T12:08:51.742756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm = LGBMClassifier(n_estimators=300, random_seed=42)\nlgbm.fit(train_X, train_y)\n\ncbc = CatBoostClassifier(silent=True, random_seed=42)\ncbc.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T12:09:08.161784Z","iopub.execute_input":"2022-08-20T12:09:08.162464Z","iopub.status.idle":"2022-08-20T12:15:58.654962Z","shell.execute_reply.started":"2022-08-20T12:09:08.162429Z","shell.execute_reply":"2022-08-20T12:15:58.653455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Prediction**","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_parquet('/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet').groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()\ncust_id = test_data.index\ntest_data.drop(drop_cols, inplace=True, axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_names = test_data.columns\ntest_data = pd.DataFrame(imputer.transform(test_data))\ntest_data.columns = col_names\n\ntest_data = pd.DataFrame(scaler.transform(test_data), index=test_data.index, columns=test_data.columns)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_probs_lgbm = lgbm.predict_proba(test_data)\npreds_probs_cbc = cbc.predict_proba(test_data)\n\noutput_lgbm = pd.DataFrame({'customer_ID': cust_id, 'prediction': pd.DataFrame(preds_probs_lgbm, dtype='float64')[1]})\noutput_cbc = pd.DataFrame({'customer_ID': cust_id, 'prediction': pd.DataFrame(preds_probs_cbc, dtype='float64')[1]})\n\noutput_lgbm.to_csv('/kaggle/working/lgbm_submission.csv', index=False)\noutput_cbc.to_csv('/kaggle/working/cbc_submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined = output_cbc.merge(output_lgbm, left_on='customer_ID', right_on='customer_ID', suffixes=('_cbc', '_lgbm'))\ncombined['Average'] = combined.mean(numeric_only=True, axis=1)\n\ncombined[['customer_ID', 'Average']].copy().rename({'Average': 'prediction'}, axis=1).to_csv('/kaggle/working/avg_submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}