{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, GridSearchCV\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay, accuracy_score\nfrom sklearn.ensemble import RandomForestClassifier\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n# Models\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-31T12:05:14.256679Z","iopub.execute_input":"2022-05-31T12:05:14.257268Z","iopub.status.idle":"2022-05-31T12:05:18.087622Z","shell.execute_reply.started":"2022-05-31T12:05:14.257155Z","shell.execute_reply":"2022-05-31T12:05:18.086288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(\"../input/amex-parquet/train_data.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:05:18.090051Z","iopub.execute_input":"2022-05-31T12:05:18.091005Z","iopub.status.idle":"2022-05-31T12:05:55.555971Z","shell.execute_reply.started":"2022-05-31T12:05:18.090948Z","shell.execute_reply":"2022-05-31T12:05:55.555105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:05:55.557558Z","iopub.execute_input":"2022-05-31T12:05:55.558015Z","iopub.status.idle":"2022-05-31T12:05:55.566016Z","shell.execute_reply.started":"2022-05-31T12:05:55.557966Z","shell.execute_reply":"2022-05-31T12:05:55.565173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:05:55.568175Z","iopub.execute_input":"2022-05-31T12:05:55.568716Z","iopub.status.idle":"2022-05-31T12:06:42.394906Z","shell.execute_reply.started":"2022-05-31T12:05:55.568680Z","shell.execute_reply":"2022-05-31T12:06:42.393549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Missing Column Analysis","metadata":{}},{"cell_type":"code","source":"missing_df = train.isnull().sum().reset_index()\nmissing_df = missing_df.rename(columns={\"index\":\"columns\",0:\"missing_value\"})\nmissing_df = missing_df.query(\"missing_value>0\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:42.396326Z","iopub.execute_input":"2022-05-31T12:06:42.396858Z","iopub.status.idle":"2022-05-31T12:06:46.632772Z","shell.execute_reply.started":"2022-05-31T12:06:42.396801Z","shell.execute_reply":"2022-05-31T12:06:46.631739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:46.634708Z","iopub.execute_input":"2022-05-31T12:06:46.635207Z","iopub.status.idle":"2022-05-31T12:06:46.653340Z","shell.execute_reply.started":"2022-05-31T12:06:46.635160Z","shell.execute_reply":"2022-05-31T12:06:46.652295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(missing_df.describe())\nprint(missing_df.max())\nprint(missing_df.min())","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:46.654713Z","iopub.execute_input":"2022-05-31T12:06:46.655741Z","iopub.status.idle":"2022-05-31T12:06:46.833358Z","shell.execute_reply.started":"2022-05-31T12:06:46.655695Z","shell.execute_reply":"2022-05-31T12:06:46.832487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(y=missing_df[\"columns\"],x=missing_df[\"missing_value\"])\nfig.update_layout(showlegend=False, \n                  title_text=\"Column Wise Null Value Distribution\", \n                  title_x=0.5,\n                  xaxis_title=\"Missing Value Count\",\n                  yaxis_title=\"Column Name\")\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:46.834665Z","iopub.execute_input":"2022-05-31T12:06:46.835582Z","iopub.status.idle":"2022-05-31T12:06:47.967981Z","shell.execute_reply.started":"2022-05-31T12:06:46.835545Z","shell.execute_reply":"2022-05-31T12:06:47.966948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations:\n\n* We have 122 NULL value columns in our train dataset. Almost 63% of columns in our dataset have NULL values.\n\n* Column B_13 has the least number of null values (1 NULL value)\n\n* Column S_9 has the maximum number of null values (5527586)\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# Different feature types","metadata":{}},{"cell_type":"markdown","source":"Features are anonymized and normalized, and fall into the following general categories:\n\nD_* = Delinquency variables\n\nS_* = Spend variables\n\nP_* = Payment variables\n\nB_* = Balance variables\n\nR_* = Risk variables","metadata":{}},{"cell_type":"code","source":"d_feats = [c for c in train.columns if c.startswith('D_')]\ns_feats = [c for c in train.columns if c.startswith('S_')]\np_feats = [c for c in train.columns if c.startswith('P_')]\nb_feats = [c for c in train.columns if c.startswith('B_')]\nr_feats = [c for c in train.columns if c.startswith('R_')]","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:47.969236Z","iopub.execute_input":"2022-05-31T12:06:47.969617Z","iopub.status.idle":"2022-05-31T12:06:47.977729Z","shell.execute_reply.started":"2022-05-31T12:06:47.969583Z","shell.execute_reply":"2022-05-31T12:06:47.976728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes = train.dtypes.reset_index()\ndtypes = dtypes.rename(columns={\"index\":\"Column_name\",0:\"dtype_name\"})\ndtypes = dtypes.groupby(by=[\"dtype_name\"]).size().reset_index(name=\"counts\")\ndtypes","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:47.980483Z","iopub.execute_input":"2022-05-31T12:06:47.981241Z","iopub.status.idle":"2022-05-31T12:06:48.008371Z","shell.execute_reply.started":"2022-05-31T12:06:47.981191Z","shell.execute_reply":"2022-05-31T12:06:48.007504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 4 object columns, 185 float and 2 integer columns ","metadata":{}},{"cell_type":"markdown","source":"# Target Analysis","metadata":{}},{"cell_type":"code","source":"target_ana = train.groupby(by=[\"target\"]).size().reset_index(name=\"counts\")\nfig = px.bar(data_frame=target_ana,x=\"target\",y=\"counts\",color = 'target')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:48.009622Z","iopub.execute_input":"2022-05-31T12:06:48.010095Z","iopub.status.idle":"2022-05-31T12:06:48.224049Z","shell.execute_reply.started":"2022-05-31T12:06:48.010061Z","shell.execute_reply":"2022-05-31T12:06:48.222776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations\n\nOur training dataset has unequal target distribution.\n\nWe have lesser number of defaulters than the ones who do not default, which does make sense considering a real world scenario.\n\nUsing stratified k fold could be a strategy we employ while training the model.","metadata":{}},{"cell_type":"markdown","source":"# Card statements per user","metadata":{}},{"cell_type":"code","source":"target_cust = train.groupby(by=[\"customer_ID\"]).size().reset_index(name=\"counts\")\ntarget_cust = target_cust.groupby(by=[\"counts\"]).size().reset_index(name=\"number_per_count\")\nfig = px.pie(target_cust,names=\"counts\",values=\"number_per_count\",title=\"NUmber of statements per customer id\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:48.225633Z","iopub.execute_input":"2022-05-31T12:06:48.226121Z","iopub.status.idle":"2022-05-31T12:06:49.858576Z","shell.execute_reply.started":"2022-05-31T12:06:48.226073Z","shell.execute_reply":"2022-05-31T12:06:49.857716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations\n\nMost of the customers almost 84% have 13 card statements (almost over a year)\n\nBut for some customers we have statements for only a single month.\n\nIn that case we would need to have a strategy that takes this into consideration while modelling.","metadata":{}},{"cell_type":"markdown","source":"# Visualising categorical object features","metadata":{}},{"cell_type":"code","source":"train.select_dtypes(include=['object'])","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:49.859979Z","iopub.execute_input":"2022-05-31T12:06:49.860327Z","iopub.status.idle":"2022-05-31T12:06:50.073980Z","shell.execute_reply.started":"2022-05-31T12:06:49.860296Z","shell.execute_reply":"2022-05-31T12:06:50.073187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_by_date = train.groupby(by=[\"S_2\"]).size().reset_index(name=\"counts\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:50.075011Z","iopub.execute_input":"2022-05-31T12:06:50.075830Z","iopub.status.idle":"2022-05-31T12:06:50.666086Z","shell.execute_reply.started":"2022-05-31T12:06:50.075794Z","shell.execute_reply":"2022-05-31T12:06:50.664758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.line(target_by_date,x=\"S_2\",y=\"counts\",title=\"Number of statements generated by date\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:50.667731Z","iopub.execute_input":"2022-05-31T12:06:50.668107Z","iopub.status.idle":"2022-05-31T12:06:50.763253Z","shell.execute_reply.started":"2022-05-31T12:06:50.668074Z","shell.execute_reply":"2022-05-31T12:06:50.762483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations\n\nOn analysing the pattern from months March to May 2017, I could see that in a week most statements are generated on a  Saturday and there is a considerable drop in the number of statements generated on Sundays, after which throughout the week the number of statements show an upward trend maxing on Saturday.","metadata":{}},{"cell_type":"code","source":"target_by_d63 = train.groupby(by=[\"D_63\"]).size().reset_index(name=\"counts\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:50.764383Z","iopub.execute_input":"2022-05-31T12:06:50.764838Z","iopub.status.idle":"2022-05-31T12:06:51.304587Z","shell.execute_reply.started":"2022-05-31T12:06:50.764808Z","shell.execute_reply":"2022-05-31T12:06:51.303510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_by_d63","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:51.306180Z","iopub.execute_input":"2022-05-31T12:06:51.307188Z","iopub.status.idle":"2022-05-31T12:06:51.318652Z","shell.execute_reply.started":"2022-05-31T12:06:51.307141Z","shell.execute_reply":"2022-05-31T12:06:51.317414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.bar(target_by_d63,x=\"D_63\",y=\"counts\",color=\"D_63\",title=\"Distribution of D_63\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:51.320047Z","iopub.execute_input":"2022-05-31T12:06:51.320461Z","iopub.status.idle":"2022-05-31T12:06:51.432271Z","shell.execute_reply.started":"2022-05-31T12:06:51.320396Z","shell.execute_reply":"2022-05-31T12:06:51.431511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations:\n\nDelinquency means minor crime, these variables seem to be depicting some sort of negligience by the credit card holder.\n\nCO is the category with the most number of counts.","metadata":{}},{"cell_type":"code","source":"target_by_d64 = train.groupby(by=[\"D_64\"]).size().reset_index(name=\"counts\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:51.433538Z","iopub.execute_input":"2022-05-31T12:06:51.434067Z","iopub.status.idle":"2022-05-31T12:06:52.020689Z","shell.execute_reply.started":"2022-05-31T12:06:51.434028Z","shell.execute_reply":"2022-05-31T12:06:52.019266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.bar(target_by_d64,x=\"D_64\",y=\"counts\",color=\"D_64\",title=\"Distribution of D_64\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:52.022483Z","iopub.execute_input":"2022-05-31T12:06:52.023010Z","iopub.status.idle":"2022-05-31T12:06:52.112427Z","shell.execute_reply.started":"2022-05-31T12:06:52.022946Z","shell.execute_reply":"2022-05-31T12:06:52.111250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualising integer features","metadata":{}},{"cell_type":"code","source":"train.select_dtypes(include=['int'])","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:52.114047Z","iopub.execute_input":"2022-05-31T12:06:52.114680Z","iopub.status.idle":"2022-05-31T12:06:52.163540Z","shell.execute_reply.started":"2022-05-31T12:06:52.114638Z","shell.execute_reply":"2022-05-31T12:06:52.162472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_by_b31 = train.groupby(by=[\"B_31\"]).size().reset_index(name=\"counts\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:52.164836Z","iopub.execute_input":"2022-05-31T12:06:52.165189Z","iopub.status.idle":"2022-05-31T12:06:52.300878Z","shell.execute_reply.started":"2022-05-31T12:06:52.165158Z","shell.execute_reply":"2022-05-31T12:06:52.299914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.bar(target_by_b31,x=\"B_31\",y=\"counts\",color=\"B_31\",title=\"Distribution of B_31\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:52.301992Z","iopub.execute_input":"2022-05-31T12:06:52.302759Z","iopub.status.idle":"2022-05-31T12:06:52.379326Z","shell.execute_reply.started":"2022-05-31T12:06:52.302721Z","shell.execute_reply":"2022-05-31T12:06:52.378250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation of variables with the target feature","metadata":{}},{"cell_type":"code","source":"corr = train.sample(frac=0.1, random_state=42).corr()\nmask = np.triu(np.ones_like(corr, dtype=np.bool))\nplt.figure(figsize=(11, 9))\nsns.heatmap(corr,mask=mask)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:06:52.380703Z","iopub.execute_input":"2022-05-31T12:06:52.381045Z","iopub.status.idle":"2022-05-31T12:07:43.529375Z","shell.execute_reply.started":"2022-05-31T12:06:52.381015Z","shell.execute_reply":"2022-05-31T12:07:43.528492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Some features are highly correlated.**","metadata":{}},{"cell_type":"code","source":"#Code reference: https://stackoverflow.com/questions/29294983/how-to-calculate-correlation-between-all-columns-and-remove-highly-correlated-on\n\niters = range(len(corr.columns) - 1)\ndrop_cols = []\nthreshold = 0.9\nprint(\"Highly correlated features with their correlation values\")\n    # Iterate through the correlation matrix and compare correlations\nfor i in iters:\n        for j in range(i+1):\n            item = corr.iloc[j:(j+1), (i+1):(i+2)]\n            col = item.columns\n            row = item.index\n            val = abs(item.values)\n\n            # If correlation exceeds the threshold\n            if val >= threshold:\n                # Print the correlated features and the correlation value\n                print(col.values[0], \"|\", row.values[0], \"|\", round(val[0][0], 2))\n                drop_cols.append(col.values[0])\n","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:43.530489Z","iopub.execute_input":"2022-05-31T12:07:43.531519Z","iopub.status.idle":"2022-05-31T12:07:45.042780Z","shell.execute_reply.started":"2022-05-31T12:07:43.531468Z","shell.execute_reply":"2022-05-31T12:07:45.041587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's remove the highly correlated features**","metadata":{}},{"cell_type":"code","source":"# Drop one of each pair of correlated columns\ndrops = set(drop_cols)\n#train = train.drop(columns=drops)\n","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:45.044575Z","iopub.execute_input":"2022-05-31T12:07:45.045308Z","iopub.status.idle":"2022-05-31T12:07:45.050780Z","shell.execute_reply.started":"2022-05-31T12:07:45.045257Z","shell.execute_reply":"2022-05-31T12:07:45.049362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:45.052272Z","iopub.execute_input":"2022-05-31T12:07:45.052866Z","iopub.status.idle":"2022-05-31T12:07:45.080425Z","shell.execute_reply.started":"2022-05-31T12:07:45.052819Z","shell.execute_reply":"2022-05-31T12:07:45.079627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling\n\n**Reference:https://www.kaggle.com/code/munumbutt/simple-lgbm-starter**","metadata":{}},{"cell_type":"code","source":"%%time\n# Keep the last statement month per customer\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327094\ntrain =  (train\n            .groupby('customer_ID')\n            .tail(1)\n            .set_index('customer_ID', drop=True)\n            .sort_index()\n            .drop(['S_2'], axis='columns'))","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:45.083863Z","iopub.execute_input":"2022-05-31T12:07:45.084332Z","iopub.status.idle":"2022-05-31T12:07:48.360456Z","shell.execute_reply.started":"2022-05-31T12:07:45.084300Z","shell.execute_reply":"2022-05-31T12:07:48.359446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:48.361962Z","iopub.execute_input":"2022-05-31T12:07:48.362514Z","iopub.status.idle":"2022-05-31T12:07:48.368234Z","shell.execute_reply.started":"2022-05-31T12:07:48.362477Z","shell.execute_reply":"2022-05-31T12:07:48.367205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_cols = train.columns.to_list()\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nnum_cols = [col for col in total_cols if col not in cat_cols + [\"target\"]]","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:48.369436Z","iopub.execute_input":"2022-05-31T12:07:48.369852Z","iopub.status.idle":"2022-05-31T12:07:48.380726Z","shell.execute_reply.started":"2022-05-31T12:07:48.369822Z","shell.execute_reply":"2022-05-31T12:07:48.379638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train[cat_cols+num_cols]\ny = train[\"target\"]\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:48.382851Z","iopub.execute_input":"2022-05-31T12:07:48.383971Z","iopub.status.idle":"2022-05-31T12:07:48.523069Z","shell.execute_reply.started":"2022-05-31T12:07:48.383774Z","shell.execute_reply":"2022-05-31T12:07:48.521842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nenc = OrdinalEncoder(handle_unknown='use_encoded_value',unknown_value=np.nan)\nX[cat_cols] = enc.fit_transform(X[cat_cols])","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:48.524275Z","iopub.execute_input":"2022-05-31T12:07:48.524729Z","iopub.status.idle":"2022-05-31T12:07:50.612029Z","shell.execute_reply.started":"2022-05-31T12:07:48.524682Z","shell.execute_reply":"2022-05-31T12:07:50.610946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:50.613492Z","iopub.execute_input":"2022-05-31T12:07:50.614616Z","iopub.status.idle":"2022-05-31T12:07:51.682193Z","shell.execute_reply.started":"2022-05-31T12:07:50.614561Z","shell.execute_reply":"2022-05-31T12:07:51.681116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = LGBMClassifier(\n    n_estimators=50000,\n    random_state=72,\n    extra_trees=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:51.683497Z","iopub.execute_input":"2022-05-31T12:07:51.683951Z","iopub.status.idle":"2022-05-31T12:07:51.689520Z","shell.execute_reply.started":"2022-05-31T12:07:51.683910Z","shell.execute_reply":"2022-05-31T12:07:51.688498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nclf.fit(\n    X_train, y_train, \n    eval_set=[(X_test,y_test)],\n    callbacks=[early_stopping(50), log_evaluation(0)]\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:07:51.691100Z","iopub.execute_input":"2022-05-31T12:07:51.691612Z","iopub.status.idle":"2022-05-31T12:08:50.874626Z","shell.execute_reply.started":"2022-05-31T12:07:51.691562Z","shell.execute_reply":"2022-05-31T12:08:50.873725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ndel train, X, y, X_test, X_train, y_train, y_test\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:46:12.379392Z","iopub.execute_input":"2022-05-31T12:46:12.381671Z","iopub.status.idle":"2022-05-31T12:46:12.881206Z","shell.execute_reply.started":"2022-05-31T12:46:12.381607Z","shell.execute_reply":"2022-05-31T12:46:12.880370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del test","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:48:29.388770Z","iopub.execute_input":"2022-05-31T12:48:29.389299Z","iopub.status.idle":"2022-05-31T12:48:29.409007Z","shell.execute_reply.started":"2022-05-31T12:48:29.389258Z","shell.execute_reply":"2022-05-31T12:48:29.407472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_feather(\"../input/amexfeather/test_data.ftr\")","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:48:32.121487Z","iopub.execute_input":"2022-05-31T12:48:32.121882Z","iopub.status.idle":"2022-05-31T12:48:47.259176Z","shell.execute_reply.started":"2022-05-31T12:48:32.121851Z","shell.execute_reply":"2022-05-31T12:48:47.258176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test =  (\n    test\n    .groupby('customer_ID')\n    .tail(1)\n    .set_index('customer_ID', drop=True)\n    .sort_index()\n    .drop(['S_2'], axis='columns')\n)\ntest[cat_cols] = enc.transform(test[cat_cols])\ntest[\"prediction\"] = clf.predict_proba(test[cat_cols + num_cols])[:,1]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:49:11.618019Z","iopub.execute_input":"2022-05-31T12:49:11.618484Z","iopub.status.idle":"2022-05-31T12:49:32.443478Z","shell.execute_reply.started":"2022-05-31T12:49:11.618449Z","shell.execute_reply":"2022-05-31T12:49:32.442646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pd.DataFrame()\n#pred[\"customer_ID\"] = test[\"customer_ID\"]\npred[\"prediction\"] = test[\"prediction\"]","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:51:48.394324Z","iopub.execute_input":"2022-05-31T12:51:48.394741Z","iopub.status.idle":"2022-05-31T12:51:48.600213Z","shell.execute_reply.started":"2022-05-31T12:51:48.394707Z","shell.execute_reply":"2022-05-31T12:51:48.598648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"].to_csv(\"submission.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T12:53:57.450395Z","iopub.execute_input":"2022-05-31T12:53:57.450891Z","iopub.status.idle":"2022-05-31T12:54:02.275888Z","shell.execute_reply.started":"2022-05-31T12:53:57.450854Z","shell.execute_reply":"2022-05-31T12:54:02.274469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Work in Progress","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}