{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Initial notebook with some simple ideas","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LogisticRegression, LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split\n%matplotlib inline\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:54:24.346665Z","iopub.execute_input":"2022-05-25T22:54:24.347364Z","iopub.status.idle":"2022-05-25T22:54:24.360133Z","shell.execute_reply.started":"2022-05-25T22:54:24.347302Z","shell.execute_reply":"2022-05-25T22:54:24.358432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reference https://www.kaggle.com/competitions/amex-default-prediction/discussion/327162\ndef amex_metric(y_true: pd.Series, y_pred: pd.Series) -> float:\n\n    def top_four_percent_captured(df) -> float:\n        \n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(df) -> float:\n        \n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true, df) -> float:\n        y_true_pred = y_true.rename('prediction')\n        true_df = pd.concat([y_true, y_true_pred], axis='columns').sort_values('prediction', ascending=False)\n        return weighted_gini(df) / weighted_gini(true_df)\n\n    df = pd.DataFrame({'target': y_true, 'prediction': y_pred}).sort_values('prediction', ascending=False)\n    g = normalized_weighted_gini(y_true, df.copy())\n    d = top_four_percent_captured(df.copy())\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:54:24.738055Z","iopub.execute_input":"2022-05-25T22:54:24.738516Z","iopub.status.idle":"2022-05-25T22:54:24.754194Z","shell.execute_reply.started":"2022-05-25T22:54:24.738471Z","shell.execute_reply":"2022-05-25T22:54:24.752702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\n    input_path / 'train_data.csv',\n    index_col='customer_ID',\n    nrows=1_000_000)\n\ntrain_labels = pd.read_csv(input_path / 'train_labels.csv', index_col='customer_ID', nrows=1_000_000)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:54:25.023629Z","iopub.execute_input":"2022-05-25T22:54:25.024115Z","iopub.status.idle":"2022-05-25T22:55:43.717398Z","shell.execute_reply.started":"2022-05-25T22:54:25.024078Z","shell.execute_reply":"2022-05-25T22:55:43.716114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get only the labels of the customers of the first 1M rows of the train data\ntrain_labels = train_labels[train_labels.index.isin(train_data.index)]","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:43.720645Z","iopub.execute_input":"2022-05-25T22:55:43.721252Z","iopub.status.idle":"2022-05-25T22:55:43.837326Z","shell.execute_reply.started":"2022-05-25T22:55:43.721189Z","shell.execute_reply":"2022-05-25T22:55:43.836288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What is the single best predictor?","metadata":{}},{"cell_type":"code","source":"# We are going to use only the last month\nlast_month_train_data = train_data.groupby('customer_ID').tail(1)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:43.838591Z","iopub.execute_input":"2022-05-25T22:55:43.839667Z","iopub.status.idle":"2022-05-25T22:55:44.330742Z","shell.execute_reply.started":"2022-05-25T22:55:43.839617Z","shell.execute_reply":"2022-05-25T22:55:44.329708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data = last_month_train_data.merge(train_labels, on='customer_ID', how='inner', validate='one_to_one')","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:44.332645Z","iopub.execute_input":"2022-05-25T22:55:44.333011Z","iopub.status.idle":"2022-05-25T22:55:44.973249Z","shell.execute_reply.started":"2022-05-25T22:55:44.332979Z","shell.execute_reply":"2022-05-25T22:55:44.97204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data.corr()['target'].abs().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:44.97503Z","iopub.execute_input":"2022-05-25T22:55:44.975582Z","iopub.status.idle":"2022-05-25T22:55:52.49233Z","shell.execute_reply.started":"2022-05-25T22:55:44.975529Z","shell.execute_reply":"2022-05-25T22:55:52.491155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyzing P_2","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(final_data.loc[final_data['target'].eq(0), 'P_2'])\nsns.kdeplot(final_data.loc[final_data['target'].eq(1), 'P_2'])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:52.493818Z","iopub.execute_input":"2022-05-25T22:55:52.494213Z","iopub.status.idle":"2022-05-25T22:55:53.366759Z","shell.execute_reply.started":"2022-05-25T22:55:52.494178Z","shell.execute_reply":"2022-05-25T22:55:53.365678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can use a simple heuristic and say that someone who has less than 0.5 P_2 has target = 0, how this algorithm will work?","metadata":{}},{"cell_type":"code","source":"final_data['target_test_P_2'] = final_data['P_2'].lt(0.5)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:53.368258Z","iopub.execute_input":"2022-05-25T22:55:53.369367Z","iopub.status.idle":"2022-05-25T22:55:53.376878Z","shell.execute_reply.started":"2022-05-25T22:55:53.369313Z","shell.execute_reply":"2022-05-25T22:55:53.375941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(final_data['target'],\n            final_data['target_test_P_2'])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:53.379386Z","iopub.execute_input":"2022-05-25T22:55:53.379912Z","iopub.status.idle":"2022-05-25T22:55:53.648415Z","shell.execute_reply.started":"2022-05-25T22:55:53.379861Z","shell.execute_reply":"2022-05-25T22:55:53.646877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyzing D_48","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(final_data.loc[final_data['target'].eq(0), 'D_48'])\nsns.kdeplot(final_data.loc[final_data['target'].eq(1), 'D_48'])\n\nplt.legend(['target 0', 'target 1'])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:53.651103Z","iopub.execute_input":"2022-05-25T22:55:53.651669Z","iopub.status.idle":"2022-05-25T22:55:54.491676Z","shell.execute_reply.started":"2022-05-25T22:55:53.651618Z","shell.execute_reply":"2022-05-25T22:55:54.490835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are just some values bigger than 1, very strange\nfinal_data['D_48'].ge(1).sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:54.496015Z","iopub.execute_input":"2022-05-25T22:55:54.496833Z","iopub.status.idle":"2022-05-25T22:55:54.505907Z","shell.execute_reply.started":"2022-05-25T22:55:54.496774Z","shell.execute_reply":"2022-05-25T22:55:54.504884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Redo chart without outliers\ntmp = final_data.loc[final_data['D_48'].le(1)]\nsns.kdeplot(tmp.loc[tmp['target'].eq(0), 'D_48'])\nsns.kdeplot(tmp.loc[tmp['target'].eq(1), 'D_48'])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:54.507592Z","iopub.execute_input":"2022-05-25T22:55:54.508292Z","iopub.status.idle":"2022-05-25T22:55:55.333381Z","shell.execute_reply.started":"2022-05-25T22:55:54.508242Z","shell.execute_reply":"2022-05-25T22:55:55.332322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try to use another simple heuristic, if a value has D_48 < 0.45 then it is target = 0","metadata":{}},{"cell_type":"code","source":"final_data['target_test_D_48'] = final_data['D_48'].ge(0.45)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:55.335169Z","iopub.execute_input":"2022-05-25T22:55:55.335885Z","iopub.status.idle":"2022-05-25T22:55:55.343157Z","shell.execute_reply.started":"2022-05-25T22:55:55.335835Z","shell.execute_reply":"2022-05-25T22:55:55.342366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(final_data['target'],\n            final_data['target_test_D_48'])","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-25T22:55:55.344979Z","iopub.execute_input":"2022-05-25T22:55:55.345771Z","iopub.status.idle":"2022-05-25T22:55:55.561275Z","shell.execute_reply.started":"2022-05-25T22:55:55.345693Z","shell.execute_reply":"2022-05-25T22:55:55.560207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Combining P_2 and D_48","metadata":{}},{"cell_type":"code","source":"tmp = final_data.loc[final_data['D_48'].le(1)]\nsns.kdeplot(x=tmp['P_2'], y=tmp['D_48'], hue=tmp['target'])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T22:55:55.563114Z","iopub.execute_input":"2022-05-25T22:55:55.563773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see there is a clear separation in the data, let's try to create a simple logistic regression model using just P_2 and D_48 on the last month","metadata":{}},{"cell_type":"markdown","source":"## Initial model","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\n    input_path / 'train_data.csv',\n    usecols=['P_2', 'D_48', 'customer_ID'])\n\ntrain_labels = pd.read_csv(input_path / 'train_labels.csv', index_col='customer_ID')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_train_data = train_data.groupby('customer_ID').tail(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_train_data = last_month_train_data.merge(train_labels, on='customer_ID', how='inner',\n                                                    validate='one_to_one')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # over sample\n# max_size = last_month_train_data['target'].value_counts().max()\n# lst = [last_month_train_data]\n# for class_index, group in last_month_train_data.groupby('target'):\n#     lst.append(group.sample(max_size-len(group), replace=True))\n# last_month_train_data_over_sampled = pd.concat(lst)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = LinearRegression()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr.fit(last_month_train_data[['P_2', 'D_48']].fillna(-999), \n       last_month_train_data['target'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\n    input_path / 'test_data.csv',\n    usecols=['P_2', 'D_48', 'customer_ID'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data = test_data.groupby('customer_ID').tail(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = lr.predict(last_month_test_data[['P_2', 'D_48']].fillna(-999))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data['prediction'] = y_pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data['prediction'].mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data[['customer_ID', 'prediction']].to_csv('inicial_submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf.fit(last_month_train_data[['P_2', 'D_48']].fillna(-999), \n       last_month_train_data['target'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = rf.predict(last_month_test_data[['P_2', 'D_48']].fillna(-999))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data['prediction'] = y_pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data['prediction'].mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data[['customer_ID', 'prediction']].to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}