{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":3696790,"sourceType":"datasetVersion","datasetId":2211601}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX Default Prediction","metadata":{}},{"cell_type":"markdown","source":"# Prelude","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: /kaggle/input/amex-default-prediction/train_data.csv\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # plotting library\nimport seaborn as sns # library for advanced plotting\nfrom tqdm import tqdm # to make use of progress bar\nfrom IPython.display import clear_output\nimport gc # for calling python garbage collection or memory handling\n\nimport warnings\nwarnings.filterwarnings('ignore')\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading and Data Manipulation","metadata":{}},{"cell_type":"code","source":"df=pd.read_parquet('/kaggle/input/amex-parquet/train_data.parquet')\nprint(f'Loaded: {df.index.size} rows')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['S_2']=pd.to_datetime(df['S_2']) # to convert date data into datetime type\n\n\n# to remove collumns with more than 20% of the data null\ncolumns_toremove=[]\n\nfor col in df.columns:\n    null_percent=round((df[col].isna().sum()/df.index.size)*100,2)\n    clear_output(wait=True)\n    if(null_percent>=20):\n        columns_toremove.append(col)\n        print(f'Column ({col}) : Null Percent: {null_percent}% ')\n    \ndf=df.drop(columns_toremove,axis=1)\nprint(f'Deleted {len(columns_toremove)} columns')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Extraction","metadata":{}},{"cell_type":"code","source":"# Extract Numerical and Categorical Columns\ntarget_columns=[c for c in df.columns if c not in ['customer_ID','S_2','target']] # taking all columns except customer id and S_2 (contains date data)\ncat_features=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_68'] # contains all categorical features (taken from data card)\nnum_features=[c for c in target_columns if c not in cat_features] # to get all numerical features\nprint('Column Extraction Complete')\n\n#notes:\n# D_66 was removed from the cat_features list as it was removed at the null removing part of the code","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Aggregating the Numerical Features based on Customer ID\nnum_agg=df.groupby(['customer_ID'])[num_features].agg(['mean','max','last'])\nnum_agg.columns=['_'.join(x) for x in num_agg.columns]\nnum_agg=num_agg.reset_index(drop=False)\nprint('Completed Operation!')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Converting the customer_ID feature from object to string for efficient memory merge\n\n#to remove all the rows with any nulls in any of the features from the df dataframe\nprint('Deleting Rows with any null in features...(for df dataframe)')\ndf=df[df[df.columns[1:]].isnull().all(axis=1)==False]\n\ndf = df[['customer_ID','target'] + cat_features].dropna().groupby('customer_ID').last().reset_index()\ndf['customer_ID']=df['customer_ID'].astype('category')\n\ndf[cat_features]=df[cat_features].astype('category')\n\ngc.collect()\nprint(f'Rows After: {df.index.size}')\nprint(f'Columns: {df.columns.size}')\n\n#to remove all the rows with any nulls in any of the features from the num_agg dataframe\nprint('\\nDeleting Rows with any null in features...(for num_agg dataframe)')\nprint(f'Rows Before: {num_agg.index.size}')\nnum_agg['customer_ID']=num_agg['customer_ID'].astype('category') #convert customer_ID from object to string for efficnet memory use\ngc.collect()\nprint(f'Rows After: {num_agg.index.size}')\nprint(f'Columns: {num_agg.columns.size}')\n\nprint('Row clean-up complete!')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# To merge the numerical, categorical and target features into the same dataframe\ndf=num_agg.merge(df,on='customer_ID',how='inner')\nprint(f'Tables have been merged {df.index.size}')\n\n#For clearing memory\ndel num_agg\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Summary","metadata":{}},{"cell_type":"code","source":"print(f\"\"\"\nSummary (df)\nRows: {df.index.size}\nCols: {df.columns.size} \n\"\"\")\ndf.head(3)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Building","metadata":{}},{"cell_type":"markdown","source":"### Building the custom eval function\nNote: The eval function was mentioned in the competition","metadata":{}},{"cell_type":"code","source":"    #official evaluation metric function drom the compition hosters\n    def amex_metric(y_pred, train_data) -> tuple:\n        \n        y_true=train_data.get_label()\n        y_true=pd.DataFrame({'target':y_true})\n        \n        y_pred=pd.DataFrame({'prediction':y_pred})\n        \n        # print(y_true.head())\n        # print(y_pred.head())\n        # raise Exception('Force breaker')\n        \n        def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n            df = (pd.concat([y_true, y_pred], axis='columns')\n                  .sort_values('prediction', ascending=False))\n            df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n            four_pct_cutoff = int(0.04 * df['weight'].sum())\n            df['weight_cumsum'] = df['weight'].cumsum()\n            df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n            return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n            \n        def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n            df = (pd.concat([y_true, y_pred], axis='columns')\n                  .sort_values('prediction', ascending=False))\n            df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n            df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n            total_pos = (df['target'] * df['weight']).sum()\n            df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n            df['lorentz'] = df['cum_pos_found'] / total_pos\n            df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n            return df['gini'].sum()\n    \n        def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n            y_true_pred = y_true.rename(columns={'target': 'prediction'})\n            return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n    \n        g = normalized_weighted_gini(y_true, y_pred)\n        d = top_four_percent_captured(y_true, y_pred)\n    \n        return 'amex_metric',0.5 * (g + d),True\n    \n    print('Created Official Evaluation Metric Function!')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### LightGBM Model","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\n\nprint('Starting model building process')\n\nX=df.drop(columns=['customer_ID','target'])\ny=df['target']\n\nxtrain,xtest,ytrain,ytest=train_test_split(X,y,random_state=45,test_size=0.2,stratify=y)\n\ntrain_data=lgb.Dataset(xtrain,label=ytrain,categorical_feature=cat_features)\ntest_data=lgb.Dataset(xtest,label=ytest,categorical_feature=cat_features)\n\nparams={\n    'objective':'binary',\n    'metric':'None',\n    'boosting_type':'gbdt',\n    'num_leaves':31,\n    'learning_rate':0.05,\n    'feature_fraction':0.9,\n    'bagging_fraction':0.8,\n    'bagging_freq':5\n}\n\nprint('Starting Training')\nmodel=lgb.train(\n    params,\n    train_data,\n    valid_sets=[test_data],\n    num_boost_round=1000,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=50),\n        lgb.log_evaluation(50)\n    ],\n    feval=amex_metric\n)\n\nprint('Training Complete!')\n\n# model.save_model('lgb_modelAMEX.txt')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### To Save the mode (will be shown in output)","metadata":{}},{"cell_type":"code","source":"model.save_model('lgb_modelAMEX.txt')\nprint('Model saved!')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Testing Case\nNote: Make sure to \"restart and clear all outputs\" before running the below. Otherwise the RAM might bloat","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # plotting library\nimport seaborn as sns # library for advanced plotting\nfrom tqdm import tqdm # to make use of progress bar\nfrom IPython.display import clear_output\nimport gc # for calling python garbage collection or memory handling\n\nimport warnings\nwarnings.filterwarnings('ignore')\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:02:13.877552Z","iopub.execute_input":"2025-06-02T19:02:13.877984Z","iopub.status.idle":"2025-06-02T19:02:15.143609Z","shell.execute_reply.started":"2025-06-02T19:02:13.877939Z","shell.execute_reply":"2025-06-02T19:02:15.142470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\n# model=lgb.Booster(model_file='/kaggle/input/lightgbm_amex_defaulter_predictor/AE_Defaulter_Model.cbm')\nmodel=lgb.Booster(model_file='/kaggle/working/lgb_modelAMEX.txt')\n\n\n#Import the Data\ntestdata_df=pd.read_parquet('/kaggle/input/amex-parquet/test_data.parquet')\n\ntestdata_df['S_2']=pd.to_datetime(testdata_df['S_2']) # to convert date data into datetime\nprint('Operation Complete!')\n\n# to remove columns with more than 20% of the data null\ncolumns_toremove=[]\n\nfor col in testdata_df.columns:\n    null_percent=round((testdata_df[col].isna().sum()/testdata_df.index.size)*100,2)\n    clear_output(wait=True)\n    if(null_percent>=20):\n        columns_toremove.append(col)\n        print(f'Column ({col}) : Null Percent: {null_percent}% ')\n    \ntestdata_df=testdata_df.drop(columns_toremove,axis=1)\nprint(f'Deleted {len(columns_toremove)} columns')\n\n\n# Extract Numerical and Categorical Columns\ntarget_columns=[c for c in testdata_df.columns if c not in ['customer_ID','S_2']] # taking all columns except customer id and S_2 (contains date data)\ncat_features=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_68'] # contains all categorical features (taken from data card)\nnum_features=[c for c in target_columns if c not in cat_features] # to get all numerical features\nprint('Column Extraction Complete')\n\n\n#notes:\n# D_66 was removed from the cat_features list as it was removed at the null removing part of the code\n\n# Aggregating the Numerical Features based on Customer ID\nnum_agg=testdata_df.groupby(['customer_ID'])[num_features].agg(['mean','max','last'])\nnum_agg.columns=['_'.join(x) for x in num_agg.columns]\nnum_agg=num_agg.reset_index(drop=False) #fix indexing issue\nprint('Completed Operation!')\n\n\n\n\n# Converting the customer_ID feature from object to string for efficient memory merge\n\n#Additional row operations (convert to int & category type for memory efficiency, remove any rows with null and remove duplicates)\nprint('Deleting null rows in features...(for testdata_df dataframe)')\n\n# PROBLEM IS HERE\ntest_df=testdata_df[testdata_df[testdata_df.columns[1:]].isnull().all(axis=1)==False] #new code\n\ntestdata_df = testdata_df[['customer_ID'] + cat_features].dropna().groupby('customer_ID').last().reset_index()\ntestdata_df['customer_ID']=testdata_df['customer_ID'].astype('category') #convert customer_ID from object to string for efficient memory use\n# raise Exception('force breaker')\n\ntestdata_df[cat_features]=testdata_df[cat_features].astype('category')\n\ngc.collect()\nprint(f'Rows After: {testdata_df.index.size}')\nprint(f'Columns: {testdata_df.columns.size}')\n\n\n\n#to remove all the rows with any nulls in any of the features from the num_agg dataframe\nprint('\\nDeleting Rows with any null in features...(for num_agg dataframe)')\nprint(f'Rows Before: {num_agg.index.size}')\n\nnum_agg['customer_ID']=num_agg['customer_ID'].astype('category') #convert customer_ID from object to string for efficnet memory use\ngc.collect()\nprint(f'Rows After: {num_agg.index.size}')\nprint(f'Columns: {num_agg.columns.size}')\n\nprint('Row clean-up complete!')\n\n# To merge the numerical, categorical and target features into the same dataframe\ntestdata_df=num_agg.merge(testdata_df,on='customer_ID',how='inner')\n\n#For clearing memory\ndel num_agg\ngc.collect()\n\n#Running model\nX=testdata_df.drop(columns=['customer_ID'])\n\n# test_data=lgb.Dataset(X,categorical_feature=cat_features)\nprint('prediction')\ny_pred=model.predict(X,num_iteration=model.best_iteration)\nprint('Prediction Complete!')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:02:17.368739Z","iopub.execute_input":"2025-06-02T19:02:17.369269Z","iopub.status.idle":"2025-06-02T19:04:02.784074Z","shell.execute_reply.started":"2025-06-02T19:02:17.369237Z","shell.execute_reply":"2025-06-02T19:04:02.782858Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### To save submission file","metadata":{}},{"cell_type":"code","source":"\nsubmission_data=pd.DataFrame({'customer_ID':testdata_df['customer_ID'],'prediction':np.round(y_pred,2)})\nsubmission_data.to_csv('submission.csv',index=False)\nprint('Saved submission.csv file.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:24:00.152570Z","iopub.execute_input":"2025-06-02T19:24:00.152916Z","iopub.status.idle":"2025-06-02T19:24:03.898973Z","shell.execute_reply.started":"2025-06-02T19:24:00.152890Z","shell.execute_reply":"2025-06-02T19:24:03.896532Z"}},"outputs":[],"execution_count":null}]}