{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n!pip install polars\n\nimport polars as pl\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nOUTPUT_FOLDER = '/kaggle/working/'\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-16T18:46:57.115270Z","iopub.execute_input":"2023-04-16T18:46:57.116146Z","iopub.status.idle":"2023-04-16T18:47:08.806041Z","shell.execute_reply.started":"2023-04-16T18:46:57.116098Z","shell.execute_reply":"2023-04-16T18:47:08.805094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-16T18:47:08.807191Z","iopub.execute_input":"2023-04-16T18:47:08.808023Z","iopub.status.idle":"2023-04-16T18:47:08.820441Z","shell.execute_reply.started":"2023-04-16T18:47:08.807992Z","shell.execute_reply":"2023-04-16T18:47:08.819437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_FILE_PATH = '../input/amex-default-prediction/train_data.csv'\nTRAIN_LABEL_PATH = '../input/amex-default-prediction/train_labels.csv'\nTEST_FILE_PATH = '../input/amex-default-prediction/test_data.csv'","metadata":{"execution":{"iopub.status.busy":"2023-04-16T18:47:08.822399Z","iopub.execute_input":"2023-04-16T18:47:08.822741Z","iopub.status.idle":"2023-04-16T18:47:08.833494Z","shell.execute_reply.started":"2023-04-16T18:47:08.822687Z","shell.execute_reply":"2023-04-16T18:47:08.832381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = pl.read_csv(TRAIN_FILE_PATH,rechunk=False).to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T18:47:08.834568Z","iopub.execute_input":"2023-04-16T18:47:08.834839Z","iopub.status.idle":"2023-04-16T18:48:30.010321Z","shell.execute_reply.started":"2023-04-16T18:47:08.834812Z","shell.execute_reply":"2023-04-16T18:48:30.008012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label = pd.read_csv(TRAIN_LABEL_PATH, nrows = 10**4, index_col = 'customer_ID')\n#df = pd.read_csv(TRAIN_FILE_PATH, nrows = 10**4, index_col = ['customer_ID','S_2'])","metadata":{"execution":{"iopub.status.busy":"2023-04-16T18:48:30.014152Z","iopub.execute_input":"2023-04-16T18:48:30.015894Z","iopub.status.idle":"2023-04-16T18:48:30.022712Z","shell.execute_reply.started":"2023-04-16T18:48:30.015822Z","shell.execute_reply":"2023-04-16T18:48:30.020925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\n- D_* = Delinquency variables\n- S_* = Spend variables\n- P_* = Payment variables\n- B_* = Balance variables\n- R_* = Risk variables","metadata":{}},{"cell_type":"code","source":"%%time\ngc.collect()\nvariables_mapping = {'D_' : 'Delinquency' , \n                     'S_' : 'Spend', \n                     'P_' : 'Payment', \n                     'B_' : 'Balance', \n                     'R_' : 'Risk'}\n\nvariables={}\nfor key, value in variables_mapping.items():\n    variables[value] = [col for col in df.columns if col.startswith(key)]\n    print(value +' variables consist of ',len(variables[value]),' features')\n    df[['customer_ID']+variables[value]].to_parquet(OUTPUT_FOLDER+'train_data_'+value+'.parquet.gzip', index = False, compression = 'gzip')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T18:48:30.024407Z","iopub.execute_input":"2023-04-16T18:48:30.024937Z","iopub.status.idle":"2023-04-16T18:56:39.075991Z","shell.execute_reply.started":"2023-04-16T18:48:30.024793Z","shell.execute_reply":"2023-04-16T18:56:39.074973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nfor key,value in variables.items():\n    print('\\n#########################\\n')\n    print(key+' variables\\n')\n    data = df[value]\n    print(data.info())\n    print(data.dtypes.value_counts())\n    print(data.notna().mean().sort_values(ascending = False))\n    print('\\n#########################\\n')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:15:35.144358Z","iopub.execute_input":"2023-04-16T19:15:35.145334Z","iopub.status.idle":"2023-04-16T19:15:46.925703Z","shell.execute_reply.started":"2023-04-16T19:15:35.145260Z","shell.execute_reply":"2023-04-16T19:15:46.924796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key,value in variables.items():\n\n    data = df[value].select_dtypes(exclude='number')\n    \n    if len(data.columns)>0:\n        print('\\n#########################\\n')\n        print(key+' variables\\n')\n        print(data.head(5))\n        print(data.describe(include='object'))\n        print('\\n#########################\\n')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T18:32:10.228971Z","iopub.execute_input":"2023-04-16T18:32:10.229631Z","iopub.status.idle":"2023-04-16T18:32:26.419130Z","shell.execute_reply.started":"2023-04-16T18:32:10.229579Z","shell.execute_reply":"2023-04-16T18:32:26.417507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:16:35.278960Z","iopub.execute_input":"2023-04-16T19:16:35.279380Z","iopub.status.idle":"2023-04-16T19:16:35.332094Z","shell.execute_reply.started":"2023-04-16T19:16:35.279348Z","shell.execute_reply":"2023-04-16T19:16:35.330318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''df = pl.read_csv(TEST_FILE_PATH,rechunk=False).to_pandas()\nfor key, value in variables_mapping.items():\n    df[['customer_ID']+variables[value]].to_parquet(OUTPUT_FOLDER+'test_data_'+value+'.parquet.gzip', index = False, compression = 'gzip')\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}