{"metadata":{"colab":{"provenance":[{"file_id":"17MsV4f8Bap8y2MU8BcKBNXRdTNE2ftP3","timestamp":1707704331804}]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Kaggle competition -  Home Credit Risk Prediction - Utility**","metadata":{"id":"mS7njSg799kz"}},{"cell_type":"markdown","source":"Inspired by notebooks by other kagglers:\n\n1) https://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook\n\n2) https://www.kaggle.com/code/andreynesterov/home-credit-baseline-data\n","metadata":{"id":"BuRooUJZVIzl"}},{"cell_type":"markdown","source":"## **Loading necessary libraries**","metadata":{"id":"CgKYO1eP91dt"}},{"cell_type":"code","source":"#import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing\nimport polars as pl # CSV file I/O (e.g. pl.read_csv)\nimport datetime as dt\n\n# import further libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# display all columns\npd.set_option('display.max_columns', None)\n\n# library for garbage collection\nimport gc  # since the data is huge here, regularly cleaning up will free up memory\n\n# library to catch and ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"id":"qNIr85ukHVCz","executionInfo":{"status":"ok","timestamp":1710111779612,"user_tz":240,"elapsed":164,"user":{"displayName":"Varuni Rao","userId":"00194682170874035293"}},"execution":{"iopub.status.busy":"2024-03-19T23:26:36.549372Z","iopub.execute_input":"2024-03-19T23:26:36.549861Z","iopub.status.idle":"2024-03-19T23:26:36.557157Z","shell.execute_reply.started":"2024-03-19T23:26:36.549819Z","shell.execute_reply":"2024-03-19T23:26:36.556140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Loading data onto dataframes**","metadata":{"id":"zZXKOJ4Y-yKP"}},{"cell_type":"code","source":"class DataLoader:\n    comp_files = pd.DataFrame({\n    'file_name': ['base', 'static', 'static_cb', 'applprev', 'other', 'tax_registry_a', 'tax_registry_b', 'tax_registry_c', 'credit_bureau_a', 'credit_bureau_b', 'deposit', 'person',\n                  'debitcard', 'applprev', 'person', 'credit_bureau_a', 'credit_bureau_b'],\n    'depth': [0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2],\n    'train_chunk': [1, 2, 1, 2, 1, 1, 1, 1, 4, 1, 1, 1, 1, 1, 1, 11, 1],\n    'test_chunk': [1, 3, 1, 3, 1, 1, 1, 1, 5, 1, 1, 1, 1, 1, 1, 11, 1]\n    })\n    \n    file_path = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/\"\n\n    @staticmethod\n    def read_and_fix_dtype(parquet_file):\n\n      # read the data in the parquet file into polars dataframe    \n      df = pl.read_parquet(parquet_file)                  # polars read file\n      print(\"DataFrame shape for\", parquet_file, \"is :\", df.shape)    # display dataframe shape\n\n      # set datatype for each of the columns - because the column names\n      # have been transformed to indicate the dtype as per the creator of the competition\n      for column in df.columns:\n        if column in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n           df.with_columns(pl.col(column).cast(pl.Int64, strict = False))\n        elif column[-1] in (\"P\", \"A\"):\n           df = df.with_columns(pl.col(column).cast(pl.Float64, strict = False))\n        elif column[-1] in (\"M\"):\n           df = df.with_columns(pl.col(column).cast(pl.Categorical, strict = False))\n        elif column in [\"date_decision\"] or column[-1] in (\"D\"):\n           df = df.with_columns(pl.col(column).cast(pl.Date, strict = False))\n\n      # return polars dataframe\n      return df\n\n\n    @staticmethod\n    def get_dataframe(subtype, depth = 0, file_df = comp_files):\n      df = pl.DataFrame()\n      files_count = len(file_df[file_df['depth'] == depth])\n      no_of_files = 1\n    \n      file_path = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/\"\n        \n      for n in (0, files_count):\n          no_of_files = no_of_files + file_df[str(subtype + '_chunk')].iloc[n]\n      print(\"Number of files to merge are:\", no_of_files)\n\n      for i in range(0, files_count):\n        if i == 0:\n          filename = subtype + '_' + file_df['file_name'].iloc[i] + '.parquet'\n          print('Base file:', filename)\n          df = DataLoader.read_and_fix_dtype(file_path + subtype + \"/\" + filename)\n        else:\n          chunk = subtype + '_chunk'\n          if file_df[chunk].iloc[i] == 1:\n            filename = subtype + '_' + file_df['file_name'].iloc[i] + '_' + str(depth) + '.parquet'\n            print('No chunk file:', filename)\n            df = df.join(DataLoader.read_and_fix_dtype(file_path + subtype + \"/\" + filename), on = \"case_id\", how = \"left\")\n          else:\n            for j in range(0, file_df[chunk].iloc[i]):\n              if j == 0:\n                filename = subtype + '_' + file_df['file_name'].iloc[i] + '_' + str(depth) + '_' + str(j) + '.parquet'\n                print('   Chunk file:', filename)\n                df_chunk = DataLoader.read_and_fix_dtype(file_path + subtype + \"/\" + filename)\n              else:\n                filename = subtype + '_' + file_df['file_name'].iloc[i] + '_' + str(depth) + '_' + str(j) + '.parquet'\n                print('   Chunk file:', filename)\n                df_chunk = pl.concat([df_chunk, DataLoader.read_and_fix_dtype(file_path + subtype + \"/\" + filename)], how = \"vertical_relaxed\")\n            df = df.join(df_chunk, on = \"case_id\", how = \"left\")\n\n      print(\"\\n\", subtype, \"dataframe shape is:\", df.shape)\n      return df","metadata":{"id":"gKGWEveRWZls","executionInfo":{"status":"ok","timestamp":1710111779761,"user_tz":240,"elapsed":3,"user":{"displayName":"Varuni Rao","userId":"00194682170874035293"}},"execution":{"iopub.status.busy":"2024-03-19T23:26:36.559214Z","iopub.execute_input":"2024-03-19T23:26:36.559842Z","iopub.status.idle":"2024-03-19T23:26:36.587153Z","shell.execute_reply.started":"2024-03-19T23:26:36.559797Z","shell.execute_reply":"2024-03-19T23:26:36.586049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MemoryOptimizer:\n\n    @staticmethod\n    def reduce_memory_usage(train, test, col_list):\n\n      # display the memory usage before optimization\n      start_mem_train = train.memory_usage().sum() / 1024 ** 2;\n      start_mem_test = test.memory_usage().sum() / 1024 ** 2;\n      gc.collect()\n      print('Memory usage of train dataframe before optimization is {:.2f} MB'.format(start_mem_train))\n      print('Memory usage of test dataframe before optimization is {:.2f} MB'.format(start_mem_test))\n      print(\"\")\n\n      # iterate through all columns\n      for col in col_list:\n        col_dtype = train[col].dtype # get column dtype\n\n        # treating integer dtypes\n        if col_dtype == 'int64':\n          c_min = train[col].min()\n          c_max = train[col].max()\n          #print(\"For {}, cmin: {}, and cmax: {}\".format(col, c_min, c_max))\n\n          # treating int64 types\n          if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n              #print(\"Changing {} from {} to int8, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.int8).min, np.iinfo(np.int8).max))\n              train[col] = train[col].astype(np.int8)\n              if col != 'target':\n                test[col] = test[col].astype(np.int8)\n          elif c_min > np.iinfo(np.uint8).min and c_max < np.iinfo(np.uint8).max:\n              #print(\"Changing {} from {} to uint8, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.uint8).min, np.iinfo(np.uint8).max))\n              train[col] = train[col].astype(np.uint8)\n              if col != 'target':\n                test[col] = test[col].astype(np.uint8)\n          elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n              #print(\"Changing {} from {} to int16, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.int16).min, np.iinfo(np.int16).max))\n              train[col] = train[col].astype(np.int16)\n              if col != 'target':\n                test[col] = test[col].astype(np.int16)\n          elif c_min > np.iinfo(np.uint16).min and c_max < np.iinfo(np.uint16).max:\n              #print(\"Changing {} from {} to uint16, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.uint16).min, np.iinfo(np.uint16).max))\n              train[col] = train[col].astype(np.uint16)\n              if col != 'target':\n                test[col] = test[col].astype(np.uint16)\n          elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n              #print(\"Changing {} from {} to int32, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.int32).min, np.iinfo(np.int32).max))\n              train[col] = train[col].astype(np.int32)\n              if col != 'target':\n                test[col] = test[col].astype(np.int32)\n          elif c_min > np.iinfo(np.uint32).min and c_max < np.iinfo(np.uint32).max:\n              #print(\"Changing {} from {} to uint32, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.uint32).min, np.iinfo(np.uint32).max))\n              train[col] = train[col].astype(np.uint32)\n              if col != 'target':\n                test[col] = test[col].astype(np.uint32)\n          #elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n          #    print(\"Changing {} from {} to int64, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.int64).min, np.iinfo(np.int64).max))\n          #    train[col] = train[col].astype(np.int64)\n          #    if col != 'target':\n          #      test[col] = test[col].astype(np.int64)\n          #elif c_min > np.iinfo(np.uint64).min and c_max < np.iinfo(np.uint64).max:\n          #    print(\"Changing {} from {} to uint64, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.iinfo(np.uint64).min, np.iinfo(np.uint64).max))\n          #    train[col] = train[col].astype(np.uint64)\n          #    if col != 'target':\n          #      test[col] = test[col].astype(np.uint64)\n\n        # treating float64 types\n        elif col_dtype == 'float64':\n          c_min = train[col].min()\n          c_max = train[col].max()\n          #print(\"For {}, cmin: {}, and cmax: {}\".format(col, c_min, c_max))\n\n          if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n              #print(\"Changing {} from {} to float16, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.finfo(np.float16).min, np.finfo(np.float16).max))\n              train[col] = train[col].astype(np.float16)\n              test[col] = test[col].astype(np.float16)\n          elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n              #print(\"Changing {} from {} to float32, , since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.finfo(np.float32).min, np.finfo(np.float32).max))\n              train[col] = train[col].astype(np.float32)\n              test[col] = test[col].astype(np.float32)\n          #elif c_min > np.finfo(np.float64).min and c_max < np.finfo(np.float64).max:\n          #    print(\"Changing {} from {} to float64, since cmin > {} and cmax < {}\".format(col,train[col].dtype, np.finfo(np.float64).min, np.finfo(np.float64).max))\n          #    train[col] = train[col].astype(np.float64)\n          #    test[col] = test[col].astype(np.float64)\n\n        # treating object types\n        elif col_dtype == 'object':\n          #print(\"Changing {} from {} to category\".format(col, train[col].dtype))\n          train[col] = train[col].astype('category')\n          test[col] = test[col].astype('category')\n\n        #print(\"New dtype of {} is {}\".format(col, train[col].dtype))\n        #print(\"-\"*100)\n\n      # collect garbage to free memory\n      gc.collect()\n\n      # determine memory usage after optimization\n      end_mem_train = train.memory_usage().sum() / 1024 ** 2\n      end_mem_test = test.memory_usage().sum() / 1024 ** 2\n      print('Memory usage after optimization for train dataframe is: {:.3f} MB'.format(end_mem_train))\n      print('Decreased by {:.1f}%'.format(100 * (start_mem_train - end_mem_train) / start_mem_train))\n      print('Memory usage after optimization for test dataframe is: {:.3f} MB'.format(end_mem_test))\n      print('Decreased by {:.1f}%'.format(100 * (start_mem_test - end_mem_test) / start_mem_test))\n\n      # return the dataframe\n      return train, test","metadata":{"id":"epDSa5riCRWy","executionInfo":{"status":"ok","timestamp":1710111779761,"user_tz":240,"elapsed":3,"user":{"displayName":"Varuni Rao","userId":"00194682170874035293"}},"execution":{"iopub.status.busy":"2024-03-19T23:26:36.736913Z","iopub.execute_input":"2024-03-19T23:26:36.737317Z","iopub.status.idle":"2024-03-19T23:26:36.776268Z","shell.execute_reply.started":"2024-03-19T23:26:36.737287Z","shell.execute_reply":"2024-03-19T23:26:36.775074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preprocessing**\n","metadata":{"id":"g--YNWlwRa3B"}},{"cell_type":"code","source":"class DataPreprocessor:\n  @staticmethod\n  def filtercols(train, test, base=False):\n      # drop month column if base dataframe\n      if base:\n        #print(\"Dropping column MONTH........\")\n        train.drop('MONTH', axis = 1, inplace = True)\n        test.drop('MONTH', axis = 1, inplace = True)\n\n      #drop columns that have more than 90% missing data\n      for col in train.columns:\n        #print(\"Considering column {}\".format(str(col)))\n        if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n           if train[col].isnull().mean() > 0.9:\n              #print(\"Dropping {} from train and test dataframe since percentage of missing value is: {:.2f}%\".format(col, train[col].isnull().mean()*100))\n              train.drop(col, axis=1, inplace=True)\n              test.drop(col, axis=1, inplace=True)\n\n      # drop columns that have cardinality of 1 or more than 100 for categorical columns\n      cat_col = train.select_dtypes(['object', 'category']).columns\n      for col in cat_col:\n        if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]):\n          freq = train[col].nunique()\n\n          if (freq == 1) | (freq > 100):\n            #print(\"Dropping {} from train and test set because the cardinality of column is: {}\".format(col, freq))\n            train.drop(col, axis=1, inplace=True)\n            test.drop(col, axis=1, inplace=True)\n\n      return train, test\n\n  @staticmethod\n  # function that returns either years or days or months information based on the dt_type\n  def getdatedelta(df, col, dt_type = 'years'):\n    if dt_type == 'years':\n        return np.round(((df['date_decision'] - df[col]).dt.days)/365, 0)\n    elif dt_type == 'months':\n        return np.round(((df['date_decision'] - df[col]).dt.days)/30, 0)\n    elif dt_type == 'days':\n        return (df['date_decision'] - df[col]).dt.days\n\n  @staticmethod\n  # create new Categorical dtype for each category by adding 'Unknown' to the list of unique categories of the column\n  def createnewcatdtype(train, cat_cols):\n    for col in cat_cols:\n      new_categories = train[col].cat.categories.to_list() + [\"Unknown\"]\n      new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n      train[col] = train[col].astype(new_dtype)\n\n    return train\n\n  @staticmethod\n  # this will set the new categorical dtype to train and test\n  def setnewcatdtype(train, test, cat_cols):\n    for col in cat_cols:\n      train_categories = set(train[col].cat.categories)\n      test_categories = set(test[col].cat.categories)\n      new_categories = test_categories - train_categories\n\n      # Set the categories for the Categorical column\n      new_dtype = pd.CategoricalDtype(categories=train_categories, ordered=True)\n      train[col] = train[col].astype(new_dtype)\n      test[col] = test[col].astype(new_dtype)\n\n      # Replace new categories with \"Unknown\" in the test DataFrame\n      if len(new_categories) > 0:\n        #print(\"Replacing new categories {} with Unknown for {}.......\".format(new_categories, col))\n        test.loc[test[col].isin(new_categories), col] = \"Unknown\"\n\n    return train, test","metadata":{"id":"di2gipiLp7NS","executionInfo":{"status":"ok","timestamp":1710111779761,"user_tz":240,"elapsed":2,"user":{"displayName":"Varuni Rao","userId":"00194682170874035293"}},"execution":{"iopub.status.busy":"2024-03-19T23:26:36.779348Z","iopub.execute_input":"2024-03-19T23:26:36.780138Z","iopub.status.idle":"2024-03-19T23:26:36.800302Z","shell.execute_reply.started":"2024-03-19T23:26:36.780101Z","shell.execute_reply":"2024-03-19T23:26:36.799006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"6CSp-1mwR9IQ","executionInfo":{"status":"ok","timestamp":1710111779761,"user_tz":240,"elapsed":2,"user":{"displayName":"Varuni Rao","userId":"00194682170874035293"}}},"execution_count":null,"outputs":[]}]}