{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-23T08:51:03.724465Z","iopub.execute_input":"2024-02-23T08:51:03.724960Z","iopub.status.idle":"2024-02-23T08:51:04.097823Z","shell.execute_reply.started":"2024-02-23T08:51:03.724918Z","shell.execute_reply":"2024-02-23T08:51:04.096926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Required Libraries\nimport seaborn as sns\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.linear_model import LogisticRegressionCV\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:04.099733Z","iopub.execute_input":"2024-02-23T08:51:04.100785Z","iopub.status.idle":"2024-02-23T08:51:05.539619Z","shell.execute_reply.started":"2024-02-23T08:51:04.100750Z","shell.execute_reply":"2024-02-23T08:51:05.538632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path configuration for easy file accessibility\nbase_path = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\npath_train_csv = base_path + \"csv_files/train/\"\npath_test_csv = base_path + \"csv_files/test/\"\npath_train_parquet = base_path + \"parquet_files/train/\"\npath_test_parquet = base_path + \"parquet_files/test/\"\n\n# Helper function for file listing and count\ndef file_count_and_list(directory_path):\n    file_list = os.listdir(directory_path)\n    print(f\"\\nFiles in {directory_path}:\\n{file_list}\")\n    print(f\"\\nTotal files: {len(file_list)}\")\n    return file_list","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:05.540868Z","iopub.execute_input":"2024-02-23T08:51:05.541216Z","iopub.status.idle":"2024-02-23T08:51:05.549613Z","shell.execute_reply.started":"2024-02-23T08:51:05.541167Z","shell.execute_reply":"2024-02-23T08:51:05.547669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Executing basic file check\nprint(\"Training CSV Files:\")\ntrain_csv_files = file_count_and_list(path_train_csv)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:05.550852Z","iopub.execute_input":"2024-02-23T08:51:05.551546Z","iopub.status.idle":"2024-02-23T08:51:05.564071Z","shell.execute_reply.started":"2024-02-23T08:51:05.551513Z","shell.execute_reply":"2024-02-23T08:51:05.563024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n* The directory holds various datasets, including credit bureau records, applicant demographics, and financial attributes.\n\n* It offers a comprehensive approach to modelling credit risk through multidimensional data analysis.","metadata":{}},{"cell_type":"code","source":"print(\"\\nTraining Parquet Files:\")\ntrain_parquet_files = file_count_and_list(path_train_parquet)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:05.566514Z","iopub.execute_input":"2024-02-23T08:51:05.566755Z","iopub.status.idle":"2024-02-23T08:51:05.572224Z","shell.execute_reply.started":"2024-02-23T08:51:05.566734Z","shell.execute_reply":"2024-02-23T08:51:05.571566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n\n* Both CSV and training Parquet files contain identical home credit and credit risk modelling datasets.\n\n* The move to the Parquet file format enhances storage efficiency and processing performance for large-scale data operations while maintaining data consistency.","metadata":{}},{"cell_type":"code","source":"print(\"\\nTest CSV Files:\")\ntest_csv_files = file_count_and_list(path_test_csv)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:05.573107Z","iopub.execute_input":"2024-02-23T08:51:05.574249Z","iopub.status.idle":"2024-02-23T08:51:05.580714Z","shell.execute_reply.started":"2024-02-23T08:51:05.574215Z","shell.execute_reply":"2024-02-23T08:51:05.579772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n\n* The project directory for Home Credit and Credit Risk Model Stability contains 36 CSV files with different datasets related to credit bureau records, applicant information, previous application histories, tax registry data, and other financial attributes.","metadata":{}},{"cell_type":"code","source":"print(\"\\nTest Parquet Files:\")\ntest_parquet_files = file_count_and_list(path_test_parquet)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:05.581738Z","iopub.execute_input":"2024-02-23T08:51:05.582065Z","iopub.status.idle":"2024-02-23T08:51:05.590032Z","shell.execute_reply.started":"2024-02-23T08:51:05.582033Z","shell.execute_reply":"2024-02-23T08:51:05.589232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n\n* The directory has 36 test Parquet files for the Home Credit and Credit Risk Model Stability project. \n\n* These files match the test CSVs, showing consistent data structure and efforts to maintain data integrity, facilitating analysis.","metadata":{}},{"cell_type":"code","source":"# Reading the base datasets\ntrain_base = pd.read_csv(path_train_csv + \"train_base.csv\")\ntest_base = pd.read_csv(path_test_csv + \"test_base.csv\")\n\n# Display basic information about train and test base datasets\nprint(\"\\nFirst 5 rows of Train Base Dataset:\\n\", train_base.head())\nprint(\"\\nFirst 5 rows of Test Base Dataset:\\n\", test_base.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:05.591186Z","iopub.execute_input":"2024-02-23T08:51:05.591471Z","iopub.status.idle":"2024-02-23T08:51:06.447099Z","shell.execute_reply.started":"2024-02-23T08:51:05.591448Z","shell.execute_reply":"2024-02-23T08:51:06.446179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations:\n\n* Both datasets have a \"case_id\" column, which probably functions as a unique identifier for each entry.\n\n* Both datasets have a \"date_decision\" column indicating the decision date. \"MONTH\" likely represents the decision month in YYYYMM format, and \"WEEK_NUM\" shows the decision week number.\n\n* The \"Train Base\" dataset includes a \"target\" column, suggesting it may be the variable to predict in the training set.","metadata":{}},{"cell_type":"code","source":"print(\"\\nShape of Train Base Dataset:\", train_base.shape)\nprint(\"Shape of Test Base Dataset:\", test_base.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:06.448462Z","iopub.execute_input":"2024-02-23T08:51:06.449151Z","iopub.status.idle":"2024-02-23T08:51:06.453742Z","shell.execute_reply.started":"2024-02-23T08:51:06.449123Z","shell.execute_reply":"2024-02-23T08:51:06.452694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations:\n\nThe Test Base Dataset is smaller than the Train Base Dataset, which is common in machine learning tasks. ","metadata":{}},{"cell_type":"code","source":"# Checking for missing values\nprint(\"\\nMissing Values in Train Base Dataset:\\n\", train_base.isnull().sum())\nprint(\"\\nMissing Values in Test Base Dataset:\\n\", test_base.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:06.454789Z","iopub.execute_input":"2024-02-23T08:51:06.455073Z","iopub.status.idle":"2024-02-23T08:51:06.610383Z","shell.execute_reply.started":"2024-02-23T08:51:06.455041Z","shell.execute_reply":"2024-02-23T08:51:06.609425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations:\n\n* Train and Test Base Datasets contain no missing values, making them ideal for analysis and modelling with no need for preprocessing.","metadata":{}},{"cell_type":"code","source":"# Basic Data Information\nprint(\"\\nTrain Base Dataset Info:\")\ntrain_base.info()\n\nprint(\"\\nTest Base Dataset Info:\")\ntest_base.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:06.611402Z","iopub.execute_input":"2024-02-23T08:51:06.611667Z","iopub.status.idle":"2024-02-23T08:51:06.770670Z","shell.execute_reply.started":"2024-02-23T08:51:06.611644Z","shell.execute_reply":"2024-02-23T08:51:06.769836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations:\n\n* Both datasets have similar columns, such as case ID, decision date, month, and week number. \n\n* The Train Base dataset has an extra column named \"target\", which could be the target variable for machine learning. \n\n* The data types are appropriate, except for the date_decision column, which needs conversion to datetime for further analysis.","metadata":{}},{"cell_type":"code","source":"def tr_dtypes(df):\n    \"\"\"\n    Transforms column data types in a DataFrame\n    \n    Parameters:\n    df (pd.DataFrame): DataFrame to be transformed\n    \n    Returns:\n    pd.DataFrame: DataFrame with transformed data types\n    \"\"\"\n    for col in df.columns:\n        if col == \"date_decision\":\n            df[col] = pd.to_datetime(df[col])\n        elif col in (\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"):\n            df[col] = df[col].astype(\"int64\")\n        elif col[-1] in (\"A\", \"P\"):\n            df[col] = df[col].astype(\"float64\")\n        elif col[-1] == \"D\":\n            df[col] = pd.to_datetime(df[col])\n        elif col[-1] in (\"M\", \"L\", \"T\"):\n            df[col] = df[col].astype(\"category\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:51:06.771874Z","iopub.execute_input":"2024-02-23T08:51:06.772430Z","iopub.status.idle":"2024-02-23T08:51:06.884512Z","shell.execute_reply.started":"2024-02-23T08:51:06.772399Z","shell.execute_reply":"2024-02-23T08:51:06.883833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = tr_dtypes(train_base)\ndf_test = tr_dtypes(test_base)\n\n# To verify data type transformations\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:54:55.651216Z","iopub.execute_input":"2024-02-23T08:54:55.651981Z","iopub.status.idle":"2024-02-23T08:54:55.720698Z","shell.execute_reply.started":"2024-02-23T08:54:55.651946Z","shell.execute_reply":"2024-02-23T08:54:55.719627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations:\n\n* The `date_decision` column has been converted to datetime format (`datetime64[ns]`) to handle and analyse date-related information better\n\n* 'datetime' data type has no significant impact on the memory footprint of the dataset after conversion\n\n* The column names and their respective counts remained unchanged, indicating that no columns were added or removed during conversion","metadata":{}},{"cell_type":"code","source":"# Visualization and Transformation Libraries\nplt.rcParams['figure.figsize'] = [10, 5] # Set default figure size\nwarnings.filterwarnings('ignore') # Ignore warnings\n\n# Simple line plot showing target mean over time (WEEK_NUM)\ndf_viz = df_train.groupby('WEEK_NUM')['target'].mean().reset_index()\nsns.lineplot(data=df_viz, x='WEEK_NUM', y='target').set_title('Average Target Value Over Time')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:54:57.629956Z","iopub.execute_input":"2024-02-23T08:54:57.630353Z","iopub.status.idle":"2024-02-23T08:54:57.976640Z","shell.execute_reply.started":"2024-02-23T08:54:57.630321Z","shell.execute_reply":"2024-02-23T08:54:57.975796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n\n* The line graph shows the average value of the target variable over time, segmented by weeks, with some peaks and troughs. \n\n* A sharp peak around week 60 suggests a significant change or event affecting the target variable.","metadata":{}},{"cell_type":"code","source":"# Distribution of target variable\nsns.displot(df_train['target'], kde=True)\nplt.title('Distribution of Target Variable')","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:54:21.141633Z","iopub.execute_input":"2024-02-23T08:54:21.142299Z","iopub.status.idle":"2024-02-23T08:54:28.084979Z","shell.execute_reply.started":"2024-02-23T08:54:21.142266Z","shell.execute_reply":"2024-02-23T08:54:28.084044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n\n* Most of the data points are in the '0' class while the '1' class has significantly fewer data points, indicating an extreme imbalance between the two classes","metadata":{}},{"cell_type":"code","source":"# Percentage visualization for the distribution of the target variable\ntarget_counts = df_train['target'].value_counts(normalize=True) * 100\nplt.figure(figsize=(8, 6))\nplt.pie(target_counts, labels=target_counts.index, autopct='%1.1f%%', startangle=140)\nplt.title('Percentage Distribution of Target Variable')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:54:17.263054Z","iopub.execute_input":"2024-02-23T08:54:17.263431Z","iopub.status.idle":"2024-02-23T08:54:17.376590Z","shell.execute_reply.started":"2024-02-23T08:54:17.263404Z","shell.execute_reply":"2024-02-23T08:54:17.375704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations :\n\n* It illustrates a binary target variable with 96.9% for the '0' class and only 3.1% for the '1' class, indicating an imbalanced class distribution. \n\n* It could lead to bias in machine learning models towards predicting the majority class.","metadata":{}}]}