{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":167900474,"sourceType":"kernelVersion"},{"sourceId":166996856,"sourceType":"kernelVersion"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing libraries for data handling and manipulation\nimport polars as pl  # Similar to pandas, but offers better performance for large datasets.\nimport pandas as pd  # Library for importing and manipulating CSV files.\nimport numpy as np  # Library for matrix operations and numerical processing.\n\n# Importing machine learning utilities\nfrom sklearn.metrics import roc_auc_score  # Importing the ROC AUC score for model evaluation.\n# StratifiedKFold considers the proportion of each category when splitting the dataset, unlike simple KFold.\nfrom sklearn.model_selection import StratifiedKFold  \nfrom sklearn.decomposition import TruncatedSVD  # Truncated Singular Value Decomposition for dimensionality reduction.\n\n# Importing utilities for serialization and memory management\nimport dill  # For serialization and deserialization (e.g., saving and loading tree models).\nimport gc  # Garbage collection module for managing memory.\n\n# Importing the time module for time-related operations\nimport time  # Standard library's time module.\n# Recording the current time for reference, useful to avoid versioning errors when calling trained models later.\n# The time.strftime() function formats a time object into a string, time.localtime() returns the current local time.\ncurrent_time = time.strftime(\"%Y-%m-%d %H:%M:%S\", time.localtime())\nprint(\"this notebook training time is \", current_time)  # Printing the current training time for this notebook.\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-05T22:28:52.189899Z","iopub.execute_input":"2024-04-05T22:28:52.191429Z","iopub.status.idle":"2024-04-05T22:28:56.486285Z","shell.execute_reply.started":"2024-04-05T22:28:52.191301Z","shell.execute_reply":"2024-04-05T22:28:56.484512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#1. Configuration and Seed Setting\nclass Config():\n    # Configuration parameters for reproducibility and batch processing\n    seed = 2024  # Random seed for consistency across runs\n    num_folds = 10  # Number of folds for cross-validation\n    TARGET_NAME = 'target'  # Name of the target variable\n    batch_size = 1000  # Size of batches for processing to manage memory usage\n\nimport random\nimport numpy as np\n\ndef seed_everything(seed):\n    \"\"\"Sets the random seed for numpy and built-in random module.\"\"\"\n    np.random.seed(seed)  # Set numpy random seed\n    random.seed(seed)  # Set Python random seed\n\nseed_everything(Config.seed)  # Initialize the seed setting\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T22:29:04.350731Z","iopub.execute_input":"2024-04-05T22:29:04.351464Z","iopub.status.idle":"2024-04-05T22:29:04.361377Z","shell.execute_reply.started":"2024-04-05T22:29:04.351422Z","shell.execute_reply":"2024-04-05T22:29:04.359565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2. Reading Data Types for Columns\nimport pandas as pd\nimport polars as pl\n\n# Read a CSV file mapping column names to their data types\ncolname2dtype = pd.read_csv(\"/kaggle/input/home-credit-inconsistent-data-types/colname2dtype.csv\")\ncolname = colname2dtype['Column'].values  # Extract column names\ndtype = colname2dtype['DataType'].values  # Extract corresponding data types\n\n# Dictionary to map pandas data types to Polars data types\ndtype2pl = {\n    'Int64': pl.Int64,\n    'Float64': pl.Float64,\n    'String': pl.String,\n    'Boolean': pl.Boolean  # Corrected to 'pl.Boolean'\n}\n\n# Mapping column names to their Polars data types\ncolname2dtype = {}\nfor idx in range(len(colname)):\n    colname2dtype[colname[idx]] = dtype2pl[dtype[idx]]\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T23:22:58.815916Z","iopub.execute_input":"2024-04-05T23:22:58.816417Z","iopub.status.idle":"2024-04-05T23:22:58.848554Z","shell.execute_reply.started":"2024-04-05T23:22:58.816385Z","shell.execute_reply":"2024-04-05T23:22:58.847390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3. Custom Function: Finding Columns with High Missing Value Ratios\ndef find_df_null_col(df, margin=0.975):\n    \"\"\"Finds columns in a DataFrame with a high ratio of missing values.\"\"\"\n    cols = []  # Initialize an empty list to store column names\n    for col in df.columns:  # Iterate over all columns\n        if df[col].isna().mean() > margin:  # Check if missing value ratio exceeds margin\n            cols.append(col)  # Add column name to list\n    return cols  # Return list of columns with high missing value ratio\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T23:23:03.767039Z","iopub.execute_input":"2024-04-05T23:23:03.767521Z","iopub.status.idle":"2024-04-05T23:23:03.775585Z","shell.execute_reply.started":"2024-04-05T23:23:03.767485Z","shell.execute_reply":"2024-04-05T23:23:03.774286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#4. Custom Function: Retaining Last Record per case_id\ndef find_last_case_id(df, id='case_id'):\n    \"\"\"Keeps only the last record for each case_id in the DataFrame.\"\"\"\n    df_copy = df.clone()  # Make a copy of the DataFrame\n    df_tail = df.tail(1)  # Extract the last row\n    # Shift case_id column by -1 and check for changes to find the last occurrence\n    df_copy = df_copy.with_columns(pl.col(id).shift(-1).alias(f\"{id}_shift_-1\"))\n    # Filter rows where the current case_id is different from the next one\n    df_last = df_copy.filter(pl.col(id) - pl.col(f'{id}_shift_-1') != 0).drop(f'{id}_shift_-1')\n    df_last = pl.concat([df_last, df_tail])  # Add the very last row back\n    # Clean up to free memory\n    del df_copy, df_tail\n    gc.collect()\n    return df_last  # Return DataFrame with last occurrences\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T23:23:07.693511Z","iopub.execute_input":"2024-04-05T23:23:07.694902Z","iopub.status.idle":"2024-04-05T23:23:07.702616Z","shell.execute_reply.started":"2024-04-05T23:23:07.694860Z","shell.execute_reply":"2024-04-05T23:23:07.701639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#5. Custom Function: Filling Missing Values\ndef df_fillna(df, col, method=None):\n    \"\"\"Fills missing values in a DataFrame column.\"\"\"\n    if method is None:\n        pass  # Do nothing if no method is specified\n    elif method == \"forward\":\n        # Forward fill missing values\n        df = df.select([pl.col(col).fill_null('forward')])\n    else:\n        # Fill missing values with a specified method/value\n        df = df.with_columns(pl.col(col).fill_null(method).alias(col))\n    return df  # Return the DataFrame with filled values\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T23:23:11.349614Z","iopub.execute_input":"2024-04-05T23:23:11.350688Z","iopub.status.idle":"2024-04-05T23:23:11.359204Z","shell.execute_reply.started":"2024-04-05T23:23:11.350648Z","shell.execute_reply":"2024-04-05T23:23:11.357757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#6. Custom Function: One-Hot Encoding\ndef one_hot_encoder(df, col, unique):\n    \"\"\"Applies one-hot encoding to a specified column.\"\"\"\n    if len(unique) == 2:\n        # If only two categories, encode with a single binary column\n        df = df.with_columns((pl.col(col) == unique[0]).cast(pl.Int8).alias(f\"{col}_{unique[0]}\"))\n    else:\n        # For more than two categories, create a binary column for each category\n        for idx in range(len(unique)):\n            df = df.with_columns((pl.col(col) == unique[idx]).cast(pl.Int8).alias(f\"{col}_{unique[idx]}\"))\n    return df.drop(col)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T23:23:14.251283Z","iopub.execute_input":"2024-04-05T23:23:14.251736Z","iopub.status.idle":"2024-04-05T23:23:14.260561Z","shell.execute_reply.started":"2024-04-05T23:23:14.251706Z","shell.execute_reply":"2024-04-05T23:23:14.259113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Data Cleaning and Preprocessing Summary:\\n\")\n\n# Configuration and Seed Setting\nprint(\"1. Configuration parameters set for reproducibility.\")\nprint(f\"   - Random seed set to: {Config.seed}\")\nprint(f\"   - Number of folds for cross-validation: {Config.num_folds}\")\nprint(f\"   - Target variable name: '{Config.TARGET_NAME}'\")\nprint(f\"   - Batch size for processing: {Config.batch_size}\\n\")\n\n# Reading Data Types\nprint(\"2. Data types for columns were read and mapped for Polars processing.\\n\")\n\n# Finding Columns with High Missing Value Ratios\nprint(\"3. Identified columns with high missing value ratios to guide further cleaning actions.\\n\")\n\n# Retaining Last Record per `case_id`\nprint(\"4. Retained the last occurrence for each `case_id`, ensuring the most recent data is used.\\n\")\n\n# Filling Missing Values\nprint(\"5. Applied strategies for filling missing values, including forward fill and custom value imputation.\\n\")\n\n# One-Hot Encoding\nprint(\"6. Executed one-hot encoding on categorical variables to prepare them for machine learning models.\\n\")\n\nprint(\"Following these steps, the dataset is now cleaned, with missing values addressed, data types correctly set for efficient processing, and categorical variables properly encoded. Ready for further analysis or model training.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T23:33:12.191750Z","iopub.execute_input":"2024-04-05T23:33:12.192276Z","iopub.status.idle":"2024-04-05T23:33:12.201670Z","shell.execute_reply.started":"2024-04-05T23:33:12.192242Z","shell.execute_reply":"2024-04-05T23:33:12.200244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}