{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7600559,"sourceType":"datasetVersion","datasetId":4424545},{"sourceId":7815606,"sourceType":"datasetVersion","datasetId":4578531}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-07T11:13:07.207961Z","iopub.execute_input":"2024-04-07T11:13:07.209274Z","iopub.status.idle":"2024-04-07T11:13:07.339790Z","shell.execute_reply.started":"2024-04-07T11:13:07.209234Z","shell.execute_reply":"2024-04-07T11:13:07.338411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CONCAT FILE","metadata":{}},{"cell_type":"code","source":"applprev_1_0 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_0.parquet')\napplprev_1_1 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_1.parquet')\ntrain_base = pd.read_parquet(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_base.parquet\")\n# combine train_applprev_1_0 + train_applprev_1_1\ndf = pd.concat([applprev_1_0,applprev_1_1 ],axis=0).reset_index(drop=True)\ndf","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:13:07.342507Z","iopub.execute_input":"2024-04-07T11:13:07.343399Z","iopub.status.idle":"2024-04-07T11:13:31.104152Z","shell.execute_reply.started":"2024-04-07T11:13:07.343352Z","shell.execute_reply":"2024-04-07T11:13:31.102988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PERCENT MISSING_VALUE","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\n# Load the DataFrame\n#applprev_1_0 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_0.parquet')\n\n# Set seed for reproducibility\nnp.random.seed(0)\n\n# Calculate percent missing values of each column\npercent_missing = df.isnull().sum() * 100 / len(df)\nmissing_applprev = pd.DataFrame({'percent_missing': percent_missing})\n\n# Identify columns with percent missing value > 80\nmorethan_80 = missing_applprev[missing_applprev['percent_missing'] > 80].index\n\n# Drop columns with percent missing value > 80\ndrop_applprev_1_0 = df.drop(columns=morethan_80)\ndrop_applprev_1_0\n# Fill NaN values in the column \"actualdpd_943P\" with 0\n#drop_applprev_1_0[\"actualdpd_943P\"].fillna(0, inplace=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:13:31.105890Z","iopub.execute_input":"2024-04-07T11:13:31.106604Z","iopub.status.idle":"2024-04-07T11:13:48.522922Z","shell.execute_reply.started":"2024-04-07T11:13:31.106564Z","shell.execute_reply":"2024-04-07T11:13:48.521772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# EXTRACT TYPE(L A T P M) FUNCTION","metadata":{}},{"cell_type":"code","source":"# feature engineering function demo and testing\ndef applepre_feature_eng(data,fealist,agglist):\n\n    result = pd.DataFrame({'case_id':list(set(data.case_id))})\n    for f in fealist:\n        print(f,'......')\n        df_table = data.loc[:,['case_id',f,'num_group1']].\\\n            sort_values(['case_id','num_group1']).\\\n            groupby(['case_id'])[f].\\\n            agg(agglist).\\\n            add_suffix('_'+ f).\\\n            reset_index()\n        result = result.merge(df_table,on=['case_id'],how='left')\n    return result","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:13:48.525963Z","iopub.execute_input":"2024-04-07T11:13:48.526330Z","iopub.status.idle":"2024-04-07T11:13:48.534210Z","shell.execute_reply.started":"2024-04-07T11:13:48.526300Z","shell.execute_reply":"2024-04-07T11:13:48.533002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# DROP COL D TYPE set option to look all column","metadata":{}},{"cell_type":"code","source":"\n\n# Drop columns from applprev_1_0\ndrop_applprev_1_0 = df.drop(columns=morethan_80)\n\n# Iterate through columns and drop columns ending with 'D'\nfor col in drop_applprev_1_0.columns:\n    if col.endswith(\"D\"):\n        drop_applprev_1_0.drop(col, axis=1, inplace=True)\n\n#Print the DataFrame\npd.set_option('display.max_columns', None)\ndrop_applprev_1_0","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:13:48.535843Z","iopub.execute_input":"2024-04-07T11:13:48.536231Z","iopub.status.idle":"2024-04-07T11:14:02.699534Z","shell.execute_reply.started":"2024-04-07T11:13:48.536201Z","shell.execute_reply":"2024-04-07T11:14:02.698394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What to fill function and automatically fill nan that calculate from table that already exclude outlier","metadata":{}},{"cell_type":"code","source":"def exclude_outliers(nf, threshold=3):\n    z_scores = np.abs((nf - nf.mean()) / nf.std())\n    outliers = z_scores > threshold\n    return nf[~outliers.any(axis=1)]\n\ndef suggest_fill_method(original_df):\n    fill_methods = {}  # Dictionary to store fill methods for each column\n    filled_df = original_df.copy()\n    for col in filled_df.columns:\n        # Check if the column contains numerical or categorical data\n        if pd.api.types.is_numeric_dtype(filled_df[col]):\n            # Exclude outliers\n            modified_df = exclude_outliers(filled_df[[col]])\n            # Calculate mean and median\n            mean_val = modified_df[col].mean()\n            median_val = modified_df[col].median()\n            # Fill missing values with mean or median\n            if filled_df[col].isnull().any():\n                if filled_df[col].skew() > 1:\n                    fill_methods[col] = 'median'\n                    original_df[col].fillna(median_val, inplace=True)  # Filling original DataFrame\n                else:\n                    fill_methods[col] = 'mean'\n                    original_df[col].fillna(mean_val, inplace=True)  # Filling original DataFrame\n        else:\n            # Fill missing values with mode for categorical data\n            original_df[col].fillna(filled_df[col].mode()[0], inplace=True)\n            fill_methods[col] = 'mode'\n\n    return original_df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:02.701047Z","iopub.execute_input":"2024-04-07T11:14:02.701422Z","iopub.status.idle":"2024-04-07T11:14:02.713535Z","shell.execute_reply.started":"2024-04-07T11:14:02.701392Z","shell.execute_reply":"2024-04-07T11:14:02.711848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndef suggest_fill_method(dataFrame, col):\n\n    # Check if the column contains numerical or categorical data\n    if pd.api.types.is_numeric_dtype(dataFrame[col]):\n        # If numerical, check if the data is skewed\n        skewness = dataFrame[col].skew()\n        if abs(skewness) > 1:\n            fill_method = 'median'  # If highly skewed, suggest filling with median\n        else:\n            fill_method = 'mean'  # If not highly skewed, suggest filling with mean\n    else:\n        fill_method = 'mode'  # If categorical, suggest filling with mode\n    \n    # Fill missing values with the suggested method\n    if fill_method == 'mode':\n        dataFrame[col].fillna(dataFrame[col].mode()[0], inplace=True)\n    elif fill_method == 'median':\n        dataFrame[col].fillna(dataFrame[col].median(), inplace=True)\n\n    return fill_method","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:02.715626Z","iopub.execute_input":"2024-04-07T11:14:02.715977Z","iopub.status.idle":"2024-04-07T11:14:02.726860Z","shell.execute_reply.started":"2024-04-07T11:14:02.715948Z","shell.execute_reply":"2024-04-07T11:14:02.725612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remove outlier function\n","metadata":{}},{"cell_type":"code","source":"# Define a function to detect and exclude outliers\ndef exclude_outliers(nf, threshold=3):\n    z_scores = np.abs((nf - nf.mean()) / nf.std())\n    outliers = z_scores > threshold\n    return nf[~outliers.any(axis=1)]\n\n# Apply the function to exclude outliers from the DataFrame","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:02.728629Z","iopub.execute_input":"2024-04-07T11:14:02.729024Z","iopub.status.idle":"2024-04-07T11:14:02.739443Z","shell.execute_reply.started":"2024-04-07T11:14:02.728978Z","shell.execute_reply":"2024-04-07T11:14:02.738351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# P FEA (Group case id)","metadata":{}},{"cell_type":"code","source":"\nP_fea = ['actualdpd_943P',\"maxdpdtolerance_577P\"]\nP_fea_table = applepre_feature_eng(data = drop_applprev_1_0, fealist = P_fea,agglist=['mean'])\nP_fea_table\n#P_fea_table[mean_actualdpd_943P].rename('actualdpd_943P')","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:02.740824Z","iopub.execute_input":"2024-04-07T11:14:02.741224Z","iopub.status.idle":"2024-04-07T11:14:07.353799Z","shell.execute_reply.started":"2024-04-07T11:14:02.741194Z","shell.execute_reply":"2024-04-07T11:14:07.352672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"percent_missing = P_fea_table.isnull().sum() * 100 / len(df)\npercent_missing","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:07.358265Z","iopub.execute_input":"2024-04-07T11:14:07.358638Z","iopub.status.idle":"2024-04-07T11:14:07.377295Z","shell.execute_reply.started":"2024-04-07T11:14:07.358608Z","shell.execute_reply":"2024-04-07T11:14:07.376166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What to fill and automatically fill nan","metadata":{}},{"cell_type":"code","source":"for col in P_fea_table.columns:\n    # Call the function for each column and update the original DataFrame\n    fill_method = suggest_fill_method(P_fea_table, col)\n    print(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\nP_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:07.378859Z","iopub.execute_input":"2024-04-07T11:14:07.391034Z","iopub.status.idle":"2024-04-07T11:14:07.504037Z","shell.execute_reply.started":"2024-04-07T11:14:07.390983Z","shell.execute_reply":"2024-04-07T11:14:07.502936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A FEA (Group case id)","metadata":{}},{"cell_type":"code","source":"A_fea = ['annuity_853A',\"credacc_credlmt_575A\",\"credamount_590A\",\"currdebt_94A\",\"downpmt_134A\",\"mainoccupationinc_437A\",\"outstandingdebt_522A\"]\nA_fea_table = applepre_feature_eng(data = drop_applprev_1_0, fealist = A_fea,agglist=['mean'])\ndrop_A_fea = A_fea_table.drop(columns='mean_currdebt_94A')\ndrop_A_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:07.505661Z","iopub.execute_input":"2024-04-07T11:14:07.507021Z","iopub.status.idle":"2024-04-07T11:14:18.790264Z","shell.execute_reply.started":"2024-04-07T11:14:07.506978Z","shell.execute_reply":"2024-04-07T11:14:18.789360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What to fill and automatically fill nan","metadata":{}},{"cell_type":"code","source":"for col in drop_A_fea.columns:\n    # Call the function for each column and update the original DataFrame\n    fill_method = suggest_fill_method(drop_A_fea, col)\n    print(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\ndrop_A_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:18.791511Z","iopub.execute_input":"2024-04-07T11:14:18.792021Z","iopub.status.idle":"2024-04-07T11:14:19.079902Z","shell.execute_reply.started":"2024-04-07T11:14:18.791984Z","shell.execute_reply":"2024-04-07T11:14:19.078514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check that there are no missing values\npercent_missing = drop_A_fea.isnull().sum() * 100 / len(df)\npercent_missing","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:19.081688Z","iopub.execute_input":"2024-04-07T11:14:19.082377Z","iopub.status.idle":"2024-04-07T11:14:19.105807Z","shell.execute_reply.started":"2024-04-07T11:14:19.082345Z","shell.execute_reply":"2024-04-07T11:14:19.104600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# L FEA (Group case id)","metadata":{}},{"cell_type":"code","source":"L_fea = [\"byoccupationinc_3656910L\",\"childnum_21L\",\"pmtnum_8L\",\"tenor_203L\"]\nL_fea_table = applepre_feature_eng(data = drop_applprev_1_0, fealist = L_fea,agglist=['mean'])\ndrop_L_fea = L_fea_table.drop(columns='mean_pmtnum_8L')\ndrop_L_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:19.107278Z","iopub.execute_input":"2024-04-07T11:14:19.107629Z","iopub.status.idle":"2024-04-07T11:14:26.532559Z","shell.execute_reply.started":"2024-04-07T11:14:19.107600Z","shell.execute_reply":"2024-04-07T11:14:26.531289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in drop_L_fea.columns:\n    # Call the function for each column and update the original DataFrame\n    fill_method = suggest_fill_method(drop_L_fea, col)\n    print(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\ndrop_L_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:26.533689Z","iopub.execute_input":"2024-04-07T11:14:26.533995Z","iopub.status.idle":"2024-04-07T11:14:26.698175Z","shell.execute_reply.started":"2024-04-07T11:14:26.533970Z","shell.execute_reply":"2024-04-07T11:14:26.696891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n\n\n\n\n\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# What to fill and automatically fill nan","metadata":{}},{"cell_type":"code","source":"L_fea_table_cat = applepre_feature_eng(data = drop_applprev_1_0, fealist = ['credtype_587L','inittransactioncode_279L','isbidproduct_390L','status_219L',\"familystate_726L\"],agglist=['first'])\ndrop_L_fea_cat = L_fea_table_cat.drop(columns='first_isbidproduct_390L')\ndrop_L_fea_cat","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:26.699582Z","iopub.execute_input":"2024-04-07T11:14:26.699912Z","iopub.status.idle":"2024-04-07T11:14:42.158987Z","shell.execute_reply.started":"2024-04-07T11:14:26.699883Z","shell.execute_reply":"2024-04-07T11:14:42.157718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What to fill and automatically fill nan","metadata":{}},{"cell_type":"code","source":"for col in drop_L_fea_cat.columns:\n    # Call the function for each column and update the original DataFrame\n    fill_method = suggest_fill_method(drop_L_fea_cat, col)\n    print(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\ndrop_L_fea_cat","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:42.160835Z","iopub.execute_input":"2024-04-07T11:14:42.161335Z","iopub.status.idle":"2024-04-07T11:14:43.483027Z","shell.execute_reply.started":"2024-04-07T11:14:42.161292Z","shell.execute_reply":"2024-04-07T11:14:43.482163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# M FEA (Group case id)","metadata":{}},{"cell_type":"code","source":"M_fea = [\"cancelreason_3545846M\",\"district_544M\",\"education_1138M\",\"postype_4733339M\",\"rejectreason_755M\",\"rejectreasonclient_4145042M\"]\nM_fea_table = applepre_feature_eng(data = drop_applprev_1_0, fealist = M_fea,agglist=['first'])\ndrop_M_fea = M_fea_table.drop(columns=\"first_postype_4733339M\")\ndrop_M_fea = M_fea_table.drop(columns=\"first_district_544M\")\ndrop_M_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:14:43.484401Z","iopub.execute_input":"2024-04-07T11:14:43.484733Z","iopub.status.idle":"2024-04-07T11:15:02.388977Z","shell.execute_reply.started":"2024-04-07T11:14:43.484706Z","shell.execute_reply":"2024-04-07T11:15:02.387542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in drop_M_fea.columns:\n# Call the function for each column and update the original DataFrame\n fill_method = suggest_fill_method(drop_M_fea, col)\nprint(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\ndrop_M_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:02.391144Z","iopub.execute_input":"2024-04-07T11:15:02.391644Z","iopub.status.idle":"2024-04-07T11:15:04.011133Z","shell.execute_reply.started":"2024-04-07T11:15:02.391602Z","shell.execute_reply":"2024-04-07T11:15:04.010011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndata = train_base.\\\n    merge(P_fea_table,on=['case_id'],how = 'left').\\\n    merge(drop_M_fea,on=['case_id'],how = 'left').\\\n    merge(drop_A_fea,on=['case_id'],how = 'left').\\\n    merge(drop_L_fea,on=['case_id'],how = 'left').\\\n    merge(drop_L_fea_cat,on=['case_id'],how = 'left')\ndata","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:04.012393Z","iopub.execute_input":"2024-04-07T11:15:04.013255Z","iopub.status.idle":"2024-04-07T11:15:07.495681Z","shell.execute_reply.started":"2024-04-07T11:15:04.013223Z","shell.execute_reply":"2024-04-07T11:15:07.494562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in data.columns:\n    # Call the function for each column and update the original DataFrame\n    fill_method = suggest_fill_method(data, col)\n    print(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\ndata","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:07.497338Z","iopub.execute_input":"2024-04-07T11:15:07.497783Z","iopub.status.idle":"2024-04-07T11:15:11.908977Z","shell.execute_reply.started":"2024-04-07T11:15:07.497743Z","shell.execute_reply":"2024-04-07T11:15:11.907784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_value = data['mean_tenor_203L'].mean() # Calculate the mode\n\n# Fill NaN values with the mode\ndata['mean_tenor_203L'].fillna(mean_value, inplace=True)\ndata\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:11.910693Z","iopub.execute_input":"2024-04-07T11:15:11.911150Z","iopub.status.idle":"2024-04-07T11:15:11.967247Z","shell.execute_reply.started":"2024-04-07T11:15:11.911109Z","shell.execute_reply":"2024-04-07T11:15:11.966067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature engineering","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming 'data' is your DataFrame\n\n# Calculate debt-to-monthly-annuity ratio\ndata['debt_to_monthly_annuity_ratio'] = data['mean_credamount_590A'] / data['mean_annuity_853A']\n\n# Calculate debt-to-income ratio\ndata['debt_to_income_ratio'] = data['mean_credamount_590A'] / data['mean_byoccupationinc_3656910L']\n\n# Drop the original columns used in calculations\ncolumns_to_drop = ['mean_credamount_590A', 'mean_annuity_853A', 'mean_byoccupationinc_3656910L']\ndata = data.drop(columns=columns_to_drop)\n\n# Display the updated DataFrame\nprint(data)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:11.968571Z","iopub.execute_input":"2024-04-07T11:15:11.968920Z","iopub.status.idle":"2024-04-07T11:15:12.245653Z","shell.execute_reply.started":"2024-04-07T11:15:11.968890Z","shell.execute_reply":"2024-04-07T11:15:12.244160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Columns to exclude from normalization\nexclude_cols = ['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target']\n# Columns to normalize\ncols_to_normalize = [col for col in data.columns if col not in exclude_cols]\nfor col in cols_to_normalize:\n    data[col] = label_encoder.fit_transform(data[col])\n    data[col] = data[col].astype(float)\ndata","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:12.248183Z","iopub.execute_input":"2024-04-07T11:15:12.249310Z","iopub.status.idle":"2024-04-07T11:15:17.741004Z","shell.execute_reply.started":"2024-04-07T11:15:12.249263Z","shell.execute_reply":"2024-04-07T11:15:17.739810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\n\n\n# Columns to exclude from normalization\nexclude_cols = ['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target']\n\n# Columns to normalize\ncols_to_normalize = [col for col in data.columns if col not in exclude_cols]\n\n# Apply z-score normalization to each column except excluded ones\nnormalized_data = data.copy()\nfor col in cols_to_normalize:\n    normalized_data[col] = (data[col] - data[col].mean()) / data[col].std()\n\nprint(\"\\nNormalized Data (Standardization):\")\nnormalized_data","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:15:17.742378Z","iopub.execute_input":"2024-04-07T11:15:17.743327Z","iopub.status.idle":"2024-04-07T11:15:18.590147Z","shell.execute_reply.started":"2024-04-07T11:15:17.743294Z","shell.execute_reply":"2024-04-07T11:15:18.588886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\n\nfrom sklearn.feature_selection import mutual_info_classif\n\n# Assuming 'date_decision' needs to be converted to datetime and extracted components\nnormalized_data['date_decision'] = pd.to_datetime(normalized_data['date_decision'])\nnormalized_data['year'] = normalized_data['date_decision'].dt.year\nnormalized_data['month'] = normalized_data['date_decision'].dt.month\nnormalized_data['day'] = normalized_data['date_decision'].dt.day\n\n# Assuming the target column is named 'target' and it's a classification problem\n# Adjust the features list if there are any columns to exclude\nfeatures = normalized_data.drop(['target'], axis=1).select_dtypes(include=[np.number])\ntarget = normalized_data['target']\n\n# Calculate mutual information scores\nmi_scores = mutual_info_classif(features, target, discrete_features='auto')\n\n# Create a Series for the MI scores\nmi_scores_series = pd.Series(mi_scores, index=features.columns)\n\n# Display the mutual information scores\nprint(mi_scores_series.sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:21:09.456726Z","iopub.execute_input":"2024-04-07T11:21:09.457157Z","iopub.status.idle":"2024-04-07T11:23:10.203895Z","shell.execute_reply.started":"2024-04-07T11:21:09.457114Z","shell.execute_reply":"2024-04-07T11:23:10.202178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nimport pandas as pd\n\n# Assuming 'data' is your DataFrame containing features and target\n# Define input and output features\n# Drop 'case_id' assuming it's an identifier column\nx = normalized_data.drop(['case_id', 'date_decision','target'], axis=1)  # Drop date column if it's not used as a feature\ny = normalized_data.target\n\n# Convert date column to datetime if needed\n# data['date_column'] = pd.to_datetime(data['date_column'])\n\n# Extract features from the date column if needed\n# For example, extract year, month, day, etc.\n# data['year'] = data['date_column'].dt.year\n# data['month'] = data['date_column'].dt.month\n# data['day'] = data['date_column'].dt.day\n\n# Train and test split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.33, random_state=42)\n\n# Define and train LightGBM classifier\nmodel = lgb.LGBMClassifier(learning_rate=0.09, random_state=42, verbosity=1)  # Adjust verbosity here\nmodel.fit(x_train, y_train, eval_set=[(x_test, y_test)], eval_metric='logloss')\n\n# Evaluate model performance\ntrain_accuracy = model.score(x_train, y_train)\ntest_accuracy = model.score(x_test, y_test)\n\nprint('Training accuracy: {:.4f}'.format(train_accuracy))\nprint('Testing accuracy: {:.4f}'.format(test_accuracy))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:20:52.000404Z","iopub.status.idle":"2024-04-07T11:20:52.000886Z","shell.execute_reply.started":"2024-04-07T11:20:52.000659Z","shell.execute_reply":"2024-04-07T11:20:52.000680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(model)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:20:52.002341Z","iopub.status.idle":"2024-04-07T11:20:52.002779Z","shell.execute_reply.started":"2024-04-07T11:20:52.002575Z","shell.execute_reply":"2024-04-07T11:20:52.002594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_data.to_csv('applple1_0-1_1.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:23:16.502673Z","iopub.execute_input":"2024-04-07T11:23:16.503139Z"},"trusted":true},"execution_count":null,"outputs":[]}]}