{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-07T13:02:58.342215Z","iopub.execute_input":"2024-04-07T13:02:58.342640Z","iopub.status.idle":"2024-04-07T13:02:58.364929Z","shell.execute_reply.started":"2024-04-07T13:02:58.342607Z","shell.execute_reply":"2024-04-07T13:02:58.362891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_person_1= pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_person_1.parquet')\ntrain_base = pd.read_parquet(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_base.parquet\")\n# combine train_applprev_1_0 + train_applprev_1_1\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:02:58.367895Z","iopub.execute_input":"2024-04-07T13:02:58.368500Z","iopub.status.idle":"2024-04-07T13:03:01.747733Z","shell.execute_reply.started":"2024-04-07T13:02:58.368452Z","shell.execute_reply":"2024-04-07T13:03:01.746695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\n# Load the DataFrame\n#applprev_1_0 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_0.parquet')\n\n# Set seed for reproducibility\nnp.random.seed(0)\n\n# Calculate percent missing values of each column\npercent_missing = train_person_1.isnull().sum() * 100 / len(train_person_1)\nmissing_applprev = pd.DataFrame({'percent_missing': percent_missing})\n\n# Identify columns with percent missing value > 80\nmorethan_80 = missing_applprev[missing_applprev['percent_missing'] > 80].index\n\n# Drop columns with percent missing value > 80\ntrain_person_1 = train_person_1.drop(columns=morethan_80)\ntrain_person_1\n# Fill NaN values in the column \"actualdpd_943P\" with 0\n#drop_applprev_1_0[\"actualdpd_943P\"].fillna(0, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:01.749814Z","iopub.execute_input":"2024-04-07T13:03:01.750696Z","iopub.status.idle":"2024-04-07T13:03:08.372908Z","shell.execute_reply.started":"2024-04-07T13:03:01.750652Z","shell.execute_reply":"2024-04-07T13:03:08.371458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndef suggest_fill_method(dataFrame, col):\n\n    # Check if the column contains numerical or categorical data\n    if pd.api.types.is_numeric_dtype(dataFrame[col]):\n        # If numerical, check if the data is skewed\n        skewness = dataFrame[col].skew()\n        if abs(skewness) > 1:\n            fill_method = 'median'  # If highly skewed, suggest filling with median\n        else:\n            fill_method = 'mean'  # If not highly skewed, suggest filling with mean\n    else:\n        fill_method = 'mode'  # If categorical, suggest filling with mode\n    \n    # Fill missing values with the suggested method\n    if fill_method == 'mode':\n        dataFrame[col].fillna(dataFrame[col].mode()[0], inplace=True)\n    elif fill_method == 'median':\n        dataFrame[col].fillna(dataFrame[col].median(), inplace=True)\n\n    return fill_method","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:08.374979Z","iopub.execute_input":"2024-04-07T13:03:08.375952Z","iopub.status.idle":"2024-04-07T13:03:08.385625Z","shell.execute_reply.started":"2024-04-07T13:03:08.375904Z","shell.execute_reply":"2024-04-07T13:03:08.384065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature engineering function demo and testing\ndef applepre_feature_eng(data,fealist,agglist):\n\n    result = pd.DataFrame({'case_id':list(set(data.case_id))})\n    for f in fealist:\n        print(f,'......')\n        df_table = data.loc[:,['case_id',f,'num_group1']].\\\n            sort_values(['case_id','num_group1']).\\\n            groupby(['case_id'])[f].\\\n            agg(agglist).\\\n            add_suffix('_'+ f).\\\n            reset_index()\n        result = result.merge(df_table,on=['case_id'],how='left')\n    return result","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:08.388419Z","iopub.execute_input":"2024-04-07T13:03:08.388798Z","iopub.status.idle":"2024-04-07T13:03:08.402137Z","shell.execute_reply.started":"2024-04-07T13:03:08.388767Z","shell.execute_reply":"2024-04-07T13:03:08.400587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Iterate through columns and drop columns ending with 'D'\nfor col in train_person_1.columns:\n    if col.endswith(\"D\"):\n        train_person_1.drop(col, axis=1, inplace=True)\n#Print the DataFrame\npd.set_option('display.max_columns', None)\ntrain_person_1","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:08.406526Z","iopub.execute_input":"2024-04-07T13:03:08.407181Z","iopub.status.idle":"2024-04-07T13:03:09.372871Z","shell.execute_reply.started":"2024-04-07T13:03:08.407131Z","shell.execute_reply":"2024-04-07T13:03:09.371664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndef suggest_fill_method(dataFrame, col):\n\n    # Check if the column contains numerical or categorical data\n    if pd.api.types.is_numeric_dtype(dataFrame[col]):\n        # If numerical, check if the data is skewed\n        skewness = dataFrame[col].skew()\n        if abs(skewness) > 1:\n            fill_method = 'median'  # If highly skewed, suggest filling with median\n        else:\n            fill_method = 'mean'  # If not highly skewed, suggest filling with mean\n    else:\n        fill_method = 'mode'  # If categorical, suggest filling with mode\n    \n    # Fill missing values with the suggested method\n    if fill_method == 'mode':\n        dataFrame[col].fillna(dataFrame[col].mode()[0], inplace=True)\n    elif fill_method == 'median':\n        dataFrame[col].fillna(dataFrame[col].median(), inplace=True)\n\n\n    return fill_method","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:09.374188Z","iopub.execute_input":"2024-04-07T13:03:09.374518Z","iopub.status.idle":"2024-04-07T13:03:09.383113Z","shell.execute_reply.started":"2024-04-07T13:03:09.374490Z","shell.execute_reply":"2024-04-07T13:03:09.382092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fea = [\"contaddr_district_15M\", \"contaddr_matchlist_1032L\", \"contaddr_zipcode_807M\",\n       \"empladdr_district_926M\", \"empladdr_zipcode_114M\", \"registaddr_district_1083M\",\n       \"persontype_1072L\", \"sex_738L\"]\n\ndrop_fea = train_person_1.drop(columns=fea)\ndrop_fea","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:09.384340Z","iopub.execute_input":"2024-04-07T13:03:09.385301Z","iopub.status.idle":"2024-04-07T13:03:09.853752Z","shell.execute_reply.started":"2024-04-07T13:03:09.385268Z","shell.execute_reply":"2024-04-07T13:03:09.852659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"M_fea = [\"education_927M\",\"registaddr_zipcode_184M\"]\nM_fea_table = applepre_feature_eng(data = drop_fea, fealist = M_fea,agglist=['first'])\nM_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:09.854889Z","iopub.execute_input":"2024-04-07T13:03:09.855862Z","iopub.status.idle":"2024-04-07T13:03:14.072372Z","shell.execute_reply.started":"2024-04-07T13:03:09.855828Z","shell.execute_reply":"2024-04-07T13:03:14.071093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in M_fea_table.columns:\n# Call the function for each column and update the original DataFrame\n fill_method = suggest_fill_method(M_fea_table, col)\nprint(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\nM_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:14.073790Z","iopub.execute_input":"2024-04-07T13:03:14.074128Z","iopub.status.idle":"2024-04-07T13:03:14.850252Z","shell.execute_reply.started":"2024-04-07T13:03:14.074101Z","shell.execute_reply":"2024-04-07T13:03:14.848924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check that there are no missing values\npercent_missing = M_fea_table.isnull().sum() * 100 / len(M_fea_table)\npercent_missing","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:14.851767Z","iopub.execute_input":"2024-04-07T13:03:14.852201Z","iopub.status.idle":"2024-04-07T13:03:15.198306Z","shell.execute_reply.started":"2024-04-07T13:03:14.852167Z","shell.execute_reply":"2024-04-07T13:03:15.196798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"L_fea = [\"personindex_1023L\",\"persontype_792L\"]\nL_fea_table = applepre_feature_eng(data = drop_fea, fealist = L_fea,agglist=['mean'])\nL_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:15.199791Z","iopub.execute_input":"2024-04-07T13:03:15.200207Z","iopub.status.idle":"2024-04-07T13:03:18.101730Z","shell.execute_reply.started":"2024-04-07T13:03:15.200178Z","shell.execute_reply":"2024-04-07T13:03:18.100494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in L_fea_table.columns:\n# Call the function for each column and update the original DataFrame\n fill_method = suggest_fill_method(L_fea_table, col)\nprint(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\nL_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:18.103415Z","iopub.execute_input":"2024-04-07T13:03:18.103777Z","iopub.status.idle":"2024-04-07T13:03:18.191975Z","shell.execute_reply.started":"2024-04-07T13:03:18.103747Z","shell.execute_reply":"2024-04-07T13:03:18.190756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check that there are no missing values\npercent_missing = L_fea_table.isnull().sum() * 100 / len(L_fea_table)\npercent_missing","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:18.199050Z","iopub.execute_input":"2024-04-07T13:03:18.199421Z","iopub.status.idle":"2024-04-07T13:03:18.215757Z","shell.execute_reply.started":"2024-04-07T13:03:18.199391Z","shell.execute_reply":"2024-04-07T13:03:18.214356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"L_fea = [\"contaddr_smempladdr_334L\",\"familystate_447L\",\"remitter_829L\",\"role_1084L\",\"safeguarantyflag_411L\",\"type_25L\"]\nL_fea_cat = applepre_feature_eng(data = drop_fea, fealist = L_fea,agglist=['first'])\nL_fea_cat","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:18.217167Z","iopub.execute_input":"2024-04-07T13:03:18.217736Z","iopub.status.idle":"2024-04-07T13:03:28.276671Z","shell.execute_reply.started":"2024-04-07T13:03:18.217683Z","shell.execute_reply":"2024-04-07T13:03:28.275310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in L_fea_cat.columns:\n# Call the function for each column and update the original DataFrame\n fill_method = suggest_fill_method(L_fea_cat, col)\nprint(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\nL_fea_cat","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:28.278421Z","iopub.execute_input":"2024-04-07T13:03:28.278799Z","iopub.status.idle":"2024-04-07T13:03:30.636638Z","shell.execute_reply.started":"2024-04-07T13:03:28.278770Z","shell.execute_reply":"2024-04-07T13:03:30.635310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"T_fea = [\"incometype_1044T\",\"relationshiptoclient_642T\",\"relationshiptoclient_415T\"]\nT_fea_table = applepre_feature_eng(data = drop_fea, fealist = T_fea,agglist=['first'])\nT_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:30.638224Z","iopub.execute_input":"2024-04-07T13:03:30.638550Z","iopub.status.idle":"2024-04-07T13:03:35.746965Z","shell.execute_reply.started":"2024-04-07T13:03:30.638523Z","shell.execute_reply":"2024-04-07T13:03:35.745731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in T_fea_table.columns:\n# Call the function for each column and update the original DataFrame\n fill_method = suggest_fill_method(T_fea_table, col)\nprint(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\nT_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:35.750667Z","iopub.execute_input":"2024-04-07T13:03:35.751055Z","iopub.status.idle":"2024-04-07T13:03:36.586113Z","shell.execute_reply.started":"2024-04-07T13:03:35.751026Z","shell.execute_reply":"2024-04-07T13:03:36.584782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A_fea = [\"mainoccupationinc_384A\"]\nA_fea_table = applepre_feature_eng(data = drop_fea, fealist = A_fea,agglist=['first'])\nA_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:36.587937Z","iopub.execute_input":"2024-04-07T13:03:36.588577Z","iopub.status.idle":"2024-04-07T13:03:38.708343Z","shell.execute_reply.started":"2024-04-07T13:03:36.588531Z","shell.execute_reply":"2024-04-07T13:03:38.707055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in A_fea_table.columns:\n# Call the function for each column and update the original DataFrame\n fill_method = suggest_fill_method(A_fea_table, col)\nprint(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\nA_fea_table","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:38.710373Z","iopub.execute_input":"2024-04-07T13:03:38.710761Z","iopub.status.idle":"2024-04-07T13:03:38.791411Z","shell.execute_reply.started":"2024-04-07T13:03:38.710729Z","shell.execute_reply":"2024-04-07T13:03:38.790051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train_base.\\\n    merge(T_fea_table,on=['case_id'],how = 'left').\\\n    merge(M_fea_table,on=['case_id'],how = 'left').\\\n    merge(A_fea_table,on=['case_id'],how = 'left').\\\n    merge(L_fea_table,on=['case_id'],how = 'left').\\\n    merge(L_fea_cat,on=['case_id'],how = 'left')\ndata","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:38.793462Z","iopub.execute_input":"2024-04-07T13:03:38.793929Z","iopub.status.idle":"2024-04-07T13:03:40.634068Z","shell.execute_reply.started":"2024-04-07T13:03:38.793887Z","shell.execute_reply":"2024-04-07T13:03:40.632771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in data.columns:\n    # Call the function for each column and update the original DataFrame\n    fill_method = suggest_fill_method(data, col)\n    print(f\"For column '{col}', suggested fill method:\", fill_method)\n\n# Now cleaned_P has been updated with the filled missing values\nprint(\"Updated DataFrame:\")\ndata","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:40.635671Z","iopub.execute_input":"2024-04-07T13:03:40.636148Z","iopub.status.idle":"2024-04-07T13:03:44.365563Z","shell.execute_reply.started":"2024-04-07T13:03:40.636106Z","shell.execute_reply":"2024-04-07T13:03:44.364298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Columns to exclude from normalization\nexclude_cols = ['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target']\n# Columns to normalize\ncols_to_normalize = [col for col in data.columns if col not in exclude_cols]\nfor col in cols_to_normalize:\n    data[col] = label_encoder.fit_transform(data[col])\n    data[col] = data[col].astype(float)\ndata","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:44.367031Z","iopub.execute_input":"2024-04-07T13:03:44.367356Z","iopub.status.idle":"2024-04-07T13:03:48.270358Z","shell.execute_reply.started":"2024-04-07T13:03:44.367326Z","shell.execute_reply":"2024-04-07T13:03:48.269017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\n\n\n# Columns to exclude from normalization\nexclude_cols = ['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target']\n\n# Columns to normalize\ncols_to_normalize = [col for col in data.columns if col not in exclude_cols]\n\n# Apply z-score normalization to each column except excluded ones\nnormalized_data = data.copy()\nfor col in cols_to_normalize:\n    normalized_data[col] = (data[col] - data[col].mean()) / data[col].std()\n\nprint(\"\\nNormalized Data (Standardization):\")\nnormalized_data","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:48.271959Z","iopub.execute_input":"2024-04-07T13:03:48.272865Z","iopub.status.idle":"2024-04-07T13:03:49.000966Z","shell.execute_reply.started":"2024-04-07T13:03:48.272831Z","shell.execute_reply":"2024-04-07T13:03:48.999851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_data.drop(columns=['first_remitter_829L'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:03:49.002761Z","iopub.execute_input":"2024-04-07T13:03:49.003262Z","iopub.status.idle":"2024-04-07T13:03:49.148534Z","shell.execute_reply.started":"2024-04-07T13:03:49.003216Z","shell.execute_reply":"2024-04-07T13:03:49.147278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_data.to_csv('/kaggle/working/train_person1.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nimport pandas as pd\n\n# Assuming 'data' is your DataFrame containing features and target\n# Define input and output features'target\n# Drop 'case_id' assuming it's an identifier column\nx = normalized_data.drop(['case_id', 'date_decision','target'], axis=1)  # Drop date column if it's not used as a feature\ny = normalized_data.target\n\n# Convert date column to datetime if needed\n# data['date_column'] = pd.to_datetime(data['date_column'])\n\n# Extract features from the date column if needed\n# For example, extract year, month, day, etc.\n# data['year'] = data['date_column'].dt.year\n# data['month'] = data['date_column'].dt.month\n# data['day'] = data['date_column'].dt.day\n\n# Train and test split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.33, random_state=42)\n\n# Define and train LightGBM classifier\nmodel = lgb.LGBMClassifier(learning_rate=0.09, random_state=42, verbosity=1)  # Adjust verbosity here\nmodel.fit(x_train, y_train, eval_set=[(x_test, y_test)], eval_metric='logloss')\n\n# Evaluate model performance\ntrain_accuracy = model.score(x_train, y_train)\ntest_accuracy = model.score(x_test, y_test)\n\nprint('Training accuracy: {:.4f}'.format(train_accuracy))\nprint('Testing accuracy: {:.4f}'.format(test_accuracy))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:04:39.624402Z","iopub.execute_input":"2024-04-07T13:04:39.625815Z","iopub.status.idle":"2024-04-07T13:05:15.451059Z","shell.execute_reply.started":"2024-04-07T13:04:39.625764Z","shell.execute_reply":"2024-04-07T13:05:15.449616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(model)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:05:15.455296Z","iopub.execute_input":"2024-04-07T13:05:15.455736Z","iopub.status.idle":"2024-04-07T13:05:15.897824Z","shell.execute_reply.started":"2024-04-07T13:05:15.455687Z","shell.execute_reply":"2024-04-07T13:05:15.896561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\n\nfrom sklearn.feature_selection import mutual_info_classif\n\n# Assuming 'date_decision' needs to be converted to datetime and extracted components\nnormalized_data['date_decision'] = pd.to_datetime(normalized_data['date_decision'])\nnormalized_data['year'] = normalized_data['date_decision'].dt.year\nnormalized_data['month'] = normalized_data['date_decision'].dt.month\nnormalized_data['day'] = normalized_data['date_decision'].dt.day\n\n# Assuming the target column is named 'target' and it's a classification problem\n# Adjust the features list if there are any columns to exclude\nfeatures = normalized_data.drop(['target'], axis=1).select_dtypes(include=[np.number])\ntarget = normalized_data['target']\n\n# Calculate mutual information scores\nmi_scores = mutual_info_classif(features, target, discrete_features='auto')\n\n# Create a Series for the MI scores\nmi_scores_series = pd.Series(mi_scores, index=features.columns)\n\n# Display the mutual information scores\nprint(mi_scores_series.sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:05:15.899175Z","iopub.execute_input":"2024-04-07T13:05:15.899506Z","iopub.status.idle":"2024-04-07T13:10:12.101232Z","shell.execute_reply.started":"2024-04-07T13:05:15.899478Z","shell.execute_reply":"2024-04-07T13:10:12.099776Z"},"trusted":true},"execution_count":null,"outputs":[]}]}