{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-13T06:43:58.851812Z","iopub.execute_input":"2024-03-13T06:43:58.852779Z","iopub.status.idle":"2024-03-13T06:44:01.904715Z","shell.execute_reply.started":"2024-03-13T06:43:58.852741Z","shell.execute_reply":"2024-03-13T06:44:01.903766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"applprev1 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_1.parquet')\n\n# find percent missing values of each columns\npercent_missing = applprev1.isnull().sum() * 100 / len(applprev1)\nmissing_val = pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_val[(missing_val['percent_missing'] > 80)].index\n#drop_80 = missing_val.drop(morethan_80, inplace=True)\n\n#columns_to_drop = [col for col in applprev1 if col in applprev1.columns and (missing_val[col] == percent_missing>80).all()]\ndf = applprev1.drop(morethan_80, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-13T07:13:43.380157Z","iopub.execute_input":"2024-03-13T07:13:43.380584Z","iopub.status.idle":"2024-03-13T07:13:51.970261Z","shell.execute_reply.started":"2024-03-13T07:13:43.380552Z","shell.execute_reply":"2024-03-13T07:13:51.969323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"morethan_80","metadata":{"execution":{"iopub.status.busy":"2024-03-13T07:13:51.972010Z","iopub.execute_input":"2024-03-13T07:13:51.972385Z","iopub.status.idle":"2024-03-13T07:13:51.980354Z","shell.execute_reply.started":"2024-03-13T07:13:51.972328Z","shell.execute_reply":"2024-03-13T07:13:51.978966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-03-13T07:13:53.674299Z","iopub.execute_input":"2024-03-13T07:13:53.674779Z","iopub.status.idle":"2024-03-13T07:13:55.340701Z","shell.execute_reply.started":"2024-03-13T07:13:53.674736Z","shell.execute_reply":"2024-03-13T07:13:55.339543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"applprev1.describe()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:44:34.874223Z","iopub.execute_input":"2024-03-13T06:44:34.874785Z","iopub.status.idle":"2024-03-13T06:44:37.800491Z","shell.execute_reply.started":"2024-03-13T06:44:34.874736Z","shell.execute_reply":"2024-03-13T06:44:37.799163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_val.iloc[:]\n","metadata":{"execution":{"iopub.status.busy":"2024-03-11T15:35:02.241067Z","iopub.execute_input":"2024-03-11T15:35:02.241920Z","iopub.status.idle":"2024-03-11T15:35:02.259868Z","shell.execute_reply.started":"2024-03-11T15:35:02.241854Z","shell.execute_reply":"2024-03-11T15:35:02.258960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#applprev1_1\n\napplprev1_1 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_1.parquet')\n\n# find percent missing values of each columns\npercent_missing = applprev1_1.isnull().sum() * 100 / len(applprev1_1)\nmissing_val1 = pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_val1[(missing_val1['percent_missing'] > 80)].index\ndrop_80 = missing_val1.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:29:38.996787Z","iopub.execute_input":"2024-03-11T14:29:38.997180Z","iopub.status.idle":"2024-03-11T14:29:47.220359Z","shell.execute_reply.started":"2024-03-11T14:29:38.997151Z","shell.execute_reply":"2024-03-11T14:29:47.218900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_val1.iloc[:]","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:29:50.094092Z","iopub.execute_input":"2024-03-11T14:29:50.094475Z","iopub.status.idle":"2024-03-11T14:29:50.107346Z","shell.execute_reply.started":"2024-03-11T14:29:50.094447Z","shell.execute_reply":"2024-03-11T14:29:50.106247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#applprev2\n\napplprev2 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_2.parquet')\n\n# find percent missing values of each columns\npercent_missing = applprev2.isnull().sum() * 100 / len(applprev2)\nmissing_val2 = pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_val2[(missing_val2['percent_missing'] > 80)].index\ndrop_80 = missing_val2.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:30:00.309947Z","iopub.execute_input":"2024-03-11T14:30:00.310320Z","iopub.status.idle":"2024-03-11T14:30:06.002962Z","shell.execute_reply.started":"2024-03-11T14:30:00.310293Z","shell.execute_reply":"2024-03-11T14:30:06.001898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_val2.iloc[:]","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:30:06.004563Z","iopub.execute_input":"2024-03-11T14:30:06.004927Z","iopub.status.idle":"2024-03-11T14:30:06.015659Z","shell.execute_reply.started":"2024-03-11T14:30:06.004899Z","shell.execute_reply":"2024-03-11T14:30:06.014701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#debitcard\n\ndebitcard = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_debitcard_1.parquet')\n\n# find percent missing values of each columns\npercent_missing = debitcard.isnull().sum() * 100 / len(debitcard)\nmissing_debitcard= pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_debitcard[(missing_debitcard['percent_missing'] > 80)].index\ndrop_80 = missing_debitcard.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:29:20.693430Z","iopub.execute_input":"2024-03-11T14:29:20.693845Z","iopub.status.idle":"2024-03-11T14:29:20.750802Z","shell.execute_reply.started":"2024-03-11T14:29:20.693814Z","shell.execute_reply":"2024-03-11T14:29:20.749573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_debitcard.iloc[:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#deposit\n\ndeposit = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_deposit_1.parquet')\n\n# find percent missing values of each columns\npercent_missing = deposit.isnull().sum() * 100 / len(deposit)\nmissing_deposit= pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_deposit[(missing_deposit['percent_missing'] > 80)].index\ndrop_80 = missing_deposit.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:28:06.647406Z","iopub.execute_input":"2024-03-11T14:28:06.647837Z","iopub.status.idle":"2024-03-11T14:28:06.711220Z","shell.execute_reply.started":"2024-03-11T14:28:06.647804Z","shell.execute_reply":"2024-03-11T14:28:06.710042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_deposit.iloc[:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#other\n\nother = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_other_1.parquet')\n\n# find percent missing values of each columns\npercent_missing = other.isnull().sum() * 100 / len(other)\nmissing_other= pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_other[(missing_other['percent_missing'] > 80)].index\ndrop_80 = missing_other.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:27:27.239792Z","iopub.execute_input":"2024-03-11T14:27:27.240235Z","iopub.status.idle":"2024-03-11T14:27:27.259474Z","shell.execute_reply.started":"2024-03-11T14:27:27.240202Z","shell.execute_reply":"2024-03-11T14:27:27.258216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_other.iloc[:]","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:27:29.022248Z","iopub.execute_input":"2024-03-11T14:27:29.022666Z","iopub.status.idle":"2024-03-11T14:27:29.034951Z","shell.execute_reply.started":"2024-03-11T14:27:29.022623Z","shell.execute_reply":"2024-03-11T14:27:29.033799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#person\n\nperson= pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_person_1.parquet')\n\n# find percent missing values of each columns\npercent_missing = person.isnull().sum() * 100 / len(person)\nmissing_person= pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_person[(missing_person['percent_missing'] > 80)].index\ndrop_80 = missing_person.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:27:38.685383Z","iopub.execute_input":"2024-03-11T14:27:38.685774Z","iopub.status.idle":"2024-03-11T14:27:47.678299Z","shell.execute_reply.started":"2024-03-11T14:27:38.685743Z","shell.execute_reply":"2024-03-11T14:27:47.677041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_person.iloc[:]","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:27:53.609544Z","iopub.execute_input":"2024-03-11T14:27:53.610067Z","iopub.status.idle":"2024-03-11T14:27:53.623862Z","shell.execute_reply.started":"2024-03-11T14:27:53.610024Z","shell.execute_reply":"2024-03-11T14:27:53.622807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#person2\n\nperson2= pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_person_2.parquet')\n\n# find percent missing values of each columns\npercent_missing = person.isnull().sum() * 100 / len(person2)\nmissing_person2= pd.DataFrame({'percent_missing': percent_missing})\n\n# remove columns with percent missing value > 80\nmorethan_80 = missing_person2[(missing_person2['percent_missing'] > 80)].index\ndrop_80 = missing_person2.drop(morethan_80, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:33:35.852978Z","iopub.execute_input":"2024-03-11T14:33:35.853362Z","iopub.status.idle":"2024-03-11T14:33:41.961625Z","shell.execute_reply.started":"2024-03-11T14:33:35.853334Z","shell.execute_reply":"2024-03-11T14:33:41.960454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_person2.iloc[:]","metadata":{"execution":{"iopub.status.busy":"2024-03-11T14:33:45.820765Z","iopub.execute_input":"2024-03-11T14:33:45.821230Z","iopub.status.idle":"2024-03-11T14:33:45.833705Z","shell.execute_reply.started":"2024-03-11T14:33:45.821198Z","shell.execute_reply":"2024-03-11T14:33:45.832365Z"},"trusted":true},"execution_count":null,"outputs":[]}]}