{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Purpose\n\nPurpose of this notebook is to find importance features of all separated train data. ML model is created by each train data, then train accuracy, validation accuracy, and feature importances are recorded. Output of this notebook will be used by other notebooks for modeling. ","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport re\nimport json\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-21T15:00:13.430364Z","iopub.execute_input":"2024-03-21T15:00:13.430723Z","iopub.status.idle":"2024-03-21T15:00:17.148811Z","shell.execute_reply.started":"2024-03-21T15:00:13.430694Z","shell.execute_reply":"2024-03-21T15:00:17.147183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nfeature_definition = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.151476Z","iopub.execute_input":"2024-03-21T15:00:17.152189Z","iopub.status.idle":"2024-03-21T15:00:17.182853Z","shell.execute_reply.started":"2024-03-21T15:00:17.152145Z","shell.execute_reply":"2024-03-21T15:00:17.181830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/\"\ntrain_files = os.listdir(train_dir)\nlen(train_files)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.184579Z","iopub.execute_input":"2024-03-21T15:00:17.185607Z","iopub.status.idle":"2024-03-21T15:00:17.204681Z","shell.execute_reply.started":"2024-03-21T15:00:17.185558Z","shell.execute_reply":"2024-03-21T15:00:17.203578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/\"\ntest_files = os.listdir(test_dir)\nlen(test_files)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.207830Z","iopub.execute_input":"2024-03-21T15:00:17.208547Z","iopub.status.idle":"2024-03-21T15:00:17.226462Z","shell.execute_reply.started":"2024-03-21T15:00:17.208505Z","shell.execute_reply":"2024-03-21T15:00:17.225192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base = pd.read_parquet(train_dir + \"train_base.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.228363Z","iopub.execute_input":"2024-03-21T15:00:17.228832Z","iopub.status.idle":"2024-03-21T15:00:17.620394Z","shell.execute_reply.started":"2024-03-21T15:00:17.228769Z","shell.execute_reply":"2024-03-21T15:00:17.619430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base[\"target\"].mean()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.621444Z","iopub.execute_input":"2024-03-21T15:00:17.621757Z","iopub.status.idle":"2024-03-21T15:00:17.633465Z","shell.execute_reply.started":"2024-03-21T15:00:17.621730Z","shell.execute_reply":"2024-03-21T15:00:17.632147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_files)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.635125Z","iopub.execute_input":"2024-03-21T15:00:17.636533Z","iopub.status.idle":"2024-03-21T15:00:17.643510Z","shell.execute_reply.started":"2024-03-21T15:00:17.636495Z","shell.execute_reply":"2024-03-21T15:00:17.642135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = np.sort(train_files)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.644725Z","iopub.execute_input":"2024-03-21T15:00:17.645160Z","iopub.status.idle":"2024-03-21T15:00:17.654184Z","shell.execute_reply.started":"2024-03-21T15:00:17.645118Z","shell.execute_reply":"2024-03-21T15:00:17.652595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#imputation\ndef inputation(data):\n    \n    for c in data.columns:\n        dtype1 = data[c].dtypes\n        filter1 = data[c].isna()\n        \n        #numeric\n        if (dtype1 == float) or (dtype1 == int): \n            if np.sum(filter1) > 0:\n                data.loc[filter1, c] = -255 #np.nanmean(data[c])\n        \n        #datetime\n        elif re.search(\"date\", c) is not None:\n            if np.sum(filter1) > 0:\n                data.loc[filter1, c] = '2030-01-01' #missing date is imputed by future datetime\n            data[c] = pd.to_datetime(data[c]).astype(\"int\")\n        \n        #categorical\n        else:\n            if np.sum(filter1) > 0:\n                data.loc[filter1, c] = \"Z\"\n\n            data[c] = pd.Categorical(data[c]).codes\n                \n\n                \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.655928Z","iopub.execute_input":"2024-03-21T15:00:17.656435Z","iopub.status.idle":"2024-03-21T15:00:17.667228Z","shell.execute_reply.started":"2024-03-21T15:00:17.656394Z","shell.execute_reply":"2024-03-21T15:00:17.666331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split data into train and validation\ndef createMLData(data):\n    \n    #data for ml\n    filter1 = data[\"target\"] == 1\n    n_pos_all = np.sum(filter1)\n    n_max = 12500\n    n_pos = min(n_max, n_pos_all)\n\n\n    data_p = data.loc[filter1,:].sample(n_pos)\n    data_n = data.loc[~filter1,:].sample(n_pos)\n    data_ml = pd.concat([data_p, data_n])\n\n    np.random.seed(1)\n\n    n_train = int(n_pos*2*0.8)\n    idx_all = np.arange(n_pos*2)\n    np.random.shuffle(idx_all)\n\n    train_idx = idx_all[:n_train]\n    val_idx = idx_all[n_train:]\n    train_ml = data_ml.iloc[train_idx]\n    val_ml = data_ml.iloc[val_idx]\n    \n    return train_ml, val_ml","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.671956Z","iopub.execute_input":"2024-03-21T15:00:17.672352Z","iopub.status.idle":"2024-03-21T15:00:17.682947Z","shell.execute_reply.started":"2024-03-21T15:00:17.672323Z","shell.execute_reply":"2024-03-21T15:00:17.681735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Modeling by RandomForest\ndef RF_Modeling(train_ml, val_ml):\n\n    # rf modeling\n\n    clf = RandomForestClassifier(max_depth=5, random_state=0)\n    clf.fit(train_ml.drop([\"target\"], axis = 1), train_ml[\"target\"])\n\n    yhat_train = clf.predict(train_ml.drop([\"target\"], axis = 1))\n    filter1 = yhat_train == train_ml[\"target\"].values\n    train_acc = np.mean(filter1)\n\n    yhat_val = clf.predict(val_ml.drop([\"target\"], axis = 1))\n    filter1 = yhat_val == val_ml[\"target\"].values\n    val_acc = np.mean(filter1)\n\n    print(\"train accuracy\", train_acc, \"val accuracy\", val_acc)\n\n    # importance\n\n    imp_df = pd.DataFrame({\"name\":list(train_ml.drop([\"target\"], axis = 1).columns), \"score\":clf.feature_importances_})\n    imp_df = imp_df.sort_values(\"score\", ascending = False).reset_index().drop([\"index\"], axis = 1)\n    \n    return train_acc, val_acc, imp_df","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.684647Z","iopub.execute_input":"2024-03-21T15:00:17.685360Z","iopub.status.idle":"2024-03-21T15:00:17.698849Z","shell.execute_reply.started":"2024-03-21T15:00:17.685315Z","shell.execute_reply":"2024-03-21T15:00:17.697659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Screening","metadata":{}},{"cell_type":"code","source":"model_acc_files = {}\nN = train_files.shape[0]\n#N = 4\n\nfor ID in range(N):\n    \n    if train_files[ID] == \"train_base.parquet\":\n        continue\n        \n    # read feature file\n    data = pd.read_parquet(train_dir + train_files[ID])\n\n    # merge with label data\n    data = data.merge(train_base[[\"case_id\", \"target\"]], on = \"case_id\", how = \"left\")\n\n    data.drop([\"case_id\"], axis = 1, inplace = True)\n\n    # inputation\n    data = inputation(data)\n\n    train_ml, val_ml = createMLData(data)\n\n    train_acc, val_acc, imp_df = RF_Modeling(train_ml, val_ml)\n\n    fname = train_files[ID].split(\".\")[0]\n    imp_fname = fname + \"_FeatureImportance.csv\"\n    fname, imp_fname\n\n    model_acc_files[fname] = {\"train_accuracy\": train_acc, \"val_accuracy\":val_acc}\n\n    imp_df.to_csv(imp_fname, index = False)\n\n    \n    print(fname, \" done\")\n    \nwith open('file_accuracy.json', 'w',  encoding='utf-8') as f:\n    json.dump(model_acc_files, f)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:00:17.699942Z","iopub.execute_input":"2024-03-21T15:00:17.700285Z","iopub.status.idle":"2024-03-21T15:03:22.699027Z","shell.execute_reply.started":"2024-03-21T15:00:17.700256Z","shell.execute_reply":"2024-03-21T15:03:22.697295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_files[ID]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.700192Z","iopub.status.idle":"2024-03-21T15:03:22.700636Z","shell.execute_reply.started":"2024-03-21T15:03:22.700435Z","shell.execute_reply":"2024-03-21T15:03:22.700453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_list = []\ntrain_acc_list = []\nval_acc_list = []\nfor key in model_acc_files.keys():\n    #print(key)\n    train_acc = model_acc_files[key][\"train_accuracy\"]\n    val_acc = model_acc_files[key][\"val_accuracy\"]\n    \n    file_list.append(key)\n    train_acc_list.append(train_acc)\n    val_acc_list.append(val_acc)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.702226Z","iopub.status.idle":"2024-03-21T15:03:22.702691Z","shell.execute_reply.started":"2024-03-21T15:03:22.702493Z","shell.execute_reply":"2024-03-21T15:03:22.702511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_df = pd.DataFrame({\"file\": file_list, \"train_accuracy\": train_acc_list, \"val_accuracy\": val_acc_list})","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.704744Z","iopub.status.idle":"2024-03-21T15:03:22.705453Z","shell.execute_reply.started":"2024-03-21T15:03:22.705123Z","shell.execute_reply":"2024-03-21T15:03:22.705148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_df = accuracy_df.sort_values(\"val_accuracy\", ascending = False).reset_index().drop([\"index\"], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.708430Z","iopub.status.idle":"2024-03-21T15:03:22.709618Z","shell.execute_reply.started":"2024-03-21T15:03:22.709363Z","shell.execute_reply":"2024-03-21T15:03:22.709388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_df.to_csv(\"accuracy_file.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.710666Z","iopub.status.idle":"2024-03-21T15:03:22.711184Z","shell.execute_reply.started":"2024-03-21T15:03:22.710980Z","shell.execute_reply":"2024-03-21T15:03:22.710999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_df","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.713124Z","iopub.status.idle":"2024-03-21T15:03:22.713575Z","shell.execute_reply.started":"2024-03-21T15:03:22.713370Z","shell.execute_reply":"2024-03-21T15:03:22.713389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#NOT USED\ncategory = [\"applprev\",\n\"credit_bureau\",\n\"debitcard_1\",\n\"deposit_1\",\n\"person\",\n\"static\",\n\"tax_registry\",\n\"other\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T15:03:22.715218Z","iopub.status.idle":"2024-03-21T15:03:22.715656Z","shell.execute_reply.started":"2024-03-21T15:03:22.715450Z","shell.execute_reply":"2024-03-21T15:03:22.715470Z"},"trusted":true},"execution_count":null,"outputs":[]}]}