{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":169163610,"sourceType":"kernelVersion"},{"sourceId":169182250,"sourceType":"kernelVersion"},{"sourceId":169187922,"sourceType":"kernelVersion"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport re\nimport json\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn import metrics\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-02T07:13:07.304361Z","iopub.execute_input":"2024-04-02T07:13:07.304807Z","iopub.status.idle":"2024-04-02T07:13:10.483536Z","shell.execute_reply.started":"2024-04-02T07:13:07.304767Z","shell.execute_reply":"2024-04-02T07:13:10.482645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_json(fname):\n    with open(fname, 'r',  encoding='utf-8') as f:\n        data = json.load(f)\n        \n    return data\n\ndef write_json(data_json, fname):\n    fname_json = fname + \".json\"\n    with open(fname_json, 'w',  encoding='utf-8') as f:\n        json.dump(data_json, f)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.485242Z","iopub.execute_input":"2024-04-02T07:13:10.485701Z","iopub.status.idle":"2024-04-02T07:13:10.492085Z","shell.execute_reply.started":"2024-04-02T07:13:10.485673Z","shell.execute_reply":"2024-04-02T07:13:10.490728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createMLData(data):\n    \n    np.random.seed(1)\n    #data for ml\n    filter1 = data[\"target\"] == 1\n    n_pos_all = np.sum(filter1)\n    n_max = 20000\n    n_pos = min(n_max, n_pos_all)\n\n\n    data_p = data.loc[filter1,:].sample(n_pos)\n    data_n = data.loc[~filter1,:].sample(n_pos)\n    data_ml = pd.concat([data_p, data_n])\n    \n    return data_ml","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.493625Z","iopub.execute_input":"2024-04-02T07:13:10.494005Z","iopub.status.idle":"2024-04-02T07:13:10.504672Z","shell.execute_reply.started":"2024-04-02T07:13:10.493971Z","shell.execute_reply":"2024-04-02T07:13:10.503507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imputation_execute():\n    \n    for c in cnames:\n        if c == \"target\":\n            continue\n\n        dtype1 = train_df[c].dtype\n\n        if dtype1 == object:\n\n            #if False:\n            filter_na = train_df[c].isna()\n            train_df.loc[filter_na, c] = \"NoCaseID\"\n\n            filter_na = val_df[c].isna()\n            val_df.loc[filter_na, c] = \"NoCaseID\"\n\n            train_df[c] = pd.Categorical(train_df[c]).codes\n            val_df[c] = pd.Categorical(val_df[c]).codes\n\n        else:\n            dtype2, imputation_null, imputation_noid = imputation_plan[subcategories[ID]][c]\n            #print(imputation_plan[subcategories[ID]][c])\n\n            filter_na = train_df[c].isna()\n            train_df.loc[filter_na, c] = imputation_noid\n\n            filter_na = val_df[c].isna()\n            val_df.loc[filter_na, c] = imputation_noid\n            \n    return train_df, val_df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.506798Z","iopub.execute_input":"2024-04-02T07:13:10.507095Z","iopub.status.idle":"2024-04-02T07:13:10.519666Z","shell.execute_reply.started":"2024-04-02T07:13:10.507072Z","shell.execute_reply":"2024-04-02T07:13:10.518567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ML_execute():\n    \n    clf = RandomForestClassifier(max_depth=10, random_state=0)\n    clf.fit(train_ml.drop([\"target\"], axis = 1), train_ml[\"target\"])\n\n    yhat_train = clf.predict_proba(train_ml.drop([\"target\"], axis = 1))[:,1]\n    yhat_val = clf.predict_proba(val_df.drop([\"target\"], axis = 1))[:,1]\n\n    fpr, tpr, thresholds = metrics.roc_curve(train_ml[\"target\"], yhat_train, pos_label=1)\n    train_auc = metrics.auc(fpr, tpr)\n    train_auc\n\n    fpr, tpr, thresholds = metrics.roc_curve(val_df[\"target\"], yhat_val, pos_label=1)\n    val_auc = metrics.auc(fpr, tpr)\n    val_auc\n    \n    \n    imp_df = pd.DataFrame({\"name\":list(train_ml.drop([\"target\"], axis = 1).columns), \"score\":clf.feature_importances_})\n    imp_df = imp_df.sort_values(\"score\", ascending = False).reset_index().drop([\"index\"], axis = 1)\n    \n    return train_auc, val_auc, imp_df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.521136Z","iopub.execute_input":"2024-04-02T07:13:10.521617Z","iopub.status.idle":"2024-04-02T07:13:10.532432Z","shell.execute_reply.started":"2024-04-02T07:13:10.521578Z","shell.execute_reply":"2024-04-02T07:13:10.531345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.534058Z","iopub.execute_input":"2024-04-02T07:13:10.534486Z","iopub.status.idle":"2024-04-02T07:13:10.545507Z","shell.execute_reply.started":"2024-04-02T07:13:10.534451Z","shell.execute_reply":"2024-04-02T07:13:10.544679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_depth = read_json(\"/kaggle/input/homecredit-filesbycategory/files_depth.json\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.547291Z","iopub.execute_input":"2024-04-02T07:13:10.548084Z","iopub.status.idle":"2024-04-02T07:13:10.567540Z","shell.execute_reply.started":"2024-04-02T07:13:10.548045Z","shell.execute_reply":"2024-04-02T07:13:10.566722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputation_plan = read_json(\"/kaggle/input/homecredit-pipeline-imputationplan/imputation_plan.json\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.569298Z","iopub.execute_input":"2024-04-02T07:13:10.570082Z","iopub.status.idle":"2024-04-02T07:13:10.604853Z","shell.execute_reply.started":"2024-04-02T07:13:10.570042Z","shell.execute_reply":"2024-04-02T07:13:10.603740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subcategories = list(files_depth.keys())\nsubcategories","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.606260Z","iopub.execute_input":"2024-04-02T07:13:10.606611Z","iopub.status.idle":"2024-04-02T07:13:10.614030Z","shell.execute_reply.started":"2024-04-02T07:13:10.606583Z","shell.execute_reply":"2024-04-02T07:13:10.612858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base = pd.read_parquet(train_dir + \"train_base.parquet\")\nn_all = train_base.shape[0]\nn_train = int(n_all*0.9)\nn_val = n_all - n_train","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:10.616820Z","iopub.execute_input":"2024-04-02T07:13:10.617356Z","iopub.status.idle":"2024-04-02T07:13:11.005964Z","shell.execute_reply.started":"2024-04-02T07:13:10.617327Z","shell.execute_reply":"2024-04-02T07:13:11.004901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_base = train_base.iloc[n_train:,:]\ntrain_base = train_base.iloc[:n_train,:]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:11.006882Z","iopub.execute_input":"2024-04-02T07:13:11.007180Z","iopub.status.idle":"2024-04-02T07:13:11.012253Z","shell.execute_reply.started":"2024-04-02T07:13:11.007154Z","shell.execute_reply":"2024-04-02T07:13:11.011409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_dir = \"/kaggle/input/homecredit-pipeline-execute/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:11.013443Z","iopub.execute_input":"2024-04-02T07:13:11.013961Z","iopub.status.idle":"2024-04-02T07:13:11.021415Z","shell.execute_reply.started":"2024-04-02T07:13:11.013932Z","shell.execute_reply":"2024-04-02T07:13:11.020499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_file = {}\nfor ID in range(len(subcategories)):\n    \n    feature_df = pd.read_csv(feature_dir + subcategories[ID] + \".csv\")\n\n    if feature_df.shape[1] <= 1:\n        print(subcategories[ID], \"skipped\")\n        continue\n\n    train_df = train_base.merge(feature_df, on = \"case_id\", how = \"left\")\n    val_df = val_base.merge(feature_df, on = \"case_id\", how = \"left\")\n\n    drop_list = [\"case_id\", \"date_decision\", \"MONTH\", \"WEEK_NUM\"]\n    train_df.drop(drop_list, axis = 1, inplace = True)\n    val_df.drop(drop_list, axis = 1, inplace = True)\n\n    cnames = list(train_df.columns)\n\n    imputation_execute()\n\n    train_ml = createMLData(train_df)\n\n    train_auc, val_auc, imp_df = ML_execute()\n    print(subcategories[ID], \"train auc = \", np.round(train_auc, 4), \" val auc = \", np.round(val_auc, 4))\n\n    accuracy_file[subcategories[ID]] = [train_auc, val_auc]\n\n    fname_imp = subcategories[ID] + \"_importance.csv\"\n    imp_df.to_csv(fname_imp, index = False)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:13:11.022872Z","iopub.execute_input":"2024-04-02T07:13:11.023413Z","iopub.status.idle":"2024-04-02T07:15:28.487710Z","shell.execute_reply.started":"2024-04-02T07:13:11.023384Z","shell.execute_reply":"2024-04-02T07:15:28.486553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#write_json(accuracy_file, \"accuracies\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:15:28.489224Z","iopub.execute_input":"2024-04-02T07:15:28.489573Z","iopub.status.idle":"2024-04-02T07:15:28.494705Z","shell.execute_reply.started":"2024-04-02T07:15:28.489543Z","shell.execute_reply":"2024-04-02T07:15:28.493292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_list = []\ntrain_acc_list = []\nval_acc_list = []\nfor key in accuracy_file.keys():\n    file_list.append(key)\n    train_auc, val_auc = accuracy_file[key]\n    train_acc_list.append(train_auc)\n    val_acc_list.append(val_auc)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:15:28.496261Z","iopub.execute_input":"2024-04-02T07:15:28.496561Z","iopub.status.idle":"2024-04-02T07:15:28.507101Z","shell.execute_reply.started":"2024-04-02T07:15:28.496536Z","shell.execute_reply":"2024-04-02T07:15:28.506215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_df = pd.DataFrame({\"file\": file_list, \"train_auc\": train_acc_list, \"val_auc\": val_acc_list})\naccuracy_df = accuracy_df.sort_values(\"val_auc\", ascending = False)\naccuracy_df.to_csv(\"accuracies_df.csv\", index = False)\naccuracy_df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:15:28.508580Z","iopub.execute_input":"2024-04-02T07:15:28.509568Z","iopub.status.idle":"2024-04-02T07:15:28.532531Z","shell.execute_reply.started":"2024-04-02T07:15:28.509528Z","shell.execute_reply":"2024-04-02T07:15:28.531498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}