{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":170267939,"sourceType":"kernelVersion"},{"sourceId":170271816,"sourceType":"kernelVersion"},{"sourceId":170312534,"sourceType":"kernelVersion"},{"sourceId":170314351,"sourceType":"kernelVersion"}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Purpose\n\nThis notebook use data from other notebooks:\n\n- [HomeCredit-Pipeline_ImputationPlan](https://www.kaggle.com/code/hidetaketakahashi/homecredit-pipeline-imputationplan)\n- [HomeCredit-Pipeline_Execute](https://www.kaggle.com/code/hidetaketakahashi/homecredit-pipeline-execute)\n- [HomeCredit-ScreeningByML](https://www.kaggle.com/code/hidetaketakahashi/homecredit-screeningbyml)\n\n","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport re\nimport json\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn import metrics\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-07T05:43:13.813737Z","iopub.execute_input":"2024-04-07T05:43:13.814517Z","iopub.status.idle":"2024-04-07T05:43:17.019145Z","shell.execute_reply.started":"2024-04-07T05:43:13.814488Z","shell.execute_reply":"2024-04-07T05:43:17.017976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nfrom sklearn.metrics import roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.021203Z","iopub.execute_input":"2024-04-07T05:43:17.021820Z","iopub.status.idle":"2024-04-07T05:43:17.266021Z","shell.execute_reply.started":"2024-04-07T05:43:17.021782Z","shell.execute_reply":"2024-04-07T05:43:17.264800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_json(fname):\n    with open(fname, 'r',  encoding='utf-8') as f:\n        data = json.load(f)\n        \n    return data\n\ndef write_json(data_json, fname):\n    fname_json = fname + \".json\"\n    with open(fname_json, 'w',  encoding='utf-8') as f:\n        json.dump(data_json, f)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.267356Z","iopub.execute_input":"2024-04-07T05:43:17.267716Z","iopub.status.idle":"2024-04-07T05:43:17.272890Z","shell.execute_reply.started":"2024-04-07T05:43:17.267686Z","shell.execute_reply":"2024-04-07T05:43:17.272288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createFactorCode(vals_list):\n    factor_code = {}\n    count = 0\n    for val in vals_list:\n        factor_code[val] = count\n        if val == True:\n            factor_code[\"True\"] = count\n        if val == False:\n            factor_code[\"False\"] = count\n        count += 1\n        \n    factor_code[\"OTHERS\"] = -1\n    factor_code[\"Z\"] = -2\n    factor_code[\"NOID\"] = -3\n    \n    return factor_code\n\ndef convertFactor(df, c, factor_code_all):\n    factor_code = factor_code_all[c]\n    return df[c].apply(lambda x: factor_code[x])","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.274580Z","iopub.execute_input":"2024-04-07T05:43:17.274961Z","iopub.status.idle":"2024-04-07T05:43:17.284290Z","shell.execute_reply.started":"2024-04-07T05:43:17.274940Z","shell.execute_reply":"2024-04-07T05:43:17.283476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imputation_execute(file, convert_factor = True):\n    \n    cnames = list(imputation_plan[file].keys())\n    \n    for c in cnames:\n        \n        vals = imputation_plan[file][c]\n        \n        if c == \"target\":\n            continue\n\n        dtype1 = vals[0]\n\n        if dtype1 == 'factor':\n\n            #if False:\n            filter_na = train_df[c].isna()\n            train_df.loc[filter_na, c] = \"NOID\"\n\n            filter_na = val_df[c].isna()\n            val_df.loc[filter_na, c] = \"NOID\"\n            \n            filter_na = test_df[c].isna()\n            test_df.loc[filter_na, c] = \"NOID\"\n            \n            if convert_factor:\n                #train_df[c] = pd.Categorical(train_df[c]).codes\n                #val_df[c] = pd.Categorical(val_df[c]).codes\n                #print(c)\n                train_df[c] = convertFactor(train_df, c, factor_code_all)\n                val_df[c] = convertFactor(val_df, c, factor_code_all)\n                \n                test_df[c] = convertFactor(test_df, c, factor_code_all)\n\n        else:\n            \n            dtype2, imputation_null, imputation_noid = imputation_plan[file][c]\n\n\n            filter_na = train_df[c].isna()\n            train_df.loc[filter_na, c] = imputation_noid\n\n            filter_na = val_df[c].isna()\n            val_df.loc[filter_na, c] = imputation_noid\n            \n            filter_na = test_df[c].isna()\n            test_df.loc[filter_na, c] = imputation_noid        \n            \n            \n    return train_df, val_df, test_df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.285153Z","iopub.execute_input":"2024-04-07T05:43:17.286051Z","iopub.status.idle":"2024-04-07T05:43:17.294932Z","shell.execute_reply.started":"2024-04-07T05:43:17.286024Z","shell.execute_reply":"2024-04-07T05:43:17.294210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createMLData(data):\n    \n    np.random.seed(1)\n    #data for ml\n    filter1 = data[\"target\"] == 1\n    n_pos_all = np.sum(filter1)\n    n_pos = n_pos_all\n    #n_max = 30000\n    #n_pos = min(n_max, n_pos_all)\n\n\n    data_p = data.loc[filter1,:].sample(n_pos)\n    data_n = data.loc[~filter1,:].sample(n_pos)\n    data_ml = pd.concat([data_p, data_n])\n    \n    return data_ml","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.295825Z","iopub.execute_input":"2024-04-07T05:43:17.296422Z","iopub.status.idle":"2024-04-07T05:43:17.312800Z","shell.execute_reply.started":"2024-04-07T05:43:17.296397Z","shell.execute_reply":"2024-04-07T05:43:17.311511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(df, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = df.loc[:, [\"WEEK_NUM\", \"target\", \"yhat\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"yhat\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"yhat\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.314126Z","iopub.execute_input":"2024-04-07T05:43:17.314478Z","iopub.status.idle":"2024-04-07T05:43:17.325456Z","shell.execute_reply.started":"2024-04-07T05:43:17.314446Z","shell.execute_reply":"2024-04-07T05:43:17.324580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ML_execute(data_ml, m_depth, return_model = False):\n    \n    drop_list = [\"case_id\", \"date_decision\", \"MONTH\", \"WEEK_NUM\", \"target\"]\n    \n    clf = RandomForestClassifier(max_depth=m_depth, random_state=0, n_estimators=200)\n    clf.fit(data_ml.drop(drop_list, axis = 1), data_ml[\"target\"])\n    \n    if return_model:\n        return clf\n    yhat_train = clf.predict_proba(data_ml.drop(drop_list, axis = 1))[:,1]\n    yhat_val = clf.predict_proba(val_df.drop(drop_list, axis = 1))[:,1]\n\n    fpr, tpr, thresholds = metrics.roc_curve(data_ml[\"target\"], yhat_train, pos_label=1)\n    train_auc = metrics.auc(fpr, tpr)\n    train_auc\n\n    fpr, tpr, thresholds = metrics.roc_curve(val_df[\"target\"], yhat_val, pos_label=1)\n    val_auc = metrics.auc(fpr, tpr)\n    val_auc\n    \n    \n    drop_list = [\"case_id\", \"date_decision\", \"MONTH\", \"WEEK_NUM\", \"target\"]\n    val_df2 = val_df[[\"WEEK_NUM\", \"target\"]].copy()\n    val_df2[\"yhat\"] = yhat_val\n    val_stability = gini_stability(val_df2)\n\n    imp_df = pd.DataFrame({\"Variable\":list(data_ml.drop(drop_list, axis = 1).columns), \"score\":clf.feature_importances_})\n    imp_df = imp_df.sort_values(\"score\", ascending = False).reset_index().drop([\"index\"], axis = 1)\n    \n    print(\"train auc = \", np.round(train_auc, 4), \" val auc = \", np.round(val_auc, 4), \" val stability = \", np.round(val_stability, 4))\n    \n    return train_auc, val_auc, imp_df, val_stability","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.326543Z","iopub.execute_input":"2024-04-07T05:43:17.327101Z","iopub.status.idle":"2024-04-07T05:43:17.341261Z","shell.execute_reply.started":"2024-04-07T05:43:17.327079Z","shell.execute_reply":"2024-04-07T05:43:17.340587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Code for merge test file","metadata":{}},{"cell_type":"code","source":"def readFiles(files):\n    count = 0\n    for file in files:\n        if count == 0:\n            data = pd.read_parquet(test_dir + file)\n            \n        else:\n            tmp = pd.read_parquet(test_dir + file)\n            data = pd.concat([data, tmp])\n        \n        count += 1\n\n    \n    return data\n\ndef readFilesAggL1(files):\n    count = 0\n    \n    \n    for file in files:\n        if count == 0:\n            data = pd.read_parquet(test_dir + file)\n            filter1 = data[\"num_group1\"] == 0\n            data = data.loc[filter1]\n        \n            \n        else:\n            tmp = pd.read_parquet(test_dir + file)\n            filter1 = tmp[\"num_group1\"] == 0\n            tmp = tmp.loc[filter1]\n            \n            data = pd.concat([data, tmp])\n        \n        count += 1\n    \n    \n    return data\n\ndef readFilesAggL2(files):\n    count = 0\n    \n    for file in files:\n        if count == 0:\n            data = pd.read_parquet(test_dir + file)\n            filter1 = (data[\"num_group1\"] == 0) & (data[\"num_group2\"] == 0)\n            data = data.loc[filter1]\n        \n            \n        else:\n            tmp = pd.read_parquet(test_dir + file)\n            filter1 = (tmp[\"num_group1\"] == 0) & (tmp[\"num_group2\"] == 0)\n            tmp = tmp.loc[filter1]\n            \n            data = pd.concat([data, tmp])\n        \n        count += 1\n\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.342491Z","iopub.execute_input":"2024-04-07T05:43:17.342835Z","iopub.status.idle":"2024-04-07T05:43:17.353211Z","shell.execute_reply.started":"2024-04-07T05:43:17.342811Z","shell.execute_reply":"2024-04-07T05:43:17.352445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def impute_factor(df, c, values):\n    \n    filter_na = df[c].isna()\n    \n    set_values = set(values)\n    \n    filter1 =  df[c].apply(lambda x: x not in set_values)\n    \n    df.loc[filter1, c] = \"OTHERS\"\n    \n    df.loc[filter_na, c] = \"Z\"\n    \n    return df[c]\n\ndef impute_numeric(df, c, value):\n    filter_na = df[c].isna()\n    df.loc[filter_na, c] = value\n    return df[c]    \n\ndef impute_date(df, c, value):\n    \n    filter_na = df[c].isna()\n    \n    df.loc[:,c] = pd.to_datetime(df[c]).astype(int)\n    df.loc[filter_na, c] = int(value)\n    \n    return df[c]    ","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.355384Z","iopub.execute_input":"2024-04-07T05:43:17.356087Z","iopub.status.idle":"2024-04-07T05:43:17.368273Z","shell.execute_reply.started":"2024-04-07T05:43:17.356055Z","shell.execute_reply":"2024-04-07T05:43:17.367546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imputation_execute_test(df, imputation_plan, key):\n\n    selected_cols = list(imputation_plan[key].keys())\n\n    df = df[[\"case_id\"] + selected_cols]\n\n    for c in selected_cols:\n        vals = imputation_plan[key][c]\n        #print(vals[0], c)\n        if vals[0] == \"factor\":\n            df.loc[:,c] = impute_factor(df, c, vals[1])\n        elif vals[0] == \"date\":\n            df.loc[:,c] = impute_date(df, c, vals[1])\n        else:\n            df.loc[:,c] = impute_numeric(df, c, vals[1])\n            \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.369180Z","iopub.execute_input":"2024-04-07T05:43:17.369846Z","iopub.status.idle":"2024-04-07T05:43:17.377913Z","shell.execute_reply.started":"2024-04-07T05:43:17.369821Z","shell.execute_reply":"2024-04-07T05:43:17.376955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.379324Z","iopub.execute_input":"2024-04-07T05:43:17.379713Z","iopub.status.idle":"2024-04-07T05:43:17.402053Z","shell.execute_reply.started":"2024-04-07T05:43:17.379683Z","shell.execute_reply":"2024-04-07T05:43:17.401356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fpath = \"/kaggle/input/homecredit-filesbycategory/train_files_subcategory.json\"\nsubcategory_files_train = read_json(fpath)\n\nfpath = \"/kaggle/input/homecredit-filesbycategory/test_files_subcategory.json\"\nsubcategory_files_test2 = read_json(fpath)\n\nfpath = \"/kaggle/input/homecredit-filesbycategory/files_depth.json\"\nfile_depth = read_json(fpath)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.403060Z","iopub.execute_input":"2024-04-07T05:43:17.403433Z","iopub.status.idle":"2024-04-07T05:43:17.433119Z","shell.execute_reply.started":"2024-04-07T05:43:17.403408Z","shell.execute_reply":"2024-04-07T05:43:17.432084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_df = pd.read_csv(\"/kaggle/input/homecredit-screeningbyml/accuracies_df.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.437081Z","iopub.execute_input":"2024-04-07T05:43:17.437442Z","iopub.status.idle":"2024-04-07T05:43:17.450993Z","shell.execute_reply.started":"2024-04-07T05:43:17.437418Z","shell.execute_reply":"2024-04-07T05:43:17.450303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_dir = \"/kaggle/input/homecredit-pipeline-execute/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.519713Z","iopub.execute_input":"2024-04-07T05:43:17.523142Z","iopub.status.idle":"2024-04-07T05:43:17.527174Z","shell.execute_reply.started":"2024-04-07T05:43:17.523092Z","shell.execute_reply":"2024-04-07T05:43:17.526538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputation_plan = read_json(\"/kaggle/input/homecredit-pipeline-imputationplan/imputation_plan.json\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.698172Z","iopub.execute_input":"2024-04-07T05:43:17.699504Z","iopub.status.idle":"2024-04-07T05:43:17.708588Z","shell.execute_reply.started":"2024-04-07T05:43:17.699470Z","shell.execute_reply":"2024-04-07T05:43:17.707270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"factor_code_all = {}\nfor file in imputation_plan.keys():\n\n    for c in imputation_plan[file].keys():\n        vals = imputation_plan[file][c]\n        if vals[0] == \"factor\":\n            factor_code = createFactorCode(vals[1])\n            factor_code_all[c] = factor_code\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:17.903052Z","iopub.execute_input":"2024-04-07T05:43:17.903423Z","iopub.status.idle":"2024-04-07T05:43:17.908714Z","shell.execute_reply.started":"2024-04-07T05:43:17.903396Z","shell.execute_reply":"2024-04-07T05:43:17.907851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## important files","metadata":{}},{"cell_type":"code","source":"files = accuracy_df.query(\"val_auc > 0.55\")[\"file\"].tolist()\nfiles","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:18.535582Z","iopub.execute_input":"2024-04-07T05:43:18.535937Z","iopub.status.idle":"2024-04-07T05:43:18.556628Z","shell.execute_reply.started":"2024-04-07T05:43:18.535913Z","shell.execute_reply":"2024-04-07T05:43:18.555736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subcategory_files_test = {}  \nfor file in files:\n    subcategory_files_test[file] = subcategory_files_test2[file]\n","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:18.722205Z","iopub.execute_input":"2024-04-07T05:43:18.722579Z","iopub.status.idle":"2024-04-07T05:43:18.727364Z","shell.execute_reply.started":"2024-04-07T05:43:18.722554Z","shell.execute_reply":"2024-04-07T05:43:18.726311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subcategory_files_test","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:19.528874Z","iopub.execute_input":"2024-04-07T05:43:19.529219Z","iopub.status.idle":"2024-04-07T05:43:19.536397Z","shell.execute_reply.started":"2024-04-07T05:43:19.529192Z","shell.execute_reply":"2024-04-07T05:43:19.535269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/\"\ntest_dir = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:19.729855Z","iopub.execute_input":"2024-04-07T05:43:19.730170Z","iopub.status.idle":"2024-04-07T05:43:19.736058Z","shell.execute_reply.started":"2024-04-07T05:43:19.730150Z","shell.execute_reply":"2024-04-07T05:43:19.734511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_parquet(test_dir + \"test_base.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:19.992758Z","iopub.execute_input":"2024-04-07T05:43:19.993070Z","iopub.status.idle":"2024-04-07T05:43:20.145752Z","shell.execute_reply.started":"2024-04-07T05:43:19.993048Z","shell.execute_reply":"2024-04-07T05:43:20.145013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key in subcategory_files_test.keys():\n    files1 = subcategory_files_test[key]\n    depth1 = file_depth[key]\n    print(depth1, files1)\n    \n    if depth1 == 0:\n        data_df = readFiles(files1)\n    elif depth1 == 1:\n        data_df = readFilesAggL1(files1)\n    else:\n        data_df = readFilesAggL2(files1)\n        \n    data_df = imputation_execute_test(data_df, imputation_plan, key)\n    test_df = test_df.merge(data_df, on = \"case_id\", how = \"left\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:20.669505Z","iopub.execute_input":"2024-04-07T05:43:20.669882Z","iopub.status.idle":"2024-04-07T05:43:21.075806Z","shell.execute_reply.started":"2024-04-07T05:43:20.669854Z","shell.execute_reply":"2024-04-07T05:43:21.074806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## split data into train and validation","metadata":{}},{"cell_type":"code","source":"train_base = pd.read_parquet(train_dir + \"train_base.parquet\")\nn_all = train_base.shape[0]\nn_train = int(n_all*0.9)\nn_val = n_all - n_train","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:21.630557Z","iopub.execute_input":"2024-04-07T05:43:21.630902Z","iopub.status.idle":"2024-04-07T05:43:21.849440Z","shell.execute_reply.started":"2024-04-07T05:43:21.630877Z","shell.execute_reply":"2024-04-07T05:43:21.848604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df = train_base.iloc[n_train:,:]\ntrain_df = train_base.iloc[:n_train,:]","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:21.999157Z","iopub.execute_input":"2024-04-07T05:43:21.999594Z","iopub.status.idle":"2024-04-07T05:43:22.005773Z","shell.execute_reply.started":"2024-04-07T05:43:21.999564Z","shell.execute_reply":"2024-04-07T05:43:22.004744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax=  plt.subplots()\nax.plot(np.arange(train_df.shape[0]), train_df[\"WEEK_NUM\"])\nax.plot(np.arange(val_df.shape[0]) + train_df.shape[0], val_df[\"WEEK_NUM\"])\nax.legend()\nax.set_ylabel(\"WEEK_NUM\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:22.409124Z","iopub.execute_input":"2024-04-07T05:43:22.409505Z","iopub.status.idle":"2024-04-07T05:43:23.791846Z","shell.execute_reply.started":"2024-04-07T05:43:22.409482Z","shell.execute_reply":"2024-04-07T05:43:23.790795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## merge files","metadata":{}},{"cell_type":"code","source":"for ID in range(len(files)):\n    print(files[ID])\n    feature_df = pd.read_csv(files_dir + files[ID] + \".csv\")\n\n    train_df = train_df.merge(feature_df, on = \"case_id\", how = \"left\")\n    val_df = val_df.merge(feature_df, on = \"case_id\", how = \"left\")","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:43:23.794619Z","iopub.execute_input":"2024-04-07T05:43:23.794941Z","iopub.status.idle":"2024-04-07T05:44:32.404400Z","shell.execute_reply.started":"2024-04-07T05:43:23.794914Z","shell.execute_reply":"2024-04-07T05:44:32.403296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape, val_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:44:32.405836Z","iopub.execute_input":"2024-04-07T05:44:32.407028Z","iopub.status.idle":"2024-04-07T05:44:32.413532Z","shell.execute_reply.started":"2024-04-07T05:44:32.407003Z","shell.execute_reply":"2024-04-07T05:44:32.412563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop_list = [\"case_id\", \"date_decision\", \"MONTH\", \"WEEK_NUM\"]\n#train_df.drop(drop_list, axis = 1, inplace = True)\n#val_df.drop(drop_list, axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:44:32.415360Z","iopub.execute_input":"2024-04-07T05:44:32.415598Z","iopub.status.idle":"2024-04-07T05:44:32.425291Z","shell.execute_reply.started":"2024-04-07T05:44:32.415574Z","shell.execute_reply":"2024-04-07T05:44:32.424477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file in files:\n    print(file)\n    imputation_execute(file)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:44:32.426030Z","iopub.execute_input":"2024-04-07T05:44:32.426313Z","iopub.status.idle":"2024-04-07T05:45:03.972219Z","shell.execute_reply.started":"2024-04-07T05:44:32.426291Z","shell.execute_reply":"2024-04-07T05:45:03.971072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## create train file for ML\n\n`train_ml` has equal numbers of positive and negative labels","metadata":{}},{"cell_type":"code","source":"train_ml = createMLData(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:45:03.973822Z","iopub.execute_input":"2024-04-07T05:45:03.974163Z","iopub.status.idle":"2024-04-07T05:45:05.270454Z","shell.execute_reply.started":"2024-04-07T05:45:03.974134Z","shell.execute_reply":"2024-04-07T05:45:05.268831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ml_all = createMLData(pd.concat([train_df, val_df]))","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:45:05.272676Z","iopub.execute_input":"2024-04-07T05:45:05.273041Z","iopub.status.idle":"2024-04-07T05:45:08.077988Z","shell.execute_reply.started":"2024-04-07T05:45:05.273010Z","shell.execute_reply":"2024-04-07T05:45:08.076500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ml.shape, train_ml_all.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:45:08.079340Z","iopub.execute_input":"2024-04-07T05:45:08.079643Z","iopub.status.idle":"2024-04-07T05:45:08.086118Z","shell.execute_reply.started":"2024-04-07T05:45:08.079617Z","shell.execute_reply":"2024-04-07T05:45:08.085291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Result","metadata":{}},{"cell_type":"code","source":"m_depths = [4,7, 10,13,16,19]\n\n\ntrain_auc_list = []\nval_auc_list = []\nval_stab_list = []\nfor m_depth in m_depths:\n    print(m_depth)\n    train_auc, val_auc, imp_df, val_stab = ML_execute(train_ml, m_depth)\n    train_auc_list.append(train_auc)\n    val_auc_list.append(val_auc)\n    val_stab_list.append(val_stab)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:45:08.088685Z","iopub.execute_input":"2024-04-07T05:45:08.088955Z","iopub.status.idle":"2024-04-07T05:51:49.479647Z","shell.execute_reply.started":"2024-04-07T05:45:08.088926Z","shell.execute_reply":"2024-04-07T05:51:49.476750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n","metadata":{}},{"cell_type":"markdown","source":"It achieved validation auc > 0.7","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (6, 5))\nax.plot(m_depths, train_auc_list, label = \"train\", marker = \"x\")\nax.plot(m_depths, val_auc_list, label = \"val\", marker = \"x\")\nax.set_xlabel(\"max depth of tree\")\nax.set_ylabel(\"AUC\")\nax.axhline(y = 0.7, ls = \"-.\", color = \"green\", alpha = 0.7, label = \"threshold\")\nax.legend()\nax.grid()\nax.set_title(\"AUC\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:51:49.483498Z","iopub.execute_input":"2024-04-07T05:51:49.484149Z","iopub.status.idle":"2024-04-07T05:51:49.806777Z","shell.execute_reply.started":"2024-04-07T05:51:49.484093Z","shell.execute_reply":"2024-04-07T05:51:49.805110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (6, 4))\nax.plot(m_depths, val_stab_list, label = \"val\", marker = \"x\")\nax.set_xlabel(\"max depth of tree\")\nax.set_ylabel(\"Score\")\n#ax.axhline(y = 0.7, ls = \"-.\", color = \"green\", alpha = 0.7, label = \"threshold\")\nax.legend()\nax.grid()\nax.set_title(\"Validation Stability Score\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T05:51:49.808790Z","iopub.execute_input":"2024-04-07T05:51:49.809255Z","iopub.status.idle":"2024-04-07T05:51:50.083739Z","shell.execute_reply.started":"2024-04-07T05:51:49.809196Z","shell.execute_reply":"2024-04-07T05:51:50.080348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m_depth = 16\nmodel_rf = ML_execute(train_ml_all, m_depth, True)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T02:53:26.995273Z","iopub.execute_input":"2024-04-05T02:53:26.995810Z","iopub.status.idle":"2024-04-05T02:55:04.793542Z","shell.execute_reply.started":"2024-04-05T02:53:26.995771Z","shell.execute_reply":"2024-04-05T02:55:04.792177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_list_test = [\"case_id\", \"date_decision\", \"MONTH\", \"WEEK_NUM\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T02:55:04.798876Z","iopub.execute_input":"2024-04-05T02:55:04.799328Z","iopub.status.idle":"2024-04-05T02:55:04.805493Z","shell.execute_reply.started":"2024-04-05T02:55:04.799295Z","shell.execute_reply":"2024-04-05T02:55:04.803992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yhat_test = model_rf.predict_proba(test_df.drop(drop_list_test, axis = 1))[:,1]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T02:55:04.807656Z","iopub.execute_input":"2024-04-05T02:55:04.808562Z","iopub.status.idle":"2024-04-05T02:55:04.838048Z","shell.execute_reply.started":"2024-04-05T02:55:04.808514Z","shell.execute_reply":"2024-04-05T02:55:04.837062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"score\"] = yhat_test","metadata":{"execution":{"iopub.status.busy":"2024-04-05T02:55:04.840662Z","iopub.execute_input":"2024-04-05T02:55:04.841462Z","iopub.status.idle":"2024-04-05T02:55:04.848290Z","shell.execute_reply.started":"2024-04-05T02:55:04.841418Z","shell.execute_reply":"2024-04-05T02:55:04.847102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[[\"case_id\", \"score\"]].to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T02:55:04.849714Z","iopub.execute_input":"2024-04-05T02:55:04.850057Z","iopub.status.idle":"2024-04-05T02:55:04.863111Z","shell.execute_reply.started":"2024-04-05T02:55:04.850023Z","shell.execute_reply":"2024-04-05T02:55:04.861817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}