{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\npd.options.display.max_columns=1000\npd.options.display.max_rows=1000\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_palette(\"Pastel1\")\n\nimport warnings as wr\nwr.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-12T09:53:49.039948Z","iopub.execute_input":"2024-04-12T09:53:49.040212Z","iopub.status.idle":"2024-04-12T09:53:51.009384Z","shell.execute_reply.started":"2024-04-12T09:53:49.040187Z","shell.execute_reply":"2024-04-12T09:53:51.008537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Loading Data","metadata":{}},{"cell_type":"markdown","source":"* The codes for reading and combining data belong to this --> https://www.kaggle.com/code/giraffe85cm/simplest-baseline-using-only-depth-0-data/notebook notebook.","metadata":{}},{"cell_type":"code","source":"directory = \"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:53:51.014333Z","iopub.execute_input":"2024-04-12T09:53:51.014647Z","iopub.status.idle":"2024-04-12T09:53:51.019011Z","shell.execute_reply.started":"2024-04-12T09:53:51.014614Z","shell.execute_reply":"2024-04-12T09:53:51.018031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_column_datatypes(df):\n    \"\"\"\n    Function to set the data type based on each column name\n    :param df: Dataframe\n    :return: Dataframe with the data type set\n    \"\"\"\n    for column in df.columns:\n        if column.endswith('P') or column.endswith('A'):\n            df[column] = df[column].astype(float)\n        elif column.endswith('M'):\n            df[column] = df[column].astype(object)\n        elif column.endswith('D'):\n            df[column] = pd.to_datetime(df[column])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:53:51.021055Z","iopub.execute_input":"2024-04-12T09:53:51.021313Z","iopub.status.idle":"2024-04-12T09:53:51.030607Z","shell.execute_reply.started":"2024-04-12T09:53:51.021290Z","shell.execute_reply":"2024-04-12T09:53:51.029703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# base.csv\ntrain_base_df = pd.read_csv(directory + \"train/train_base.csv\")\ntest_base_df = pd.read_csv(directory + \"test/test_base.csv\")\n\n# static(depth=0)\n## train_static is divided into two parts, 0 and 1. read and concat one by one\ntrain_static0 = pd.read_csv(directory + \"train/train_static_0_0.csv\")\ntrain_static1 = pd.read_csv(directory + \"train/train_static_0_1.csv\")\ntrain_static = pd.concat([train_static0, train_static1], ignore_index=True)\ntrain_static = set_column_datatypes(train_static)\ndel train_static0, train_static1\n## test_static is divided into three parts: 0, 1, and 2. Processing is the same as train.\ntest_static0 = pd.read_csv(directory + \"test/test_static_0_0.csv\")\ntest_static1 = pd.read_csv(directory + \"test/test_static_0_1.csv\")\ntest_static2 = pd.read_csv(directory + \"test/test_static_0_2.csv\")\ntest_static = pd.concat([test_static0, test_static1, test_static2], ignore_index=True)\ntest_static = set_column_datatypes(test_static)\ndel test_static0, test_static1, test_static2\n\n# static_cb\ntrain_static_cb = pd.read_csv(directory + \"train/train_static_cb_0.csv\")\ntrain_static_cb = set_column_datatypes(train_static_cb)\n\ntest_static_cb = pd.read_csv(directory + \"test/test_static_cb_0.csv\")\ntest_static_cb = set_column_datatypes(test_static_cb)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:53:51.031754Z","iopub.execute_input":"2024-04-12T09:53:51.032024Z","iopub.status.idle":"2024-04-12T09:54:52.030815Z","shell.execute_reply.started":"2024-04-12T09:53:51.032000Z","shell.execute_reply":"2024-04-12T09:54:52.030025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train","metadata":{}},{"cell_type":"code","source":"# For simplicity, select only columns ending in \"A\" or \"P\" (columns of float type)\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cols.append(col)\n# print(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cb_cols.append(col)\n# print(selected_static_cb_cols)\n\n# merge\ntrain_df = pd.merge(train_base_df, train_static[[\"case_id\"]+selected_static_cols], how=\"left\", on=\"case_id\")\ntrain_df = pd.merge(train_df, train_static_cb[[\"case_id\"]+selected_static_cb_cols], how=\"left\", on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:52.031943Z","iopub.execute_input":"2024-04-12T09:54:52.032242Z","iopub.status.idle":"2024-04-12T09:54:54.506222Z","shell.execute_reply.started":"2024-04-12T09:54:52.032218Z","shell.execute_reply":"2024-04-12T09:54:54.505382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test","metadata":{}},{"cell_type":"code","source":"# For simplicity, select only columns ending in \"A\" or \"P\" (columns of float type)\nselected_static_cols = []\nfor col in test_static.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cols.append(col)\n# print(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in test_static_cb.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cb_cols.append(col)\n# print(selected_static_cb_cols)\n\n# merge\ntest_df = pd.merge(test_base_df, test_static[[\"case_id\"]+selected_static_cols], how=\"left\", on=\"case_id\")\ntest_df = pd.merge(test_df, test_static_cb[[\"case_id\"]+selected_static_cb_cols], how=\"left\", on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:54.507632Z","iopub.execute_input":"2024-04-12T09:54:54.508003Z","iopub.status.idle":"2024-04-12T09:54:54.608608Z","shell.execute_reply.started":"2024-04-12T09:54:54.507970Z","shell.execute_reply":"2024-04-12T09:54:54.607776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Checking Data","metadata":{}},{"cell_type":"code","source":"test_Id = test_df[\"case_id\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:54.609611Z","iopub.execute_input":"2024-04-12T09:54:54.609870Z","iopub.status.idle":"2024-04-12T09:54:54.621919Z","shell.execute_reply.started":"2024-04-12T09:54:54.609847Z","shell.execute_reply":"2024-04-12T09:54:54.621146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:54.622964Z","iopub.execute_input":"2024-04-12T09:54:54.623294Z","iopub.status.idle":"2024-04-12T09:54:54.634004Z","shell.execute_reply.started":"2024-04-12T09:54:54.623263Z","shell.execute_reply":"2024-04-12T09:54:54.633106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:54.637282Z","iopub.execute_input":"2024-04-12T09:54:54.637735Z","iopub.status.idle":"2024-04-12T09:54:54.644370Z","shell.execute_reply.started":"2024-04-12T09:54:54.637711Z","shell.execute_reply":"2024-04-12T09:54:54.643428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.tail(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:54.645374Z","iopub.execute_input":"2024-04-12T09:54:54.645615Z","iopub.status.idle":"2024-04-12T09:54:54.739903Z","shell.execute_reply.started":"2024-04-12T09:54:54.645591Z","shell.execute_reply":"2024-04-12T09:54:54.738983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:54.741138Z","iopub.execute_input":"2024-04-12T09:54:54.741550Z","iopub.status.idle":"2024-04-12T09:54:55.037471Z","shell.execute_reply.started":"2024-04-12T09:54:54.741514Z","shell.execute_reply":"2024-04-12T09:54:55.036551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:55.038920Z","iopub.execute_input":"2024-04-12T09:54:55.039629Z","iopub.status.idle":"2024-04-12T09:54:55.045907Z","shell.execute_reply.started":"2024-04-12T09:54:55.039592Z","shell.execute_reply":"2024-04-12T09:54:55.044855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Variable Analysis","metadata":{}},{"cell_type":"markdown","source":"## 3.1 Target Variable","metadata":{}},{"cell_type":"code","source":"label_counts = train_df[\"target\"].value_counts()\nlabel_counts","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:55.047698Z","iopub.execute_input":"2024-04-12T09:54:55.048081Z","iopub.status.idle":"2024-04-12T09:54:55.070595Z","shell.execute_reply.started":"2024-04-12T09:54:55.048047Z","shell.execute_reply":"2024-04-12T09:54:55.069607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* unbalanced data","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.pie(label_counts, labels=label_counts.index, autopct='%1.1f%%', startangle=90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:55.071756Z","iopub.execute_input":"2024-04-12T09:54:55.072140Z","iopub.status.idle":"2024-04-12T09:54:55.198545Z","shell.execute_reply.started":"2024-04-12T09:54:55.072114Z","shell.execute_reply":"2024-04-12T09:54:55.197389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.2 Numerical Values\n\n* 0   case_id                   \n* 1 date_decision          \n* 2   MONTH                 \n* 3   WEEK_NUM            \n* 4   target                          \n* 5   actualdpdtolerance_344P        - DPD of client with tolerance.  \n* 6   amtinstpaidbefduel24m_4187115A   - Number of instalments paid before due date in the last 24 months\n* 7   annuity_780A            - Monthly annuity amount.\n* 8   annuitynextmonth_57A    - Next month's amount of annuity.         \n* 9   avgdbddpdlast24m_3658932P      - Average days past or before due of payment during the last 24 months.  \n* 10  avgdbddpdlast3m_4187120P      - Average days past or before due of payment during the last 3 months.  \n* 11  avgdbdtollast24m_4525197P        - Average days of payment before due date within the last 24 months (with tolerance).\n* 12  avgdpdtolclosure24_3658938P      - Average DPD (days past due) with tolerance within the past 24 months from the maximum closure date, ...\n* 13  avginstallast24m_3658937A      - Average instalments paid by the client over the past 24 months.\n* 14  avglnamtstart24m_4525187A      - Average loan amount in the last 24 months.  \n* 15  avgmaxdpdlast9m_3716943P       - Average Days Past Due (DPD) of the client in last 9 months. \n* 16  avgoutstandbalancel6m_4187114A   - Average outstanding balance of applicant for the last 6 months.\n* 17  avgpmtlast12m_4525200A      - Average of payments made by the client in the last 12 months.     \n* 18  credamount_770A        - Loan amount or credit card limit.          \n* 19  currdebt_22A                 - Current debt amount of the client.\n* 20  currdebtcredtyperange_828A     - Current amount of debt of the applicant. \n* 21  disbursedcredamount_1113A        - Disbursed credit amount after consolidation.\n* 22  downpmt_116A                    - Amount of downpayment.\n* 23  inittransactionamount_650A      - Initial transaction amount of the credit application. \n* 24  lastapprcredamount_781A          - Credit amount from the client's last application.\n* 25  lastotherinc_902A                - Amount of other income reported by the client in their last application.\n* 26  lastotherlnsexpense_631A         - Monthly expenses on other loans from the last application.\n* 27  lastrejectcredamount_222A        - Credit amount on last rejected application\n* 28  maininc_215A                     - Client's primary income amount.\n* 29  maxannuity_159A                  - Maximum annuity previously obtained by client.\n* 30  maxannuity_4075009A              - Maximal annuity offered to the client in the current application.\n* 31  maxdbddpdlast1m_3658939P        - Maximum number of days past due in the last month. A negative value indicates the number of days bef...\n* 32  maxdbddpdtollast12m_3658940P    - Maximum number of days past due in last 12 months. A negative value implies days before due date.\n* 33  maxdbddpdtollast6m_4187119P      - Maximum number of days past due in last 6 months. This predictor takes the value as a negative number\n* 34  maxdebt4_972A                    - Maximal principal debt of the client in the history older than 4 months.\n* 35  maxdpdfrom6mto36m_3546853P       - Maximum Days Past Due (DPD) in the period ranging from 6 to 36 months.\n* 36  maxdpdinstlnum_3546846P         - Instalment number of which client was most days past due.\n* 37  maxdpdlast12m_727P               - Maximum days past due in the past 12 months.\n* 38  maxdpdlast24m_143P              - Maximal days past due in the last 24 months.\n* 39  maxdpdlast3m_392P                - Maximum number of days past due in last 3 months.\n* 40  maxdpdlast6m_474P                - Maximum days past due in the last 6 months.\n* 41  maxdpdlast9m_1059P               - Maximum days past due in last 9 months.\n* 42  maxdpdtolerance_374P            - Maximum number of days past due (with tolerance).\n* 43  maxinstallast24m_3658928A       - Maximum instalment in the last 24 months\n* 44  maxlnamtstart6m_4525199A       - Maximum loan amount started in the last 6 months.\n* 45  maxoutstandbalancel12m_4187113A  - Maximum outstanding balance in the last 12 months.\n* 46  maxpmtlast3m_4525190A            -  Maximum payment made by the client in the last 3 months.\n* 47  mindbddpdlast24m_3658935P       - Minimum days past due (or days before due) in last 24 months.\n* 48  mindbdtollast24m_4525191P        - Minimum days before due in last 24 months.\n* 49  posfpd10lastmonth_333P           - Average FPD30 (Share of contracts with first installment past due more than 30 days) from point of s...\n* 50  posfpd30lastmonth_3976960P      - Average FPD30 (Share of contracts with first installment past due more than 30 days) from point of s...\n* 51  posfstqpd30lastmonth_3976962P    - Average FSTPD30 (share of contracts with first, second, or third installment past due more than 30 d...\n* 52  price_1097A                     - Credit price.\n* 53  sumoutstandtotal_3546847A       - Sum of total outstanding amount.\n* 54  sumoutstandtotalest_4493215A     - Sum of total outstanding amount.\n* 55  totaldebt_9A                     - Total amount of debt.\n* 56  totalsettled_863A               - Sum of all payments made by the client.\n* 57  totinstallast1m_4525188A        - Total amount of monthly instalments paid in the previous month.\n* 58  pmtaverage_3A                    - Average of tax deductions.\n* 59  pmtaverage_4527227A              - Average of tax deductions.\n* 60  pmtaverage_4955615A              - Average of tax deductions.\n* 61  pmtssum_45A     - Sum of tax deductions for the client.\n","metadata":{}},{"cell_type":"code","source":"numeric = ['date_decision', 'MONTH', 'WEEK_NUM', 'target',\n       'actualdpdtolerance_344P', 'amtinstpaidbefduel24m_4187115A',\n       'annuity_780A', 'annuitynextmonth_57A', 'avgdbddpdlast24m_3658932P',\n       'avgdbddpdlast3m_4187120P', 'avgdbdtollast24m_4525197P',\n       'avgdpdtolclosure24_3658938P', 'avginstallast24m_3658937A',\n       'avglnamtstart24m_4525187A', 'avgmaxdpdlast9m_3716943P',\n       'avgoutstandbalancel6m_4187114A', 'avgpmtlast12m_4525200A',\n       'credamount_770A', 'currdebt_22A', 'currdebtcredtyperange_828A',\n       'disbursedcredamount_1113A', 'downpmt_116A',\n       'inittransactionamount_650A', 'lastapprcredamount_781A',\n       'lastotherinc_902A', 'lastotherlnsexpense_631A',\n       'lastrejectcredamount_222A', 'maininc_215A', 'maxannuity_159A',\n       'maxannuity_4075009A', 'maxdbddpdlast1m_3658939P',\n       'maxdbddpdtollast12m_3658940P', 'maxdbddpdtollast6m_4187119P',\n       'maxdebt4_972A', 'maxdpdfrom6mto36m_3546853P',\n       'maxdpdinstlnum_3546846P', 'maxdpdlast12m_727P', 'maxdpdlast24m_143P',\n       'maxdpdlast3m_392P', 'maxdpdlast6m_474P', 'maxdpdlast9m_1059P',\n       'maxdpdtolerance_374P', 'maxinstallast24m_3658928A',\n       'maxlnamtstart6m_4525199A', 'maxoutstandbalancel12m_4187113A',\n       'maxpmtlast3m_4525190A', 'mindbddpdlast24m_3658935P',\n       'mindbdtollast24m_4525191P', 'posfpd10lastmonth_333P',\n       'posfpd30lastmonth_3976960P', 'posfstqpd30lastmonth_3976962P',\n       'price_1097A', 'sumoutstandtotal_3546847A',\n       'sumoutstandtotalest_4493215A', 'totaldebt_9A', 'totalsettled_863A',\n       'totinstallast1m_4525188A', 'pmtaverage_3A', 'pmtaverage_4527227A',\n       'pmtaverage_4955615A', 'pmtssum_45A']","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:55.200576Z","iopub.execute_input":"2024-04-12T09:54:55.201410Z","iopub.status.idle":"2024-04-12T09:54:55.213608Z","shell.execute_reply.started":"2024-04-12T09:54:55.201359Z","shell.execute_reply":"2024-04-12T09:54:55.212457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist(variable):\n    plt.figure(figsize = (5,2))\n    plt.hist(train_df[variable], bins = 50)\n    plt.xlabel(variable)\n    plt.ylabel(\"Frequency\")\n    plt.title(\"{} disribution with hist\".format(variable))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:55.215583Z","iopub.execute_input":"2024-04-12T09:54:55.216038Z","iopub.status.idle":"2024-04-12T09:54:55.226190Z","shell.execute_reply.started":"2024-04-12T09:54:55.215997Z","shell.execute_reply":"2024-04-12T09:54:55.224753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for n in numeric:\n    plot_hist(n)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:54:55.228276Z","iopub.execute_input":"2024-04-12T09:54:55.229159Z","iopub.status.idle":"2024-04-12T09:55:21.556423Z","shell.execute_reply.started":"2024-04-12T09:54:55.229114Z","shell.execute_reply":"2024-04-12T09:55:21.555535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:21.557749Z","iopub.execute_input":"2024-04-12T09:55:21.558082Z","iopub.status.idle":"2024-04-12T09:55:21.823446Z","shell.execute_reply.started":"2024-04-12T09:55:21.558048Z","shell.execute_reply":"2024-04-12T09:55:21.822555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Feature Engineering","metadata":{}},{"cell_type":"code","source":"train_df.drop(columns = [\"case_id\", \"MONTH\", \"WEEK_NUM\", \"date_decision\"], inplace = True)\ntest_df.drop(columns = [\"case_id\", \"MONTH\", \"WEEK_NUM\", \"date_decision\"], inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:21.825145Z","iopub.execute_input":"2024-04-12T09:55:21.825532Z","iopub.status.idle":"2024-04-12T09:55:22.087499Z","shell.execute_reply.started":"2024-04-12T09:55:21.825477Z","shell.execute_reply":"2024-04-12T09:55:22.086723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def drop_rows_with_many_nulls(df, threshold=45):\n    \"\"\"\n    Deletes rows with null values in the DataFrame.\n\n    Args:\n        df (pandas.DataFrame): DataFrame to process.\n        threshold (int, optional): The acceptable limit of null values for a row. The default value is 50.\n\n    Returns:\n        pandas.DataFrame: DataFrame where rows with null values are deleted.\n    \"\"\"\n    # Count number of null values in rows\n    null_counts = df.isnull().sum(axis=1)\n    \n    # Delete row if number of null values exceeds threshold\n    filtered_df = df[null_counts <= threshold]\n    \n    return filtered_df","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:22.088758Z","iopub.execute_input":"2024-04-12T09:55:22.089181Z","iopub.status.idle":"2024-04-12T09:55:22.095922Z","shell.execute_reply.started":"2024-04-12T09:55:22.089149Z","shell.execute_reply":"2024-04-12T09:55:22.095030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = drop_rows_with_many_nulls(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:22.096937Z","iopub.execute_input":"2024-04-12T09:55:22.097198Z","iopub.status.idle":"2024-04-12T09:55:22.806407Z","shell.execute_reply.started":"2024-04-12T09:55:22.097161Z","shell.execute_reply":"2024-04-12T09:55:22.805411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:22.807961Z","iopub.execute_input":"2024-04-12T09:55:22.808253Z","iopub.status.idle":"2024-04-12T09:55:22.814066Z","shell.execute_reply.started":"2024-04-12T09:55:22.808228Z","shell.execute_reply":"2024-04-12T09:55:22.813131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Modelling","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nfrom lightgbm import LGBMClassifier","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:22.815171Z","iopub.execute_input":"2024-04-12T09:55:22.815476Z","iopub.status.idle":"2024-04-12T09:55:25.855442Z","shell.execute_reply.started":"2024-04-12T09:55:22.815448Z","shell.execute_reply":"2024-04-12T09:55:25.854658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(['target'], axis=1)\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:25.856553Z","iopub.execute_input":"2024-04-12T09:55:25.856837Z","iopub.status.idle":"2024-04-12T09:55:26.110707Z","shell.execute_reply.started":"2024-04-12T09:55:25.856813Z","shell.execute_reply":"2024-04-12T09:55:26.109945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Print the shapes of the training and testing datasets\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:26.111766Z","iopub.execute_input":"2024-04-12T09:55:26.112021Z","iopub.status.idle":"2024-04-12T09:55:26.948662Z","shell.execute_reply.started":"2024-04-12T09:55:26.111999Z","shell.execute_reply":"2024-04-12T09:55:26.947510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_classifier = LGBMClassifier(force_col_wise=True)\n\nlgbm_classifier.fit(X_train, y_train)\n\ny_pred = lgbm_classifier.predict(X_test)\naccuracy_score(y_test, y_pred) ","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:26.949760Z","iopub.execute_input":"2024-04-12T09:55:26.950035Z","iopub.status.idle":"2024-04-12T09:55:59.405269Z","shell.execute_reply.started":"2024-04-12T09:55:26.950011Z","shell.execute_reply":"2024-04-12T09:55:59.404337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Submission","metadata":{}},{"cell_type":"code","source":"predictions = lgbm_classifier.predict_proba(test_df)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:59.410652Z","iopub.execute_input":"2024-04-12T09:55:59.410962Z","iopub.status.idle":"2024-04-12T09:55:59.418211Z","shell.execute_reply.started":"2024-04-12T09:55:59.410937Z","shell.execute_reply":"2024-04-12T09:55:59.417116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'case_id': test_Id,\n    'score': predictions\n})","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:59.419926Z","iopub.execute_input":"2024-04-12T09:55:59.420209Z","iopub.status.idle":"2024-04-12T09:55:59.429176Z","shell.execute_reply.started":"2024-04-12T09:55:59.420185Z","shell.execute_reply":"2024-04-12T09:55:59.428262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T09:55:59.430207Z","iopub.execute_input":"2024-04-12T09:55:59.431101Z","iopub.status.idle":"2024-04-12T09:55:59.446479Z","shell.execute_reply.started":"2024-04-12T09:55:59.431047Z","shell.execute_reply":"2024-04-12T09:55:59.445425Z"},"trusted":true},"execution_count":null,"outputs":[]}]}