{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nfrom sklearn.linear_model import LinearRegression\n\n\ndef calculate_stability_metric(ginis, weeks=None):\n    if weeks is None:\n        weeks = np.arange(1, len(ginis) + 1) + 100\n    \n    ginis = np.array(ginis)\n\n    reg = LinearRegression().fit(\n        weeks.reshape((len(weeks), 1)), ginis\n    )\n\n    a = reg.coef_[0]\n    b = reg.intercept_\n    residuals = ginis - (a * weeks + b)\n    return np.mean(ginis) + 88 * min(0, a) - 0.5 * np.std(residuals)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-18T18:56:34.167335Z","iopub.execute_input":"2024-05-18T18:56:34.167760Z","iopub.status.idle":"2024-05-18T18:56:34.969833Z","shell.execute_reply.started":"2024-05-18T18:56:34.167728Z","shell.execute_reply":"2024-05-18T18:56:34.968634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ginis in [\n    [0.80, 0.80, 0.80, 0.80],\n    [1.0, 0.99, 0.99, 0.99],\n]:\n    metric = calculate_stability_metric(ginis)\n    print(ginis, metric)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T18:56:35.183111Z","iopub.execute_input":"2024-05-18T18:56:35.183490Z","iopub.status.idle":"2024-05-18T18:56:35.207036Z","shell.execute_reply.started":"2024-05-18T18:56:35.183462Z","shell.execute_reply":"2024-05-18T18:56:35.205612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\n\nvalues2 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues2[:2] = 2 * 0.5 - 1\n\nvalues3 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues3[:3] = 2 * 0.62 - 1\n\nvalues4 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues4[:4] = 2 * 0.68 - 1\n\nvalues6 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues6[:6] = 2 * 0.73 - 1\n\nvalues10 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues10[:10] = 2 * 0.775 - 1\n\nvalues20 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues20[:20] = 2 * 0.804 - 1\n\nvalues40 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues40[:40] = 2 * 0.814 - 1","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:00:17.269797Z","iopub.execute_input":"2024-05-18T03:00:17.270580Z","iopub.status.idle":"2024-05-18T03:00:17.279928Z","shell.execute_reply.started":"2024-05-18T03:00:17.270539Z","shell.execute_reply":"2024-05-18T03:00:17.278576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ginis in [\n    values,\n    values2,\n    values3,\n    values4,\n    values6,\n    values10,\n    values20,\n    values40\n]:\n    metric = calculate_stability_metric(ginis)\n    print( metric)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:00:17.570700Z","iopub.execute_input":"2024-05-18T03:00:17.571157Z","iopub.status.idle":"2024-05-18T03:00:17.587381Z","shell.execute_reply.started":"2024-05-18T03:00:17.571117Z","shell.execute_reply":"2024-05-18T03:00:17.586332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\n\nvalues2 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues2[:2] = 2 * 0.5 - 1\n\n# values3 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\n# values3[:3] = 2 * 0.62 - 1\n\nvalues4 = np.linspace(2 * 0.85 - 1, 2 * 0.81 - 1, 100)\nvalues4[:4] = 2 * 0.60 - 1\n\n\nfor ginis in [\n    values,\n    values2,\n    values4\n]:\n    metric = calculate_stability_metric(ginis)\n    print( metric)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T19:06:52.408628Z","iopub.execute_input":"2024-05-18T19:06:52.409107Z","iopub.status.idle":"2024-05-18T19:06:52.424597Z","shell.execute_reply.started":"2024-05-18T19:06:52.409078Z","shell.execute_reply":"2024-05-18T19:06:52.423171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Predict Train vs Test","metadata":{"execution":{"iopub.status.busy":"2024-05-21T17:43:08.406830Z","iopub.execute_input":"2024-05-21T17:43:08.407280Z","iopub.status.idle":"2024-05-21T17:43:08.432909Z","shell.execute_reply.started":"2024-05-21T17:43:08.407246Z","shell.execute_reply":"2024-05-21T17:43:08.431959Z"}}},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport joblib\nimport warnings\nwarnings.filterwarnings('ignore')\n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\n\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import KNNImputer\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:48:16.776771Z\",\"iopub.execute_input\":\"2024-04-12T04:48:16.777574Z\",\"iopub.status.idle\":\"2024-04-12T04:48:16.819534Z\",\"shell.execute_reply.started\":\"2024-04-12T04:48:16.777537Z\",\"shell.execute_reply\":\"2024-04-12T04:48:16.818531Z\"}}\nclass Pipeline:\n\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        \n        return df\n\n\nclass Aggregator:\n    #Please add or subtract features yourself, be aware that too many features will take up too much space.\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n    \n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return  expr_max\n    \n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return  expr_max\n    \n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return  expr_max \n    \n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] \n        return  expr_max\n    \n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs\n\ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1,2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2    \n    return df\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:48:16.820721Z\",\"iopub.execute_input\":\"2024-04-12T04:48:16.821138Z\",\"iopub.status.idle\":\"2024-04-12T04:48:16.831678Z\",\"shell.execute_reply.started\":\"2024-04-12T04:48:16.821102Z\",\"shell.execute_reply\":\"2024-04-12T04:48:16.830812Z\"}}\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:48:16.834374Z\",\"iopub.execute_input\":\"2024-04-12T04:48:16.8347Z\",\"iopub.status.idle\":\"2024-04-12T04:50:32.556024Z\",\"shell.execute_reply.started\":\"2024-04-12T04:48:16.834675Z\",\"shell.execute_reply\":\"2024-04-12T04:50:32.554914Z\"}}\n\ndata_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:50:32.557478Z\",\"iopub.execute_input\":\"2024-04-12T04:50:32.557855Z\",\"iopub.status.idle\":\"2024-04-12T04:52:13.690579Z\",\"shell.execute_reply.started\":\"2024-04-12T04:50:32.557827Z\",\"shell.execute_reply\":\"2024-04-12T04:52:13.689628Z\"}}\n\ndf_train = feature_eng(**data_store)\ndel data_store\ngc.collect()\ndf_train = df_train.pipe(Pipeline.filter_cols)\ndf_train, cat_cols = to_pandas(df_train)\ndf_train = reduce_mem_usage(df_train)\nnums=df_train.select_dtypes(exclude='category').columns\nfrom itertools import combinations, permutations\nnans_df = df_train[nums].isna()\nnans_groups={}\nfor col in nums:\n    cur_group = nans_df[col].sum()\n    try:\n        nans_groups[cur_group].append(col)\n    except:\n        nans_groups[cur_group]=[col]\ndel nans_df; x=gc.collect()\n\ndef reduce_group(grps):\n    use = []\n    for g in grps:\n        mx = 0; vx = g[0]\n        for gg in g:\n            n = df_train[gg].nunique()\n            if n>mx:\n                mx = n\n                vx = gg\n        use.append(vx)\n    return use\n\ndef group_columns_by_correlation(matrix, threshold=0.8):\n    correlation_matrix = matrix.corr()\n    groups = []\n    remaining_cols = list(matrix.columns)\n    while remaining_cols:\n        col = remaining_cols.pop(0)\n        group = [col]\n        correlated_cols = [col]\n        for c in remaining_cols:\n            if correlation_matrix.loc[col, c] >= threshold:\n                group.append(c)\n                correlated_cols.append(c)\n        groups.append(group)\n        remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n    \n    return groups\n\nuses=[]\nfor k,v in nans_groups.items():\n    if len(v)>1:\n            Vs = nans_groups[k]\n            grps= group_columns_by_correlation(df_train[Vs], threshold=0.8)\n            use=reduce_group(grps)\n            uses=uses+use\n    else:\n        uses=uses+v\ndf_train=df_train[uses]\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:52:13.706076Z\",\"iopub.execute_input\":\"2024-04-12T04:52:13.706934Z\",\"iopub.status.idle\":\"2024-04-12T04:52:13.943422Z\",\"shell.execute_reply.started\":\"2024-04-12T04:52:13.706903Z\",\"shell.execute_reply\":\"2024-04-12T04:52:13.94266Z\"}}\ndata_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:52:13.944484Z\",\"iopub.execute_input\":\"2024-04-12T04:52:13.944768Z\",\"iopub.status.idle\":\"2024-04-12T04:52:14.469687Z\",\"shell.execute_reply.started\":\"2024-04-12T04:52:13.944738Z\",\"shell.execute_reply\":\"2024-04-12T04:52:14.468749Z\"}}\ndf_test = feature_eng(**data_store)\ndel data_store\ngc.collect()\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\ndf_test, cat_cols = to_pandas(df_test)\ndf_test = reduce_mem_usage(df_test)\ngc.collect()\n\n# %% [code]\ndf_train['target']=0\ndf_test['target']=1\n\n# %% [code]\n# df_train=pd.concat([df_train,df_test])\ndf_train=reduce_mem_usage(df_train)\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-04-12T04:52:14.471468Z\",\"iopub.execute_input\":\"2024-04-12T04:52:14.471866Z\",\"iopub.status.idle\":\"2024-04-12T04:52:14.61968Z\",\"shell.execute_reply.started\":\"2024-04-12T04:52:14.471831Z\",\"shell.execute_reply\":\"2024-04-12T04:52:14.618841Z\"}}\ny = df_train[\"target\"]\n#df_train= df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\n\n# %% [code]\njoblib.dump((df_train,y,df_test),'data.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:02:34.384011Z","iopub.execute_input":"2024-05-21T18:02:34.384517Z","iopub.status.idle":"2024-05-21T18:05:02.703509Z","shell.execute_reply.started":"2024-05-21T18:02:34.384488Z","shell.execute_reply":"2024-05-21T18:05:02.702522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:06:09.728186Z","iopub.execute_input":"2024-05-21T18:06:09.728553Z","iopub.status.idle":"2024-05-21T18:06:09.734565Z","shell.execute_reply.started":"2024-05-21T18:06:09.728524Z","shell.execute_reply":"2024-05-21T18:06:09.733560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_base = pd.read_parquet(TRAIN_DIR / \"train_base.parquet\")\ndf_base.shape\n\ndf_train = df_train.merge(\n    df_base[['case_id', 'date_decision']],\n    on='case_id',\n    how='left'\n)\n\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:06:22.008243Z","iopub.execute_input":"2024-05-21T18:06:22.008584Z","iopub.status.idle":"2024-05-21T18:06:23.658417Z","shell.execute_reply.started":"2024-05-21T18:06:22.008558Z","shell.execute_reply":"2024-05-21T18:06:23.656408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.to_pickle('df_train_model_train_test.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:06:50.759264Z","iopub.execute_input":"2024-05-21T18:06:50.759982Z","iopub.status.idle":"2024-05-21T18:06:51.728481Z","shell.execute_reply.started":"2024-05-21T18:06:50.759942Z","shell.execute_reply":"2024-05-21T18:06:51.727474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['date_decision'].min(), df_train['date_decision'].max()","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:07:24.843625Z","iopub.execute_input":"2024-05-21T18:07:24.843987Z","iopub.status.idle":"2024-05-21T18:07:25.241781Z","shell.execute_reply.started":"2024-05-21T18:07:24.843957Z","shell.execute_reply":"2024-05-21T18:07:25.240887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all = df_train","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:09:44.963395Z","iopub.execute_input":"2024-05-21T18:09:44.963771Z","iopub.status.idle":"2024-05-21T18:09:44.968252Z","shell.execute_reply.started":"2024-05-21T18:09:44.963740Z","shell.execute_reply":"2024-05-21T18:09:44.967208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_week = 60\ndf_all.loc[df_all['WEEK_NUM'] < threshold_week, 'target'] = 0\ndf_all.loc[df_all['WEEK_NUM'] >= threshold_week, 'target'] = 1","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:12:20.832978Z","iopub.execute_input":"2024-05-21T18:12:20.833879Z","iopub.status.idle":"2024-05-21T18:12:20.844750Z","shell.execute_reply.started":"2024-05-21T18:12:20.833847Z","shell.execute_reply":"2024-05-21T18:12:20.843717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all.shape#, df_train.shape, df_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:13:19.970538Z","iopub.execute_input":"2024-05-21T18:13:19.971175Z","iopub.status.idle":"2024-05-21T18:13:19.976975Z","shell.execute_reply.started":"2024-05-21T18:13:19.971140Z","shell.execute_reply":"2024-05-21T18:13:19.976154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = lgb.LGBMClassifier()\nmodel.fit(df_all.drop(columns=['case_id', 'WEEK_NUM', 'date_decision', 'target']), df_all['target'])","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:13:51.918498Z","iopub.execute_input":"2024-05-21T18:13:51.918847Z","iopub.status.idle":"2024-05-21T18:15:49.590583Z","shell.execute_reply.started":"2024-05-21T18:13:51.918818Z","shell.execute_reply":"2024-05-21T18:15:49.589640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted(zip(np.array(model.feature_importances_) / np.sum(model.feature_importances_), model.feature_name_), reverse=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:17:01.655279Z","iopub.execute_input":"2024-05-21T18:17:01.655909Z","iopub.status.idle":"2024-05-21T18:17:01.678125Z","shell.execute_reply.started":"2024-05-21T18:17:01.655879Z","shell.execute_reply":"2024-05-21T18:17:01.677048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\n\nmodel_cat = CatBoostClassifier(\n    task_type='GPU'\n)\nmodel_cat.fit(df_all.drop(columns=['case_id', 'WEEK_NUM', 'date_decision', 'target']), df_all['target'])","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:40:26.806358Z","iopub.execute_input":"2024-05-21T18:40:26.806738Z","iopub.status.idle":"2024-05-21T18:43:53.383146Z","shell.execute_reply.started":"2024-05-21T18:40:26.806707Z","shell.execute_reply":"2024-05-21T18:43:53.382167Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted(zip(np.array(model_cat.feature_importances_) / np.sum(model_cat.feature_importances_), model_cat.feature_names_), reverse=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:44:14.057868Z","iopub.execute_input":"2024-05-21T18:44:14.058763Z","iopub.status.idle":"2024-05-21T18:44:14.079537Z","shell.execute_reply.started":"2024-05-21T18:44:14.058726Z","shell.execute_reply":"2024-05-21T18:44:14.078549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_all[df_all['target'] == 1]\ndf_test['y_pred'] = model.predict_proba(df_test.drop(columns=['case_id', 'WEEK_NUM', 'date_decision', 'target']))[:, 1]\nprint(df_test['WEEK_NUM'].min(), df_test['WEEK_NUM'].max())","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:18:26.423288Z","iopub.execute_input":"2024-05-21T18:18:26.424230Z","iopub.status.idle":"2024-05-21T18:18:28.121583Z","shell.execute_reply.started":"2024-05-21T18:18:26.424185Z","shell.execute_reply":"2024-05-21T18:18:28.120545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = np.percentile(df_test['y_pred'], 2.5)\nthreshold","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:27:13.416260Z","iopub.execute_input":"2024-05-21T18:27:13.416622Z","iopub.status.idle":"2024-05-21T18:27:13.426443Z","shell.execute_reply.started":"2024-05-21T18:27:13.416594Z","shell.execute_reply":"2024-05-21T18:27:13.425542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg = df_test.groupby('WEEK_NUM', as_index=False)['y_pred'].mean()\n\nimport plotly.express as px\n\nfig = px.line(df_agg, x=\"WEEK_NUM\", y=\"y_pred\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:24:03.560878Z","iopub.execute_input":"2024-05-21T18:24:03.561256Z","iopub.status.idle":"2024-05-21T18:24:05.540506Z","shell.execute_reply.started":"2024-05-21T18:24:03.561227Z","shell.execute_reply":"2024-05-21T18:24:05.539590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold","metadata":{"execution":{"iopub.status.busy":"2024-05-21T18:26:59.248515Z","iopub.execute_input":"2024-05-21T18:26:59.248882Z","iopub.status.idle":"2024-05-21T18:26:59.255519Z","shell.execute_reply.started":"2024-05-21T18:26:59.248853Z","shell.execute_reply":"2024-05-21T18:26:59.254474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.percentile(df_test['y_pred'], 2.5))\nprint(np.percentile(df_test['y_pred'], 3.0))\nprint(np.percentile(df_test['y_pred'], 3.5))\nprint(np.percentile(df_test['y_pred'], 4.0))\nprint(np.percentile(df_test['y_pred'], 4.5))","metadata":{"execution":{"iopub.status.busy":"2024-05-21T19:15:01.036684Z","iopub.execute_input":"2024-05-21T19:15:01.037419Z","iopub.status.idle":"2024-05-21T19:15:01.056533Z","shell.execute_reply.started":"2024-05-21T19:15:01.037389Z","shell.execute_reply":"2024-05-21T19:15:01.055584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[df_test['y_pred'] < np.percentile(df_test['y_pred'], 2.5)]['WEEK_NUM'].value_counts(dropna=False, normalize=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-21T19:12:25.285083Z","iopub.execute_input":"2024-05-21T19:12:25.285484Z","iopub.status.idle":"2024-05-21T19:12:25.306400Z","shell.execute_reply.started":"2024-05-21T19:12:25.285447Z","shell.execute_reply":"2024-05-21T19:12:25.305485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[df_test['y_pred'] < np.percentile(df_test['y_pred'], 4.)]['WEEK_NUM'].value_counts(dropna=False, normalize=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-21T19:15:24.279918Z","iopub.execute_input":"2024-05-21T19:15:24.280310Z","iopub.status.idle":"2024-05-21T19:15:24.305228Z","shell.execute_reply.started":"2024-05-21T19:15:24.280282Z","shell.execute_reply":"2024-05-21T19:15:24.304067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}